feat(application): integrate measured-load ingestion training and planner source
This commit is contained in:
@@ -0,0 +1,240 @@
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
from scipy.optimize import Bounds, LinearConstraint, milp
|
||||
from scipy.sparse import lil_matrix
|
||||
|
||||
DT_H = 5.0 / 60.0
|
||||
|
||||
|
||||
def train_artifact(kind):
|
||||
return {"trained": True, "type": "battery_48h_cost_milp_v3", "source": kind}
|
||||
|
||||
|
||||
def _cfg_float(config, key, default):
|
||||
try:
|
||||
value = config.get(key, default)
|
||||
return float(default if value is None or value == "" else value)
|
||||
except Exception:
|
||||
return float(default)
|
||||
|
||||
|
||||
def _cfg_bool(config, key, default=False):
|
||||
value = config.get(key, default)
|
||||
if value is None or value == "":
|
||||
return bool(default)
|
||||
if isinstance(value, bool):
|
||||
return value
|
||||
return str(value).strip().lower() in {"1", "true", "yes", "ja", "on"}
|
||||
|
||||
|
||||
def _use_dynamic(config, key):
|
||||
value = str(config.get(key, "") or "").strip().lower()
|
||||
return any(token in value for token in ("dynam", "marktpreis", "referenzmarktpreis", "market"))
|
||||
|
||||
|
||||
def _price(data_obj, timestamp, column, fallback, dynamic_enabled):
|
||||
if not dynamic_enabled:
|
||||
return fallback
|
||||
frame = data_obj["df_fut"]
|
||||
if column in frame.columns and timestamp in frame.index:
|
||||
try:
|
||||
value = float(frame.at[timestamp, column])
|
||||
if np.isfinite(value):
|
||||
return value
|
||||
except Exception:
|
||||
pass
|
||||
return fallback
|
||||
|
||||
|
||||
def _battery_meta(config, data_obj):
|
||||
cap_kwh = _cfg_float(config, "batt_capacity_kwh", 0.0)
|
||||
max_power_w = _cfg_float(config, "batt_power_kw", 0.0) * 1000.0
|
||||
min_soc = _cfg_float(config, "batt_min_soc", _cfg_float(config, "batt_min_soc_percent", 0.0))
|
||||
max_soc = _cfg_float(config, "batt_max_soc", _cfg_float(config, "batt_max_soc_percent", 100.0))
|
||||
start_soc_value = data_obj.get("current_soc_perc")
|
||||
if start_soc_value is None:
|
||||
start_soc_value = min_soc
|
||||
try:
|
||||
start_soc = float(start_soc_value)
|
||||
except (TypeError, ValueError):
|
||||
start_soc = min_soc
|
||||
min_soc = max(0.0, min(100.0, min_soc))
|
||||
max_soc = max(min_soc, min(100.0, max_soc))
|
||||
start_soc = max(min_soc, min(max_soc, start_soc))
|
||||
charge_eff = max(0.01, min(1.0, _cfg_float(config, "batt_charge_efficiency", 0.95)))
|
||||
discharge_eff = max(0.01, min(1.0, _cfg_float(config, "batt_discharge_efficiency", 0.95)))
|
||||
return cap_kwh, max_power_w, min_soc, max_soc, start_soc, charge_eff, discharge_eff
|
||||
|
||||
|
||||
def _quarter_groups(index):
|
||||
groups = {}
|
||||
for position, timestamp in enumerate(index):
|
||||
quarter = timestamp.floor("15min") if hasattr(timestamp, "floor") else position // 3
|
||||
groups.setdefault(quarter, []).append(position)
|
||||
return list(groups.values())
|
||||
|
||||
|
||||
def _fallback_plan(index, residual_w):
|
||||
grid = {timestamp: float(value) for timestamp, value in zip(index, residual_w)}
|
||||
battery = {timestamp: 0.0 for timestamp in index}
|
||||
return {"grid": grid, "battery": battery, "solver": "fallback"}
|
||||
|
||||
|
||||
def optimize_battery_plan(data_obj, pv_dict, load_dict):
|
||||
config = data_obj["config"]
|
||||
cap_kwh, max_power_w, min_soc, max_soc, start_soc, charge_eff, discharge_eff = _battery_meta(config, data_obj)
|
||||
index = list(data_obj["df_fut"].index)
|
||||
if not index:
|
||||
return {"grid": {}, "battery": {}, "solver": "empty"}
|
||||
|
||||
load_w = np.array([max(0.0, float(load_dict.get(t, 0.0))) for t in index])
|
||||
pv_w = np.array([max(0.0, float(pv_dict.get(t, 0.0))) for t in index])
|
||||
residual_w = load_w - pv_w
|
||||
if cap_kwh <= 0.0 or max_power_w <= 0.0:
|
||||
return _fallback_plan(index, residual_w)
|
||||
|
||||
import_fixed = _cfg_float(config, "tarif_bezug_fest", 0.30)
|
||||
export_fixed = _cfg_float(config, "tarif_einspeisung_fest", 0.10)
|
||||
import_dynamic = _use_dynamic(config, "tarif_bezug")
|
||||
export_dynamic = _use_dynamic(config, "tarif_einspeisung")
|
||||
import_price = np.array([
|
||||
_price(data_obj, t, "import_price", import_fixed, import_dynamic) for t in index
|
||||
])
|
||||
export_price = np.array([
|
||||
_price(data_obj, t, "export_price", export_fixed, export_dynamic) for t in index
|
||||
])
|
||||
|
||||
n = len(index)
|
||||
imp, exp, charge, discharge, curtail, soc = 0, n, 2 * n, 3 * n, 4 * n, 5 * n
|
||||
peak = 6 * n + 1
|
||||
battery_mode = peak + 1
|
||||
grid_mode = battery_mode + n
|
||||
variable_count = grid_mode + n
|
||||
|
||||
max_import_w = max(float(load_w.max(initial=0.0)) + max_power_w, max_power_w, 1.0)
|
||||
configured_import_limit = _cfg_float(config, "grid_import_limit_w", 0.0)
|
||||
if configured_import_limit > 0.0:
|
||||
max_import_w = min(max_import_w, configured_import_limit)
|
||||
max_export_w = max(float(pv_w.max(initial=0.0)) + max_power_w, max_power_w, 1.0)
|
||||
configured_export_limit = _cfg_float(config, "grid_export_limit_w", 0.0)
|
||||
if configured_export_limit > 0.0:
|
||||
max_export_w = min(max_export_w, configured_export_limit)
|
||||
|
||||
lower = np.zeros(variable_count)
|
||||
upper = np.full(variable_count, np.inf)
|
||||
upper[imp:imp + n] = max_import_w
|
||||
upper[exp:exp + n] = max_export_w
|
||||
upper[charge:charge + n] = max_power_w
|
||||
if not _cfg_bool(config, "batt_grid_charging_enabled", False):
|
||||
upper[charge:charge + n] = np.minimum(max_power_w, np.maximum(0.0, pv_w - load_w))
|
||||
upper[discharge:discharge + n] = max_power_w
|
||||
upper[curtail:curtail + n] = pv_w
|
||||
reserve_soc = max(
|
||||
min_soc,
|
||||
min(100.0, _cfg_float(config, "batt_economic_reserve_soc_percent", 10.0)),
|
||||
)
|
||||
economic_min_soc = max(min_soc, min(start_soc, reserve_soc))
|
||||
min_kwh = cap_kwh * economic_min_soc / 100.0
|
||||
max_kwh = cap_kwh * max_soc / 100.0
|
||||
lower[soc:soc + n + 1] = min_kwh
|
||||
upper[soc:soc + n + 1] = max_kwh
|
||||
current_peak_kw = max(0.0, float(data_obj.get("current_month_peak_kw", 0.0) or 0.0))
|
||||
lower[peak] = current_peak_kw
|
||||
upper[peak] = max(current_peak_kw, max_import_w / 1000.0)
|
||||
upper[battery_mode:battery_mode + n] = 1.0
|
||||
upper[grid_mode:grid_mode + n] = 1.0
|
||||
|
||||
objective = np.zeros(variable_count)
|
||||
objective[imp:imp + n] = import_price * DT_H / 1000.0
|
||||
objective[exp:exp + n] = -export_price * DT_H / 1000.0
|
||||
degradation = max(0.0, _cfg_float(config, "batt_degradation_chf_kwh", 0.03))
|
||||
objective[charge:charge + n] = (degradation / 2.0 + 1e-7) * DT_H / 1000.0
|
||||
objective[discharge:discharge + n] = (degradation / 2.0 + 1e-7) * DT_H / 1000.0
|
||||
objective[curtail:curtail + n] = 1e-9 * DT_H / 1000.0
|
||||
objective[peak] = max(0.0, _cfg_float(config, "tarif_peak_fest", 0.0))
|
||||
terminal_value = _cfg_float(config, "batt_terminal_value_chf_kwh", np.median(import_price))
|
||||
objective[soc + n] = -max(0.0, terminal_value) * discharge_eff
|
||||
|
||||
equality_rows = 2 * n + 1
|
||||
equality = lil_matrix((equality_rows, variable_count), dtype=float)
|
||||
equality_rhs = np.zeros(equality_rows)
|
||||
for i in range(n):
|
||||
equality[i, imp + i] = 1.0
|
||||
equality[i, exp + i] = -1.0
|
||||
equality[i, charge + i] = -1.0
|
||||
equality[i, discharge + i] = 1.0
|
||||
equality[i, curtail + i] = -1.0
|
||||
equality_rhs[i] = residual_w[i]
|
||||
|
||||
row = n + i
|
||||
equality[row, soc + i] = -1.0
|
||||
equality[row, soc + i + 1] = 1.0
|
||||
equality[row, charge + i] = -charge_eff * DT_H / 1000.0
|
||||
equality[row, discharge + i] = DT_H / (1000.0 * discharge_eff)
|
||||
equality[2 * n, soc] = 1.0
|
||||
equality_rhs[2 * n] = cap_kwh * start_soc / 100.0
|
||||
|
||||
quarter_groups = _quarter_groups(pd.Index(index))
|
||||
inequality_rows = 4 * n + len(quarter_groups)
|
||||
inequality = lil_matrix((inequality_rows, variable_count), dtype=float)
|
||||
inequality_upper = np.zeros(inequality_rows)
|
||||
row = 0
|
||||
for i in range(n):
|
||||
inequality[row, charge + i] = 1.0
|
||||
inequality[row, battery_mode + i] = -max_power_w
|
||||
row += 1
|
||||
inequality[row, discharge + i] = 1.0
|
||||
inequality[row, battery_mode + i] = max_power_w
|
||||
inequality_upper[row] = max_power_w
|
||||
row += 1
|
||||
inequality[row, imp + i] = 1.0
|
||||
inequality[row, grid_mode + i] = -max_import_w
|
||||
row += 1
|
||||
inequality[row, exp + i] = 1.0
|
||||
inequality[row, grid_mode + i] = max_export_w
|
||||
inequality_upper[row] = max_export_w
|
||||
row += 1
|
||||
for group in quarter_groups:
|
||||
for i in group:
|
||||
inequality[row, imp + i] = 1.0 / (len(group) * 1000.0)
|
||||
inequality[row, peak] = -1.0
|
||||
row += 1
|
||||
|
||||
integrality = np.zeros(variable_count, dtype=int)
|
||||
integrality[battery_mode:battery_mode + n] = 1
|
||||
integrality[grid_mode:grid_mode + n] = 1
|
||||
constraints = [
|
||||
LinearConstraint(equality.tocsr(), equality_rhs, equality_rhs),
|
||||
LinearConstraint(inequality.tocsr(), -np.inf, inequality_upper),
|
||||
]
|
||||
result = milp(
|
||||
objective,
|
||||
integrality=integrality,
|
||||
bounds=Bounds(lower, upper),
|
||||
constraints=constraints,
|
||||
options={"time_limit": max(5.0, _cfg_float(config, "batt_optimizer_timeout_seconds", 30.0))},
|
||||
)
|
||||
if not result.success or result.x is None:
|
||||
return _fallback_plan(index, residual_w)
|
||||
|
||||
grid_values = result.x[imp:imp + n] - result.x[exp:exp + n]
|
||||
battery_values = result.x[charge:charge + n] - result.x[discharge:discharge + n]
|
||||
threshold_w = max(25.0, max_power_w * 0.005)
|
||||
grid_values[np.abs(grid_values) < threshold_w] = 0.0
|
||||
battery_values[np.abs(battery_values) < threshold_w] = 0.0
|
||||
return {
|
||||
"grid": {t: float(v) for t, v in zip(index, grid_values)},
|
||||
"battery": {t: float(v) for t, v in zip(index, battery_values)},
|
||||
"solver": "scipy-milp",
|
||||
"objective_chf": float(result.fun),
|
||||
"planned_peak_kw": float(result.x[peak]),
|
||||
"start_soc_percent": float(start_soc),
|
||||
"economic_min_soc_percent": float(economic_min_soc),
|
||||
}
|
||||
|
||||
|
||||
def optimize_grid_setpoint(data_obj, pv_dict, load_dict, forecast_id=None):
|
||||
plan = optimize_battery_plan(data_obj, pv_dict, load_dict)
|
||||
if forecast_id is not None:
|
||||
data_obj.setdefault("battery_plans", {})[int(forecast_id)] = plan
|
||||
return plan["grid"]
|
||||
@@ -0,0 +1,32 @@
|
||||
import os
|
||||
import joblib
|
||||
|
||||
MODEL_DIR = "/app/data/models"
|
||||
|
||||
|
||||
def model_path(aid, forecast_id):
|
||||
os.makedirs(MODEL_DIR, exist_ok=True)
|
||||
return os.path.join(MODEL_DIR, f"forecast_var_{forecast_id}_{aid}.pkl")
|
||||
|
||||
|
||||
def save_model(aid, forecast_id, artifact):
|
||||
path = model_path(aid, forecast_id)
|
||||
tmp_path = f"{path}.tmp.{os.getpid()}"
|
||||
try:
|
||||
joblib.dump(artifact, tmp_path)
|
||||
os.replace(tmp_path, path)
|
||||
finally:
|
||||
if os.path.exists(tmp_path):
|
||||
os.remove(tmp_path)
|
||||
return path
|
||||
|
||||
|
||||
def load_model(aid, forecast_id):
|
||||
path = model_path(aid, forecast_id)
|
||||
if not os.path.exists(path):
|
||||
return None
|
||||
try:
|
||||
return joblib.load(path)
|
||||
except Exception as exc:
|
||||
print(f"Modell {path} konnte nicht geladen werden: {exc}")
|
||||
return None
|
||||
@@ -0,0 +1,66 @@
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
from sklearn.ensemble import RandomForestRegressor
|
||||
|
||||
from methods.common import load_model, save_model
|
||||
from shared_utils import calc_pure_math_pv, roof_features
|
||||
|
||||
FORECAST_ID = 1
|
||||
FEATURES = [
|
||||
"math_pv",
|
||||
"temp_c",
|
||||
"cloud",
|
||||
"hour_sin",
|
||||
"hour_cos",
|
||||
"sin_year",
|
||||
"cos_year",
|
||||
"pv_kwp_total",
|
||||
"roof_azimuth_sin",
|
||||
"roof_azimuth_cos",
|
||||
"roof_tilt_avg",
|
||||
"roof_south_factor",
|
||||
]
|
||||
|
||||
|
||||
def _features(frame, config):
|
||||
out = frame.copy()
|
||||
out["math_pv"] = [calc_pure_math_pv(config, t) for t in out.index]
|
||||
rf = roof_features(config)
|
||||
for k, v in rf.items():
|
||||
out[k] = v
|
||||
for col, default in [("temp_c", 15.0), ("cloud", 20.0)]:
|
||||
if col not in out.columns:
|
||||
out[col] = default
|
||||
out[col] = pd.to_numeric(out[col], errors="coerce").ffill().bfill().fillna(default)
|
||||
return out[FEATURES].astype(float)
|
||||
|
||||
|
||||
def train(data_obj):
|
||||
config = data_obj["config"]
|
||||
aid = config["anlagen_id"]
|
||||
df = data_obj.get("df_pv_training", data_obj["df_hist"]).copy()
|
||||
if "PV" not in df.columns:
|
||||
return {"trained": False, "reason": "PV fehlt"}
|
||||
X = _features(df, config)
|
||||
y = pd.to_numeric(df["PV"], errors="coerce")
|
||||
valid = X.notna().all(axis=1) & y.notna()
|
||||
X, y = X.loc[valid], y.loc[valid]
|
||||
if len(X) < 288:
|
||||
return {"trained": False, "reason": "zu wenig Daten", "samples": int(len(X))}
|
||||
model = RandomForestRegressor(n_estimators=400, max_depth=18, min_samples_leaf=2, random_state=42, n_jobs=-1)
|
||||
model.fit(X, y)
|
||||
path = save_model(aid, FORECAST_ID, {"model": model, "features": FEATURES})
|
||||
return {"trained": True, "samples": int(len(X)), "path": path}
|
||||
|
||||
|
||||
def predict(data_obj):
|
||||
config = data_obj["config"]
|
||||
aid = config["anlagen_id"]
|
||||
artifact = load_model(aid, FORECAST_ID)
|
||||
X = _features(data_obj["df_fut"], config)
|
||||
if artifact and "model" in artifact:
|
||||
values = artifact["model"].predict(X)
|
||||
else:
|
||||
values = X["math_pv"].to_numpy()
|
||||
ac_limit = float(config.get("ac_leistung", 10.0) or 10.0) * 1000.0
|
||||
return {t: max(0.0, min(float(v), ac_limit)) for t, v in zip(data_obj["df_fut"].index, values)}
|
||||
@@ -0,0 +1,12 @@
|
||||
from telemetry_quality import repeat_daily_profile
|
||||
import datetime
|
||||
from methods.common import save_model
|
||||
|
||||
|
||||
def train(data_obj):
|
||||
aid = data_obj["config"]["anlagen_id"]
|
||||
return {"trained": True, "path": save_model(aid, 10, {"type": "repeat_pv_24h"})}
|
||||
|
||||
|
||||
def predict(data_obj):
|
||||
return repeat_daily_profile(data_obj["df_hist"], data_obj["df_fut"].index, 'PV')
|
||||
@@ -0,0 +1,12 @@
|
||||
from telemetry_quality import repeat_daily_profile
|
||||
import datetime
|
||||
from methods.common import save_model
|
||||
|
||||
|
||||
def train(data_obj):
|
||||
aid = data_obj["config"]["anlagen_id"]
|
||||
return {"trained": True, "path": save_model(aid, 11, {"type": "repeat_load_24h"})}
|
||||
|
||||
|
||||
def predict(data_obj):
|
||||
return repeat_daily_profile(data_obj["df_hist"], data_obj["df_fut"].index, 'Hausverbrauch')
|
||||
@@ -0,0 +1,13 @@
|
||||
from methods.battery_optimizer import optimize_grid_setpoint, train_artifact
|
||||
from methods.common import save_model
|
||||
|
||||
FORECAST_ID = 13
|
||||
|
||||
|
||||
def train(data_obj):
|
||||
aid = data_obj["config"]["anlagen_id"]
|
||||
return {"trained": True, "path": save_model(aid, FORECAST_ID, train_artifact("var_10_11"))}
|
||||
|
||||
|
||||
def predict(data_obj, pv_dict, load_dict):
|
||||
return optimize_grid_setpoint(data_obj, pv_dict, load_dict, forecast_id=13)
|
||||
@@ -0,0 +1,169 @@
|
||||
import datetime
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
from sklearn.ensemble import HistGradientBoostingRegressor
|
||||
|
||||
from methods.common import load_model, save_model
|
||||
from telemetry_quality import profile_source_value
|
||||
|
||||
FORECAST_ID = 2
|
||||
FEATURES = [
|
||||
"temp_c",
|
||||
"cloud",
|
||||
"hour_sin",
|
||||
"hour_cos",
|
||||
"weekday",
|
||||
"is_weekday",
|
||||
"load_24h_ago",
|
||||
"load_7d_ago",
|
||||
"energy_24h_rolling",
|
||||
"load_3d_same_time_mean",
|
||||
"load_7d_same_time_mean",
|
||||
]
|
||||
|
||||
|
||||
def _history_features(df):
|
||||
out = df.copy()
|
||||
out["load_24h_ago"] = out["Hausverbrauch"].shift(288)
|
||||
out["load_7d_ago"] = out["Hausverbrauch"].shift(2016)
|
||||
out["energy_24h_rolling"] = (out["Hausverbrauch"] * 5 / 60 / 1000).shift(1).rolling(288).sum()
|
||||
same_time_lags = [out["Hausverbrauch"].shift(288 * d) for d in range(1, 8)]
|
||||
out["load_3d_same_time_mean"] = pd.concat(same_time_lags[:3], axis=1).mean(axis=1)
|
||||
out["load_7d_same_time_mean"] = pd.concat(same_time_lags, axis=1).mean(axis=1)
|
||||
for col, default in [("temp_c", 15.0), ("cloud", 20.0)]:
|
||||
if col not in out.columns:
|
||||
out[col] = default
|
||||
out[col] = pd.to_numeric(out[col], errors="coerce").ffill().bfill().fillna(default)
|
||||
return out
|
||||
|
||||
|
||||
def train(data_obj):
|
||||
aid = data_obj["config"]["anlagen_id"]
|
||||
df = data_obj.get("df_load_training", data_obj["df_hist"]).copy()
|
||||
if "Hausverbrauch" not in df.columns:
|
||||
return {"trained": False, "reason": "Hausverbrauch fehlt"}
|
||||
df = _history_features(df)
|
||||
X = df[FEATURES].astype(float)
|
||||
y = pd.to_numeric(df["Hausverbrauch"], errors="coerce")
|
||||
valid = X.notna().all(axis=1) & y.notna()
|
||||
X, y = X.loc[valid], y.loc[valid]
|
||||
if len(X) < 288:
|
||||
return {"trained": False, "reason": "zu wenig Daten", "samples": int(len(X))}
|
||||
validation_day = X.index.max().normalize()
|
||||
validation_mask = X.index >= validation_day
|
||||
if int(validation_mask.sum()) < 144:
|
||||
validation_day -= datetime.timedelta(days=1)
|
||||
validation_mask = (X.index >= validation_day) & (X.index < validation_day + datetime.timedelta(days=1))
|
||||
score = None
|
||||
if int((~validation_mask).sum()) >= 288 and int(validation_mask.sum()) >= 96:
|
||||
eval_model = HistGradientBoostingRegressor(
|
||||
max_iter=900,
|
||||
max_depth=12,
|
||||
learning_rate=0.02,
|
||||
min_samples_leaf=8,
|
||||
random_state=42,
|
||||
)
|
||||
eval_model.fit(X.loc[~validation_mask], y.loc[~validation_mask])
|
||||
score = float(eval_model.score(X.loc[validation_mask], y.loc[validation_mask]))
|
||||
|
||||
model = HistGradientBoostingRegressor(
|
||||
max_iter=1200,
|
||||
max_depth=12,
|
||||
learning_rate=0.02,
|
||||
min_samples_leaf=8,
|
||||
random_state=42,
|
||||
)
|
||||
model.fit(X, y)
|
||||
path = save_model(
|
||||
aid,
|
||||
FORECAST_ID,
|
||||
{"model": model, "features": FEATURES, "r2_last_day": score},
|
||||
)
|
||||
return {
|
||||
"trained": True,
|
||||
"samples": int(len(X)),
|
||||
"r2_last_day": score,
|
||||
"path": path,
|
||||
}
|
||||
|
||||
|
||||
def _history_value(history, ts):
|
||||
if history is None or history.empty or ts not in history.index or "Hausverbrauch" not in history.columns:
|
||||
return None
|
||||
value = history.at[ts, "Hausverbrauch"]
|
||||
return float(value) if pd.notna(value) and np.isfinite(value) and value >= 0 else None
|
||||
|
||||
|
||||
def _same_time_values(history, t, days):
|
||||
values = []
|
||||
for day in range(1, days + 1):
|
||||
value = _history_value(history, t - datetime.timedelta(days=day))
|
||||
if value is not None and np.isfinite(value):
|
||||
values.append(max(0.0, value))
|
||||
return values
|
||||
|
||||
|
||||
def predict(data_obj):
|
||||
aid = data_obj["config"]["anlagen_id"]
|
||||
artifact = load_model(aid, FORECAST_ID)
|
||||
model = artifact["model"] if artifact and "model" in artifact else None
|
||||
try:
|
||||
score = float(artifact.get("r2_last_day")) if artifact and artifact.get("r2_last_day") is not None else 0.0
|
||||
except (TypeError, ValueError):
|
||||
score = 0.0
|
||||
model_weight = min(0.35, max(0.0, score) * 0.35) if np.isfinite(score) else 0.0
|
||||
|
||||
frames = [
|
||||
frame
|
||||
for frame in (
|
||||
data_obj.get("df_load_training"),
|
||||
data_obj.get("df_hist"),
|
||||
data_obj.get("df_recent_raw"),
|
||||
)
|
||||
if frame is not None and not frame.empty and "Hausverbrauch" in frame.columns
|
||||
]
|
||||
history = pd.concat(frames).sort_index() if frames else pd.DataFrame(columns=["Hausverbrauch"])
|
||||
history = history[~history.index.duplicated(keep="last")]
|
||||
fut = data_obj["df_fut"]
|
||||
res = {}
|
||||
|
||||
for t in fut.index:
|
||||
same_values = _same_time_values(history, t, 7)
|
||||
load_24 = _history_value(history, t - datetime.timedelta(days=1))
|
||||
load_7d = _history_value(history, t - datetime.timedelta(days=7))
|
||||
if load_24 is None:
|
||||
load_24 = load_7d if load_7d is not None else profile_source_value(history, t, "Hausverbrauch")
|
||||
if load_7d is None:
|
||||
load_7d = load_24
|
||||
same_median = float(np.median(same_values)) if same_values else load_7d
|
||||
profile = max(0.0, (0.50 * load_24) + (0.30 * load_7d) + (0.20 * same_median))
|
||||
|
||||
window = history.loc[
|
||||
t - datetime.timedelta(days=1):t - datetime.timedelta(minutes=5),
|
||||
"Hausverbrauch",
|
||||
]
|
||||
energy = float((window.sum() * 5 / 60) / 1000.0) if not window.empty else 0.0
|
||||
same_3 = same_values[:3]
|
||||
row = pd.DataFrame([[
|
||||
float(fut.at[t, "temp_c"]),
|
||||
float(fut.at[t, "cloud"]),
|
||||
float(fut.at[t, "hour_sin"]),
|
||||
float(fut.at[t, "hour_cos"]),
|
||||
int(fut.at[t, "weekday"]),
|
||||
int(fut.at[t, "is_weekday"]),
|
||||
load_24,
|
||||
load_7d,
|
||||
energy,
|
||||
float(np.mean(same_3)) if same_3 else profile,
|
||||
float(np.mean(same_values)) if same_values else profile,
|
||||
]], columns=FEATURES)
|
||||
|
||||
pred = profile
|
||||
if model is not None and model_weight > 0.0:
|
||||
model_pred = max(0.0, float(model.predict(row)[0]))
|
||||
model_pred = min(max(model_pred, profile * 0.25), max(500.0, profile * 3.0))
|
||||
pred = ((1.0 - model_weight) * profile) + (model_weight * model_pred)
|
||||
|
||||
res[t] = pred
|
||||
history.loc[t, "Hausverbrauch"] = pred
|
||||
return res
|
||||
@@ -0,0 +1,62 @@
|
||||
import datetime
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
from sklearn.ensemble import RandomForestRegressor
|
||||
|
||||
from methods.common import load_model, save_model
|
||||
|
||||
FORECAST_ID = 21
|
||||
FEATURES = ["temp_c", "hour_cos", "pv_24h_ago"]
|
||||
|
||||
|
||||
def _feature_frame(df):
|
||||
out = df.copy()
|
||||
if "temp_c" not in out.columns:
|
||||
out["temp_c"] = 15.0
|
||||
out["temp_c"] = pd.to_numeric(out["temp_c"], errors="coerce").ffill().bfill().fillna(15.0)
|
||||
out["hour_float"] = out.index.hour + out.index.minute / 60.0
|
||||
out["hour_cos"] = np.cos(2 * np.pi * out["hour_float"] / 24.0)
|
||||
if "PV" in out.columns:
|
||||
out["pv_24h_ago"] = out["PV"].shift(288)
|
||||
return out
|
||||
|
||||
|
||||
def train(data_obj):
|
||||
aid = data_obj["config"]["anlagen_id"]
|
||||
df = _feature_frame(data_obj.get("df_pv_training", data_obj["df_hist"]).copy())
|
||||
if "PV" not in df.columns:
|
||||
return {"trained": False, "reason": "PV fehlt"}
|
||||
df = df.dropna(subset=FEATURES + ["PV"])
|
||||
if len(df) < 288:
|
||||
return {"trained": False, "reason": "zu wenig Daten", "samples": int(len(df))}
|
||||
|
||||
last_day = df.index.max().normalize()
|
||||
train_df = df[df.index < last_day]
|
||||
test_df = df[(df.index >= last_day) & (df.index < last_day + datetime.timedelta(days=1))]
|
||||
score = None
|
||||
if len(train_df) >= 288 and len(test_df) >= 12:
|
||||
eval_model = RandomForestRegressor(n_estimators=300, max_depth=15, random_state=42)
|
||||
eval_model.fit(train_df[FEATURES], train_df["PV"])
|
||||
score = float(eval_model.score(test_df[FEATURES], test_df["PV"]))
|
||||
print(f"[var_21] R2 PV letzter kompletter Tag: {score:.3f}")
|
||||
|
||||
model = RandomForestRegressor(n_estimators=300, max_depth=15, random_state=42)
|
||||
model.fit(df[FEATURES], df["PV"])
|
||||
path = save_model(aid, FORECAST_ID, {"model": model, "features": FEATURES, "r2_last_day": score})
|
||||
return {"trained": True, "samples": int(len(df)), "features": FEATURES, "r2_last_day": score, "path": path}
|
||||
|
||||
|
||||
def predict(data_obj):
|
||||
aid = data_obj["config"]["anlagen_id"]
|
||||
artifact = load_model(aid, FORECAST_ID)
|
||||
model = artifact["model"] if artifact and "model" in artifact else None
|
||||
hist = data_obj["df_hist"]
|
||||
fut = data_obj["df_fut"].copy()
|
||||
res = {}
|
||||
for t in fut.index:
|
||||
t_24 = t - datetime.timedelta(days=1)
|
||||
pv_24 = float(hist.at[t_24, "PV"]) if t_24 in hist.index else 0.0
|
||||
row = pd.DataFrame([[float(fut.at[t, "temp_c"]), float(fut.at[t, "hour_cos"]), pv_24]], columns=FEATURES)
|
||||
pred = float(model.predict(row)[0]) if model is not None else pv_24
|
||||
res[t] = max(0.0, pred)
|
||||
return res
|
||||
@@ -0,0 +1,121 @@
|
||||
import datetime
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
from sklearn.ensemble import HistGradientBoostingRegressor
|
||||
from sklearn.inspection import permutation_importance
|
||||
|
||||
from methods.common import load_model, save_model
|
||||
|
||||
FORECAST_ID = 22
|
||||
FEATURES = ["temp_c", "hour_cos", "load_24h_ago", "load_7d_ago", "energy_24h_rolling", "weekday"]
|
||||
|
||||
|
||||
def _feature_frame(df):
|
||||
out = df.copy()
|
||||
if "temp_c" not in out.columns:
|
||||
out["temp_c"] = 15.0
|
||||
out["temp_c"] = pd.to_numeric(out["temp_c"], errors="coerce").ffill().bfill().fillna(15.0)
|
||||
out["hour_float"] = out.index.hour + out.index.minute / 60.0
|
||||
out["hour_cos"] = np.cos(2 * np.pi * out["hour_float"] / 24.0)
|
||||
out["weekday"] = out.index.weekday
|
||||
if "Hausverbrauch" in out.columns:
|
||||
out["load_24h_ago"] = out["Hausverbrauch"].shift(288)
|
||||
out["load_7d_ago"] = out["Hausverbrauch"].shift(2016)
|
||||
out["energy_5m_kwh"] = out["Hausverbrauch"] * (5 / 60) / 1000
|
||||
out["energy_24h_rolling"] = out["energy_5m_kwh"].shift(1).rolling(window=288).sum()
|
||||
out = out.drop(columns=["energy_5m_kwh"])
|
||||
return out
|
||||
|
||||
|
||||
def _history_value(history, recent, reference, ts):
|
||||
for frame in (history, recent, reference):
|
||||
if frame is not None and not frame.empty and ts in frame.index and "Hausverbrauch" in frame.columns:
|
||||
value = frame.at[ts, "Hausverbrauch"]
|
||||
if pd.notna(value):
|
||||
return float(value)
|
||||
return None
|
||||
|
||||
|
||||
def train(data_obj):
|
||||
aid = data_obj["config"]["anlagen_id"]
|
||||
df = _feature_frame(data_obj.get("df_load_training", data_obj["df_hist"]).copy())
|
||||
if "Hausverbrauch" not in df.columns:
|
||||
return {"trained": False, "reason": "Hausverbrauch fehlt"}
|
||||
df = df.dropna(subset=FEATURES + ["Hausverbrauch"])
|
||||
if len(df) < 288:
|
||||
return {"trained": False, "reason": "zu wenig Daten", "samples": int(len(df))}
|
||||
|
||||
last_day = df.index.max().normalize()
|
||||
train_df = df[df.index < last_day]
|
||||
test_df = df[(df.index >= last_day) & (df.index < last_day + datetime.timedelta(days=1))]
|
||||
score = None
|
||||
importance = {}
|
||||
if len(train_df) >= 288 and len(test_df) >= 12:
|
||||
eval_model = HistGradientBoostingRegressor(max_iter=2500, max_depth=25, learning_rate=0.01, min_samples_leaf=1, random_state=42)
|
||||
eval_model.fit(train_df[FEATURES], train_df["Hausverbrauch"])
|
||||
score = float(eval_model.score(test_df[FEATURES], test_df["Hausverbrauch"]))
|
||||
print(f"[var_22] R2 Hausverbrauch letzter kompletter Tag: {score:.3f}")
|
||||
try:
|
||||
perm = permutation_importance(eval_model, test_df[FEATURES], test_df["Hausverbrauch"], n_repeats=10, random_state=42)
|
||||
order = perm.importances_mean.argsort()[::-1]
|
||||
importance = {FEATURES[i]: float(perm.importances_mean[i]) for i in order}
|
||||
print("[var_22] Feature-Wichtigkeit Hausverbrauch:")
|
||||
for name, val in importance.items():
|
||||
print(f" {name}: {val:.4f}")
|
||||
except Exception as exc:
|
||||
print(f"[var_22] permutation_importance nicht berechnet: {exc}")
|
||||
|
||||
model = HistGradientBoostingRegressor(max_iter=2500, max_depth=25, learning_rate=0.01, min_samples_leaf=1, random_state=42)
|
||||
model.fit(df[FEATURES], df["Hausverbrauch"])
|
||||
path = save_model(aid, FORECAST_ID, {"model": model, "features": FEATURES, "r2_last_day": score, "importance": importance})
|
||||
return {"trained": True, "samples": int(len(df)), "features": FEATURES, "r2_last_day": score, "path": path}
|
||||
|
||||
|
||||
def predict(data_obj):
|
||||
aid = data_obj["config"]["anlagen_id"]
|
||||
artifact = load_model(aid, FORECAST_ID)
|
||||
model = artifact["model"] if artifact and "model" in artifact else None
|
||||
try:
|
||||
score = float(artifact.get("r2_last_day")) if artifact and artifact.get("r2_last_day") is not None else 0.0
|
||||
except (TypeError, ValueError):
|
||||
score = 0.0
|
||||
model_weight = min(0.35, max(0.0, score) * 0.35) if np.isfinite(score) else 0.0
|
||||
hist = data_obj["df_hist"].copy()
|
||||
recent = data_obj.get("df_recent_raw", pd.DataFrame())
|
||||
reference = data_obj.get("df_load_training", pd.DataFrame())
|
||||
fut = data_obj["df_fut"]
|
||||
res = {}
|
||||
for t in fut.index:
|
||||
t_24 = t - datetime.timedelta(days=1)
|
||||
t_7d = t - datetime.timedelta(days=7)
|
||||
load_7d = _history_value(hist, recent, reference, t_7d)
|
||||
load_24 = _history_value(hist, recent, reference, t_24)
|
||||
if load_24 is None:
|
||||
load_24 = load_7d if load_7d is not None else 0.0
|
||||
if load_7d is None:
|
||||
load_7d = load_24
|
||||
ref_end = t - datetime.timedelta(days=7)
|
||||
ref_start = ref_end - datetime.timedelta(days=1)
|
||||
if not reference.empty and "Hausverbrauch" in reference.columns:
|
||||
window = reference.loc[ref_start:ref_end - datetime.timedelta(minutes=5), "Hausverbrauch"]
|
||||
else:
|
||||
window = pd.Series(dtype=float)
|
||||
if window.empty and "Hausverbrauch" in recent.columns:
|
||||
window = recent.loc[t - datetime.timedelta(days=1):t - datetime.timedelta(minutes=5), "Hausverbrauch"]
|
||||
roll_energy = float((window.sum() * 5 / 60) / 1000.0) if not window.empty else 0.0
|
||||
row = pd.DataFrame([[
|
||||
float(fut.at[t, "temp_c"]),
|
||||
float(fut.at[t, "hour_cos"]),
|
||||
load_24,
|
||||
load_7d,
|
||||
roll_energy,
|
||||
int(fut.at[t, "weekday"]),
|
||||
]], columns=FEATURES)
|
||||
profile = max(0.0, (0.65 * load_24) + (0.35 * load_7d))
|
||||
pred = profile
|
||||
if model is not None and model_weight > 0.0:
|
||||
model_pred = max(0.0, float(model.predict(row)[0]))
|
||||
pred = ((1.0 - model_weight) * profile) + (model_weight * model_pred)
|
||||
res[t] = pred
|
||||
hist.loc[t, "Hausverbrauch"] = pred
|
||||
return res
|
||||
@@ -0,0 +1,13 @@
|
||||
from methods.battery_optimizer import optimize_grid_setpoint, train_artifact
|
||||
from methods.common import save_model
|
||||
|
||||
FORECAST_ID = 23
|
||||
|
||||
|
||||
def train(data_obj):
|
||||
aid = data_obj["config"]["anlagen_id"]
|
||||
return {"trained": True, "path": save_model(aid, FORECAST_ID, train_artifact("var_21_22"))}
|
||||
|
||||
|
||||
def predict(data_obj, pv_dict, load_dict):
|
||||
return optimize_grid_setpoint(data_obj, pv_dict, load_dict, forecast_id=23)
|
||||
@@ -0,0 +1,13 @@
|
||||
from methods.battery_optimizer import optimize_grid_setpoint, train_artifact
|
||||
from methods.common import save_model
|
||||
|
||||
FORECAST_ID = 3
|
||||
|
||||
|
||||
def train(data_obj):
|
||||
aid = data_obj["config"]["anlagen_id"]
|
||||
return {"trained": True, "path": save_model(aid, FORECAST_ID, train_artifact("var_1_2"))}
|
||||
|
||||
|
||||
def predict(data_obj, pv_dict, load_dict):
|
||||
return optimize_grid_setpoint(data_obj, pv_dict, load_dict, forecast_id=3)
|
||||
Reference in New Issue
Block a user