feat(application): integrate measured-load ingestion training and planner source
This commit is contained in:
@@ -0,0 +1,121 @@
|
||||
import datetime
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
from sklearn.ensemble import HistGradientBoostingRegressor
|
||||
from sklearn.inspection import permutation_importance
|
||||
|
||||
from methods.common import load_model, save_model
|
||||
|
||||
FORECAST_ID = 22
|
||||
FEATURES = ["temp_c", "hour_cos", "load_24h_ago", "load_7d_ago", "energy_24h_rolling", "weekday"]
|
||||
|
||||
|
||||
def _feature_frame(df):
|
||||
out = df.copy()
|
||||
if "temp_c" not in out.columns:
|
||||
out["temp_c"] = 15.0
|
||||
out["temp_c"] = pd.to_numeric(out["temp_c"], errors="coerce").ffill().bfill().fillna(15.0)
|
||||
out["hour_float"] = out.index.hour + out.index.minute / 60.0
|
||||
out["hour_cos"] = np.cos(2 * np.pi * out["hour_float"] / 24.0)
|
||||
out["weekday"] = out.index.weekday
|
||||
if "Hausverbrauch" in out.columns:
|
||||
out["load_24h_ago"] = out["Hausverbrauch"].shift(288)
|
||||
out["load_7d_ago"] = out["Hausverbrauch"].shift(2016)
|
||||
out["energy_5m_kwh"] = out["Hausverbrauch"] * (5 / 60) / 1000
|
||||
out["energy_24h_rolling"] = out["energy_5m_kwh"].shift(1).rolling(window=288).sum()
|
||||
out = out.drop(columns=["energy_5m_kwh"])
|
||||
return out
|
||||
|
||||
|
||||
def _history_value(history, recent, reference, ts):
|
||||
for frame in (history, recent, reference):
|
||||
if frame is not None and not frame.empty and ts in frame.index and "Hausverbrauch" in frame.columns:
|
||||
value = frame.at[ts, "Hausverbrauch"]
|
||||
if pd.notna(value):
|
||||
return float(value)
|
||||
return None
|
||||
|
||||
|
||||
def train(data_obj):
|
||||
aid = data_obj["config"]["anlagen_id"]
|
||||
df = _feature_frame(data_obj.get("df_load_training", data_obj["df_hist"]).copy())
|
||||
if "Hausverbrauch" not in df.columns:
|
||||
return {"trained": False, "reason": "Hausverbrauch fehlt"}
|
||||
df = df.dropna(subset=FEATURES + ["Hausverbrauch"])
|
||||
if len(df) < 288:
|
||||
return {"trained": False, "reason": "zu wenig Daten", "samples": int(len(df))}
|
||||
|
||||
last_day = df.index.max().normalize()
|
||||
train_df = df[df.index < last_day]
|
||||
test_df = df[(df.index >= last_day) & (df.index < last_day + datetime.timedelta(days=1))]
|
||||
score = None
|
||||
importance = {}
|
||||
if len(train_df) >= 288 and len(test_df) >= 12:
|
||||
eval_model = HistGradientBoostingRegressor(max_iter=2500, max_depth=25, learning_rate=0.01, min_samples_leaf=1, random_state=42)
|
||||
eval_model.fit(train_df[FEATURES], train_df["Hausverbrauch"])
|
||||
score = float(eval_model.score(test_df[FEATURES], test_df["Hausverbrauch"]))
|
||||
print(f"[var_22] R2 Hausverbrauch letzter kompletter Tag: {score:.3f}")
|
||||
try:
|
||||
perm = permutation_importance(eval_model, test_df[FEATURES], test_df["Hausverbrauch"], n_repeats=10, random_state=42)
|
||||
order = perm.importances_mean.argsort()[::-1]
|
||||
importance = {FEATURES[i]: float(perm.importances_mean[i]) for i in order}
|
||||
print("[var_22] Feature-Wichtigkeit Hausverbrauch:")
|
||||
for name, val in importance.items():
|
||||
print(f" {name}: {val:.4f}")
|
||||
except Exception as exc:
|
||||
print(f"[var_22] permutation_importance nicht berechnet: {exc}")
|
||||
|
||||
model = HistGradientBoostingRegressor(max_iter=2500, max_depth=25, learning_rate=0.01, min_samples_leaf=1, random_state=42)
|
||||
model.fit(df[FEATURES], df["Hausverbrauch"])
|
||||
path = save_model(aid, FORECAST_ID, {"model": model, "features": FEATURES, "r2_last_day": score, "importance": importance})
|
||||
return {"trained": True, "samples": int(len(df)), "features": FEATURES, "r2_last_day": score, "path": path}
|
||||
|
||||
|
||||
def predict(data_obj):
|
||||
aid = data_obj["config"]["anlagen_id"]
|
||||
artifact = load_model(aid, FORECAST_ID)
|
||||
model = artifact["model"] if artifact and "model" in artifact else None
|
||||
try:
|
||||
score = float(artifact.get("r2_last_day")) if artifact and artifact.get("r2_last_day") is not None else 0.0
|
||||
except (TypeError, ValueError):
|
||||
score = 0.0
|
||||
model_weight = min(0.35, max(0.0, score) * 0.35) if np.isfinite(score) else 0.0
|
||||
hist = data_obj["df_hist"].copy()
|
||||
recent = data_obj.get("df_recent_raw", pd.DataFrame())
|
||||
reference = data_obj.get("df_load_training", pd.DataFrame())
|
||||
fut = data_obj["df_fut"]
|
||||
res = {}
|
||||
for t in fut.index:
|
||||
t_24 = t - datetime.timedelta(days=1)
|
||||
t_7d = t - datetime.timedelta(days=7)
|
||||
load_7d = _history_value(hist, recent, reference, t_7d)
|
||||
load_24 = _history_value(hist, recent, reference, t_24)
|
||||
if load_24 is None:
|
||||
load_24 = load_7d if load_7d is not None else 0.0
|
||||
if load_7d is None:
|
||||
load_7d = load_24
|
||||
ref_end = t - datetime.timedelta(days=7)
|
||||
ref_start = ref_end - datetime.timedelta(days=1)
|
||||
if not reference.empty and "Hausverbrauch" in reference.columns:
|
||||
window = reference.loc[ref_start:ref_end - datetime.timedelta(minutes=5), "Hausverbrauch"]
|
||||
else:
|
||||
window = pd.Series(dtype=float)
|
||||
if window.empty and "Hausverbrauch" in recent.columns:
|
||||
window = recent.loc[t - datetime.timedelta(days=1):t - datetime.timedelta(minutes=5), "Hausverbrauch"]
|
||||
roll_energy = float((window.sum() * 5 / 60) / 1000.0) if not window.empty else 0.0
|
||||
row = pd.DataFrame([[
|
||||
float(fut.at[t, "temp_c"]),
|
||||
float(fut.at[t, "hour_cos"]),
|
||||
load_24,
|
||||
load_7d,
|
||||
roll_energy,
|
||||
int(fut.at[t, "weekday"]),
|
||||
]], columns=FEATURES)
|
||||
profile = max(0.0, (0.65 * load_24) + (0.35 * load_7d))
|
||||
pred = profile
|
||||
if model is not None and model_weight > 0.0:
|
||||
model_pred = max(0.0, float(model.predict(row)[0]))
|
||||
pred = ((1.0 - model_weight) * profile) + (model_weight * model_pred)
|
||||
res[t] = pred
|
||||
hist.loc[t, "Hausverbrauch"] = pred
|
||||
return res
|
||||
Reference in New Issue
Block a user