import datetime import numpy as np import pandas as pd from sklearn.ensemble import RandomForestRegressor from methods.common import load_model, save_model FORECAST_ID = 21 FEATURES = ["temp_c", "hour_cos", "pv_24h_ago"] def _feature_frame(df): out = df.copy() if "temp_c" not in out.columns: out["temp_c"] = 15.0 out["temp_c"] = pd.to_numeric(out["temp_c"], errors="coerce").ffill().bfill().fillna(15.0) out["hour_float"] = out.index.hour + out.index.minute / 60.0 out["hour_cos"] = np.cos(2 * np.pi * out["hour_float"] / 24.0) if "PV" in out.columns: out["pv_24h_ago"] = out["PV"].shift(288) return out def train(data_obj): aid = data_obj["config"]["anlagen_id"] df = _feature_frame(data_obj.get("df_pv_training", data_obj["df_hist"]).copy()) if "PV" not in df.columns: return {"trained": False, "reason": "PV fehlt"} df = df.dropna(subset=FEATURES + ["PV"]) if len(df) < 288: return {"trained": False, "reason": "zu wenig Daten", "samples": int(len(df))} last_day = df.index.max().normalize() train_df = df[df.index < last_day] test_df = df[(df.index >= last_day) & (df.index < last_day + datetime.timedelta(days=1))] score = None if len(train_df) >= 288 and len(test_df) >= 12: eval_model = RandomForestRegressor(n_estimators=300, max_depth=15, random_state=42) eval_model.fit(train_df[FEATURES], train_df["PV"]) score = float(eval_model.score(test_df[FEATURES], test_df["PV"])) print(f"[var_21] R2 PV letzter kompletter Tag: {score:.3f}") model = RandomForestRegressor(n_estimators=300, max_depth=15, random_state=42) model.fit(df[FEATURES], df["PV"]) path = save_model(aid, FORECAST_ID, {"model": model, "features": FEATURES, "r2_last_day": score}) return {"trained": True, "samples": int(len(df)), "features": FEATURES, "r2_last_day": score, "path": path} def predict(data_obj): aid = data_obj["config"]["anlagen_id"] artifact = load_model(aid, FORECAST_ID) model = artifact["model"] if artifact and "model" in artifact else None hist = data_obj["df_hist"] fut = data_obj["df_fut"].copy() res = {} for t in fut.index: t_24 = t - datetime.timedelta(days=1) pv_24 = float(hist.at[t_24, "PV"]) if t_24 in hist.index else 0.0 row = pd.DataFrame([[float(fut.at[t, "temp_c"]), float(fut.at[t, "hour_cos"]), pv_24]], columns=FEATURES) pred = float(model.predict(row)[0]) if model is not None else pv_24 res[t] = max(0.0, pred) return res