feat(application): integrate measured-load ingestion training and planner source
This commit is contained in:
@@ -0,0 +1,62 @@
|
||||
import datetime
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
from sklearn.ensemble import RandomForestRegressor
|
||||
|
||||
from methods.common import load_model, save_model
|
||||
|
||||
FORECAST_ID = 21
|
||||
FEATURES = ["temp_c", "hour_cos", "pv_24h_ago"]
|
||||
|
||||
|
||||
def _feature_frame(df):
|
||||
out = df.copy()
|
||||
if "temp_c" not in out.columns:
|
||||
out["temp_c"] = 15.0
|
||||
out["temp_c"] = pd.to_numeric(out["temp_c"], errors="coerce").ffill().bfill().fillna(15.0)
|
||||
out["hour_float"] = out.index.hour + out.index.minute / 60.0
|
||||
out["hour_cos"] = np.cos(2 * np.pi * out["hour_float"] / 24.0)
|
||||
if "PV" in out.columns:
|
||||
out["pv_24h_ago"] = out["PV"].shift(288)
|
||||
return out
|
||||
|
||||
|
||||
def train(data_obj):
|
||||
aid = data_obj["config"]["anlagen_id"]
|
||||
df = _feature_frame(data_obj.get("df_pv_training", data_obj["df_hist"]).copy())
|
||||
if "PV" not in df.columns:
|
||||
return {"trained": False, "reason": "PV fehlt"}
|
||||
df = df.dropna(subset=FEATURES + ["PV"])
|
||||
if len(df) < 288:
|
||||
return {"trained": False, "reason": "zu wenig Daten", "samples": int(len(df))}
|
||||
|
||||
last_day = df.index.max().normalize()
|
||||
train_df = df[df.index < last_day]
|
||||
test_df = df[(df.index >= last_day) & (df.index < last_day + datetime.timedelta(days=1))]
|
||||
score = None
|
||||
if len(train_df) >= 288 and len(test_df) >= 12:
|
||||
eval_model = RandomForestRegressor(n_estimators=300, max_depth=15, random_state=42)
|
||||
eval_model.fit(train_df[FEATURES], train_df["PV"])
|
||||
score = float(eval_model.score(test_df[FEATURES], test_df["PV"]))
|
||||
print(f"[var_21] R2 PV letzter kompletter Tag: {score:.3f}")
|
||||
|
||||
model = RandomForestRegressor(n_estimators=300, max_depth=15, random_state=42)
|
||||
model.fit(df[FEATURES], df["PV"])
|
||||
path = save_model(aid, FORECAST_ID, {"model": model, "features": FEATURES, "r2_last_day": score})
|
||||
return {"trained": True, "samples": int(len(df)), "features": FEATURES, "r2_last_day": score, "path": path}
|
||||
|
||||
|
||||
def predict(data_obj):
|
||||
aid = data_obj["config"]["anlagen_id"]
|
||||
artifact = load_model(aid, FORECAST_ID)
|
||||
model = artifact["model"] if artifact and "model" in artifact else None
|
||||
hist = data_obj["df_hist"]
|
||||
fut = data_obj["df_fut"].copy()
|
||||
res = {}
|
||||
for t in fut.index:
|
||||
t_24 = t - datetime.timedelta(days=1)
|
||||
pv_24 = float(hist.at[t_24, "PV"]) if t_24 in hist.index else 0.0
|
||||
row = pd.DataFrame([[float(fut.at[t, "temp_c"]), float(fut.at[t, "hour_cos"]), pv_24]], columns=FEATURES)
|
||||
pred = float(model.predict(row)[0]) if model is not None else pv_24
|
||||
res[t] = max(0.0, pred)
|
||||
return res
|
||||
Reference in New Issue
Block a user