67 lines
2.1 KiB
Python
67 lines
2.1 KiB
Python
import numpy as np
|
|
import pandas as pd
|
|
from sklearn.ensemble import RandomForestRegressor
|
|
|
|
from methods.common import load_model, save_model
|
|
from shared_utils import calc_pure_math_pv, roof_features
|
|
|
|
FORECAST_ID = 1
|
|
FEATURES = [
|
|
"math_pv",
|
|
"temp_c",
|
|
"cloud",
|
|
"hour_sin",
|
|
"hour_cos",
|
|
"sin_year",
|
|
"cos_year",
|
|
"pv_kwp_total",
|
|
"roof_azimuth_sin",
|
|
"roof_azimuth_cos",
|
|
"roof_tilt_avg",
|
|
"roof_south_factor",
|
|
]
|
|
|
|
|
|
def _features(frame, config):
|
|
out = frame.copy()
|
|
out["math_pv"] = [calc_pure_math_pv(config, t) for t in out.index]
|
|
rf = roof_features(config)
|
|
for k, v in rf.items():
|
|
out[k] = v
|
|
for col, default in [("temp_c", 15.0), ("cloud", 20.0)]:
|
|
if col not in out.columns:
|
|
out[col] = default
|
|
out[col] = pd.to_numeric(out[col], errors="coerce").ffill().bfill().fillna(default)
|
|
return out[FEATURES].astype(float)
|
|
|
|
|
|
def train(data_obj):
|
|
config = data_obj["config"]
|
|
aid = config["anlagen_id"]
|
|
df = data_obj.get("df_pv_training", data_obj["df_hist"]).copy()
|
|
if "PV" not in df.columns:
|
|
return {"trained": False, "reason": "PV fehlt"}
|
|
X = _features(df, config)
|
|
y = pd.to_numeric(df["PV"], errors="coerce")
|
|
valid = X.notna().all(axis=1) & y.notna()
|
|
X, y = X.loc[valid], y.loc[valid]
|
|
if len(X) < 288:
|
|
return {"trained": False, "reason": "zu wenig Daten", "samples": int(len(X))}
|
|
model = RandomForestRegressor(n_estimators=400, max_depth=18, min_samples_leaf=2, random_state=42, n_jobs=-1)
|
|
model.fit(X, y)
|
|
path = save_model(aid, FORECAST_ID, {"model": model, "features": FEATURES})
|
|
return {"trained": True, "samples": int(len(X)), "path": path}
|
|
|
|
|
|
def predict(data_obj):
|
|
config = data_obj["config"]
|
|
aid = config["anlagen_id"]
|
|
artifact = load_model(aid, FORECAST_ID)
|
|
X = _features(data_obj["df_fut"], config)
|
|
if artifact and "model" in artifact:
|
|
values = artifact["model"].predict(X)
|
|
else:
|
|
values = X["math_pv"].to_numpy()
|
|
ac_limit = float(config.get("ac_leistung", 10.0) or 10.0) * 1000.0
|
|
return {t: max(0.0, min(float(v), ac_limit)) for t, v in zip(data_obj["df_fut"].index, values)}
|