170 lines
6.1 KiB
Python
170 lines
6.1 KiB
Python
import datetime
|
|
import numpy as np
|
|
import pandas as pd
|
|
from sklearn.ensemble import HistGradientBoostingRegressor
|
|
|
|
from methods.common import load_model, save_model
|
|
from telemetry_quality import profile_source_value
|
|
|
|
FORECAST_ID = 2
|
|
FEATURES = [
|
|
"temp_c",
|
|
"cloud",
|
|
"hour_sin",
|
|
"hour_cos",
|
|
"weekday",
|
|
"is_weekday",
|
|
"load_24h_ago",
|
|
"load_7d_ago",
|
|
"energy_24h_rolling",
|
|
"load_3d_same_time_mean",
|
|
"load_7d_same_time_mean",
|
|
]
|
|
|
|
|
|
def _history_features(df):
|
|
out = df.copy()
|
|
out["load_24h_ago"] = out["Hausverbrauch"].shift(288)
|
|
out["load_7d_ago"] = out["Hausverbrauch"].shift(2016)
|
|
out["energy_24h_rolling"] = (out["Hausverbrauch"] * 5 / 60 / 1000).shift(1).rolling(288).sum()
|
|
same_time_lags = [out["Hausverbrauch"].shift(288 * d) for d in range(1, 8)]
|
|
out["load_3d_same_time_mean"] = pd.concat(same_time_lags[:3], axis=1).mean(axis=1)
|
|
out["load_7d_same_time_mean"] = pd.concat(same_time_lags, axis=1).mean(axis=1)
|
|
for col, default in [("temp_c", 15.0), ("cloud", 20.0)]:
|
|
if col not in out.columns:
|
|
out[col] = default
|
|
out[col] = pd.to_numeric(out[col], errors="coerce").ffill().bfill().fillna(default)
|
|
return out
|
|
|
|
|
|
def train(data_obj):
|
|
aid = data_obj["config"]["anlagen_id"]
|
|
df = data_obj.get("df_load_training", data_obj["df_hist"]).copy()
|
|
if "Hausverbrauch" not in df.columns:
|
|
return {"trained": False, "reason": "Hausverbrauch fehlt"}
|
|
df = _history_features(df)
|
|
X = df[FEATURES].astype(float)
|
|
y = pd.to_numeric(df["Hausverbrauch"], errors="coerce")
|
|
valid = X.notna().all(axis=1) & y.notna()
|
|
X, y = X.loc[valid], y.loc[valid]
|
|
if len(X) < 288:
|
|
return {"trained": False, "reason": "zu wenig Daten", "samples": int(len(X))}
|
|
validation_day = X.index.max().normalize()
|
|
validation_mask = X.index >= validation_day
|
|
if int(validation_mask.sum()) < 144:
|
|
validation_day -= datetime.timedelta(days=1)
|
|
validation_mask = (X.index >= validation_day) & (X.index < validation_day + datetime.timedelta(days=1))
|
|
score = None
|
|
if int((~validation_mask).sum()) >= 288 and int(validation_mask.sum()) >= 96:
|
|
eval_model = HistGradientBoostingRegressor(
|
|
max_iter=900,
|
|
max_depth=12,
|
|
learning_rate=0.02,
|
|
min_samples_leaf=8,
|
|
random_state=42,
|
|
)
|
|
eval_model.fit(X.loc[~validation_mask], y.loc[~validation_mask])
|
|
score = float(eval_model.score(X.loc[validation_mask], y.loc[validation_mask]))
|
|
|
|
model = HistGradientBoostingRegressor(
|
|
max_iter=1200,
|
|
max_depth=12,
|
|
learning_rate=0.02,
|
|
min_samples_leaf=8,
|
|
random_state=42,
|
|
)
|
|
model.fit(X, y)
|
|
path = save_model(
|
|
aid,
|
|
FORECAST_ID,
|
|
{"model": model, "features": FEATURES, "r2_last_day": score},
|
|
)
|
|
return {
|
|
"trained": True,
|
|
"samples": int(len(X)),
|
|
"r2_last_day": score,
|
|
"path": path,
|
|
}
|
|
|
|
|
|
def _history_value(history, ts):
|
|
if history is None or history.empty or ts not in history.index or "Hausverbrauch" not in history.columns:
|
|
return None
|
|
value = history.at[ts, "Hausverbrauch"]
|
|
return float(value) if pd.notna(value) and np.isfinite(value) and value >= 0 else None
|
|
|
|
|
|
def _same_time_values(history, t, days):
|
|
values = []
|
|
for day in range(1, days + 1):
|
|
value = _history_value(history, t - datetime.timedelta(days=day))
|
|
if value is not None and np.isfinite(value):
|
|
values.append(max(0.0, value))
|
|
return values
|
|
|
|
|
|
def predict(data_obj):
|
|
aid = data_obj["config"]["anlagen_id"]
|
|
artifact = load_model(aid, FORECAST_ID)
|
|
model = artifact["model"] if artifact and "model" in artifact else None
|
|
try:
|
|
score = float(artifact.get("r2_last_day")) if artifact and artifact.get("r2_last_day") is not None else 0.0
|
|
except (TypeError, ValueError):
|
|
score = 0.0
|
|
model_weight = min(0.35, max(0.0, score) * 0.35) if np.isfinite(score) else 0.0
|
|
|
|
frames = [
|
|
frame
|
|
for frame in (
|
|
data_obj.get("df_load_training"),
|
|
data_obj.get("df_hist"),
|
|
data_obj.get("df_recent_raw"),
|
|
)
|
|
if frame is not None and not frame.empty and "Hausverbrauch" in frame.columns
|
|
]
|
|
history = pd.concat(frames).sort_index() if frames else pd.DataFrame(columns=["Hausverbrauch"])
|
|
history = history[~history.index.duplicated(keep="last")]
|
|
fut = data_obj["df_fut"]
|
|
res = {}
|
|
|
|
for t in fut.index:
|
|
same_values = _same_time_values(history, t, 7)
|
|
load_24 = _history_value(history, t - datetime.timedelta(days=1))
|
|
load_7d = _history_value(history, t - datetime.timedelta(days=7))
|
|
if load_24 is None:
|
|
load_24 = load_7d if load_7d is not None else profile_source_value(history, t, "Hausverbrauch")
|
|
if load_7d is None:
|
|
load_7d = load_24
|
|
same_median = float(np.median(same_values)) if same_values else load_7d
|
|
profile = max(0.0, (0.50 * load_24) + (0.30 * load_7d) + (0.20 * same_median))
|
|
|
|
window = history.loc[
|
|
t - datetime.timedelta(days=1):t - datetime.timedelta(minutes=5),
|
|
"Hausverbrauch",
|
|
]
|
|
energy = float((window.sum() * 5 / 60) / 1000.0) if not window.empty else 0.0
|
|
same_3 = same_values[:3]
|
|
row = pd.DataFrame([[
|
|
float(fut.at[t, "temp_c"]),
|
|
float(fut.at[t, "cloud"]),
|
|
float(fut.at[t, "hour_sin"]),
|
|
float(fut.at[t, "hour_cos"]),
|
|
int(fut.at[t, "weekday"]),
|
|
int(fut.at[t, "is_weekday"]),
|
|
load_24,
|
|
load_7d,
|
|
energy,
|
|
float(np.mean(same_3)) if same_3 else profile,
|
|
float(np.mean(same_values)) if same_values else profile,
|
|
]], columns=FEATURES)
|
|
|
|
pred = profile
|
|
if model is not None and model_weight > 0.0:
|
|
model_pred = max(0.0, float(model.predict(row)[0]))
|
|
model_pred = min(max(model_pred, profile * 0.25), max(500.0, profile * 3.0))
|
|
pred = ((1.0 - model_weight) * profile) + (model_weight * model_pred)
|
|
|
|
res[t] = pred
|
|
history.loc[t, "Hausverbrauch"] = pred
|
|
return res
|