feat(v4): consolidate audited data recovery economic replay and optional archival

This commit is contained in:
ENELIX Agent
2026-10-03 10:40:18 +00:00
parent 6c91d6bcc6
commit 8eb9688942
26 changed files with 1747 additions and 38 deletions
@@ -14,6 +14,7 @@ from statistics import median
from zoneinfo import ZoneInfo
import json
from .history_timing import validate_policy, endpoint_bridges
from . import mapping_identity
UTC = timezone.utc
LOCAL = ZoneInfo('Europe/Zurich')
@@ -42,6 +43,7 @@ def numeric(value, bound=1e12):
def schema(con):
mapping_identity.schema(con)
con.executescript('''
CREATE TABLE IF NOT EXISTS planner_data_sets(
plant TEXT NOT NULL, dataset TEXT NOT NULL, config TEXT NOT NULL,
@@ -162,10 +164,11 @@ def configuration(con, plant, dataset):
return json.loads(row[0])
def project(record, c, plant, now):
def project(record, c, plant, now, approved_mappings=()):
if not isinstance(record, dict) or type(record.get('schemaVersion')) is not int or record.get('schemaVersion') != 1 or record.get('kind') != 'raw_accounting_capture' or record.get('installationId') != plant:
raise ValueError('Wrong capture identity')
if record.get('mappingSha256') != c['mappingSha256'] or record.get('reportedInventorySha256') != c['inventorySha256']:
incoming = record.get('mappingSha256')
if not isinstance(incoming, str) or (incoming != c['mappingSha256'] and incoming not in approved_mappings) or record.get('reportedInventorySha256') != c['inventorySha256']:
raise ValueError('Wrong capture mapping or inventory')
t = epoch(record.get('capturedAt')); start = epoch(record.get('captureStartedAt'))
if start > t or t > now+30 or t < now-90*86400:
@@ -196,13 +199,14 @@ def ingest_batch(con, plant, payload, now):
records = payload['records']
if not isinstance(records,list) or not 1 <= len(records) <= 120:
raise ValueError('Batch requires 1..120 captures')
rows = [project(r,c,plant,now) for r in records]
aliases = mapping_identity.approved(con, plant, c)
rows = [project(r,c,plant,now,aliases) for r in records]
if any(a['capturedAt'] >= b['capturedAt'] for a,b in zip(rows,rows[1:])):
raise ValueError('Batch must be in increasing capture order')
stored = duplicate = 0
con.execute('BEGIN IMMEDIATE')
try:
for r in rows:
for original, r in zip(records, rows):
value = canonical(r); digest = sha256(value.encode()).hexdigest()
old = con.execute('SELECT fingerprint FROM planner_observations WHERE plant=? AND dataset=? AND captured_at=?', (plant,c['datasetId'],r['capturedAt'])).fetchone()
if old:
@@ -211,6 +215,7 @@ def ingest_batch(con, plant, payload, now):
duplicate += 1
else:
con.execute('INSERT INTO planner_observations VALUES(?,?,?,?,?,?)',(plant,c['datasetId'],r['capturedAt'],now,digest,value)); stored += 1
mapping_identity.save_origin(con, plant, c, original, r['capturedAt'], now, aliases)
con.commit()
except Exception:
con.rollback(); raise
@@ -243,7 +248,7 @@ def physical_value(values, c):
return load
def reconstruct(records, c):
def reconstruct(records, c, *, projection=None):
"""Bounded retrospective estimation, never a real-time feedback signal.
Missing observations split support. Source timestamps are not refreshed. Small
@@ -260,6 +265,11 @@ uncovered portions remain quantified and are never filled with zero.
for s in c['sources']:
if s['key'] in (sr['rawKey'],sr['scaleKey']): primary[s['key']] = s
timing_policy = validate_policy(c.get('historyTimingPolicy'), c['sources'])
if projection is not None:
configured={s['key']:s for s in c['sources']}
if set(projection)!= {'keys','calculate'} or not callable(projection['calculate']) or not projection['keys'] or not set(projection['keys']) <= set(configured):
raise ValueError('Invalid internal projection')
primary={k:configured[k] for k in projection['keys']}
first,last = records[0]['capturedAt'],records[-1]['capturedAt']
series = {k:{} for k in primary}; blocks = {k:[] for k in primary}; gaps = []
first_observed = {k:{} for k in primary}
@@ -313,7 +323,9 @@ uncovered portions remain quantified and are never filled with zero.
extended.append(k); known_at = max(known_at, bridge['availableAt'])
load = None
if usable:
try: load = physical_value(vals,c)
try:
load = physical_value(vals,c) if projection is None else projection['calculate'](vals,c)
if not numeric(load,1e9):raise ValueError('Invalid historical projection')
except ValueError: usable = False
if usable:
item['seconds'] += b-a; item['wattSeconds'] += load*(b-a);item['currentGap'] = 0
@@ -490,9 +502,10 @@ def pipeline_status(con,plant):
out=[]
for row in con.execute('SELECT dataset,config FROM planner_data_sets WHERE plant=? ORDER BY dataset',(plant,)):
ds=row[0]; c=json.loads(row[1]); state=con.execute('SELECT status,detail FROM planner_pipeline_state WHERE plant=? AND dataset=?',(plant,ds)).fetchone()
count=con.execute('SELECT COUNT(*),MIN(captured_at),MAX(captured_at) FROM planner_observations WHERE plant=? AND dataset=?',(plant,c.get('sourceDatasetId',ds))).fetchone()
count=con.execute('SELECT COUNT(*),MIN(captured_at),MAX(captured_at),MAX(received_at) FROM planner_observations WHERE plant=? AND dataset=?',(plant,c.get('sourceDatasetId',ds))).fetchone()
out.append({'datasetId':ds,'formula':c['formula'],'mappingSha256':c['mappingSha256'],'records':count[0],
'firstCapture':iso(count[1]) if count[1] else None,'lastCapture':iso(count[2]) if count[2] else None,
'lastReceived':iso(count[3]) if count[3] else None,
'status':state[0] if state else 'awaiting_measurements','detail':json.loads(state[1]) if state else {},
'minimumCoverage':c['minimumCoverage'],'maximumGapSeconds':c['maximumGapSeconds'],
'observationDatasetId':c.get('sourceDatasetId',ds),