feat(v4): consolidate audited data recovery economic replay and optional archival
This commit is contained in:
@@ -14,6 +14,7 @@ from statistics import median
|
||||
from zoneinfo import ZoneInfo
|
||||
import json
|
||||
from .history_timing import validate_policy, endpoint_bridges
|
||||
from . import mapping_identity
|
||||
|
||||
UTC = timezone.utc
|
||||
LOCAL = ZoneInfo('Europe/Zurich')
|
||||
@@ -42,6 +43,7 @@ def numeric(value, bound=1e12):
|
||||
|
||||
|
||||
def schema(con):
|
||||
mapping_identity.schema(con)
|
||||
con.executescript('''
|
||||
CREATE TABLE IF NOT EXISTS planner_data_sets(
|
||||
plant TEXT NOT NULL, dataset TEXT NOT NULL, config TEXT NOT NULL,
|
||||
@@ -162,10 +164,11 @@ def configuration(con, plant, dataset):
|
||||
return json.loads(row[0])
|
||||
|
||||
|
||||
def project(record, c, plant, now):
|
||||
def project(record, c, plant, now, approved_mappings=()):
|
||||
if not isinstance(record, dict) or type(record.get('schemaVersion')) is not int or record.get('schemaVersion') != 1 or record.get('kind') != 'raw_accounting_capture' or record.get('installationId') != plant:
|
||||
raise ValueError('Wrong capture identity')
|
||||
if record.get('mappingSha256') != c['mappingSha256'] or record.get('reportedInventorySha256') != c['inventorySha256']:
|
||||
incoming = record.get('mappingSha256')
|
||||
if not isinstance(incoming, str) or (incoming != c['mappingSha256'] and incoming not in approved_mappings) or record.get('reportedInventorySha256') != c['inventorySha256']:
|
||||
raise ValueError('Wrong capture mapping or inventory')
|
||||
t = epoch(record.get('capturedAt')); start = epoch(record.get('captureStartedAt'))
|
||||
if start > t or t > now+30 or t < now-90*86400:
|
||||
@@ -196,13 +199,14 @@ def ingest_batch(con, plant, payload, now):
|
||||
records = payload['records']
|
||||
if not isinstance(records,list) or not 1 <= len(records) <= 120:
|
||||
raise ValueError('Batch requires 1..120 captures')
|
||||
rows = [project(r,c,plant,now) for r in records]
|
||||
aliases = mapping_identity.approved(con, plant, c)
|
||||
rows = [project(r,c,plant,now,aliases) for r in records]
|
||||
if any(a['capturedAt'] >= b['capturedAt'] for a,b in zip(rows,rows[1:])):
|
||||
raise ValueError('Batch must be in increasing capture order')
|
||||
stored = duplicate = 0
|
||||
con.execute('BEGIN IMMEDIATE')
|
||||
try:
|
||||
for r in rows:
|
||||
for original, r in zip(records, rows):
|
||||
value = canonical(r); digest = sha256(value.encode()).hexdigest()
|
||||
old = con.execute('SELECT fingerprint FROM planner_observations WHERE plant=? AND dataset=? AND captured_at=?', (plant,c['datasetId'],r['capturedAt'])).fetchone()
|
||||
if old:
|
||||
@@ -211,6 +215,7 @@ def ingest_batch(con, plant, payload, now):
|
||||
duplicate += 1
|
||||
else:
|
||||
con.execute('INSERT INTO planner_observations VALUES(?,?,?,?,?,?)',(plant,c['datasetId'],r['capturedAt'],now,digest,value)); stored += 1
|
||||
mapping_identity.save_origin(con, plant, c, original, r['capturedAt'], now, aliases)
|
||||
con.commit()
|
||||
except Exception:
|
||||
con.rollback(); raise
|
||||
@@ -243,7 +248,7 @@ def physical_value(values, c):
|
||||
return load
|
||||
|
||||
|
||||
def reconstruct(records, c):
|
||||
def reconstruct(records, c, *, projection=None):
|
||||
"""Bounded retrospective estimation, never a real-time feedback signal.
|
||||
|
||||
Missing observations split support. Source timestamps are not refreshed. Small
|
||||
@@ -260,6 +265,11 @@ uncovered portions remain quantified and are never filled with zero.
|
||||
for s in c['sources']:
|
||||
if s['key'] in (sr['rawKey'],sr['scaleKey']): primary[s['key']] = s
|
||||
timing_policy = validate_policy(c.get('historyTimingPolicy'), c['sources'])
|
||||
if projection is not None:
|
||||
configured={s['key']:s for s in c['sources']}
|
||||
if set(projection)!= {'keys','calculate'} or not callable(projection['calculate']) or not projection['keys'] or not set(projection['keys']) <= set(configured):
|
||||
raise ValueError('Invalid internal projection')
|
||||
primary={k:configured[k] for k in projection['keys']}
|
||||
first,last = records[0]['capturedAt'],records[-1]['capturedAt']
|
||||
series = {k:{} for k in primary}; blocks = {k:[] for k in primary}; gaps = []
|
||||
first_observed = {k:{} for k in primary}
|
||||
@@ -313,7 +323,9 @@ uncovered portions remain quantified and are never filled with zero.
|
||||
extended.append(k); known_at = max(known_at, bridge['availableAt'])
|
||||
load = None
|
||||
if usable:
|
||||
try: load = physical_value(vals,c)
|
||||
try:
|
||||
load = physical_value(vals,c) if projection is None else projection['calculate'](vals,c)
|
||||
if not numeric(load,1e9):raise ValueError('Invalid historical projection')
|
||||
except ValueError: usable = False
|
||||
if usable:
|
||||
item['seconds'] += b-a; item['wattSeconds'] += load*(b-a);item['currentGap'] = 0
|
||||
@@ -490,9 +502,10 @@ def pipeline_status(con,plant):
|
||||
out=[]
|
||||
for row in con.execute('SELECT dataset,config FROM planner_data_sets WHERE plant=? ORDER BY dataset',(plant,)):
|
||||
ds=row[0]; c=json.loads(row[1]); state=con.execute('SELECT status,detail FROM planner_pipeline_state WHERE plant=? AND dataset=?',(plant,ds)).fetchone()
|
||||
count=con.execute('SELECT COUNT(*),MIN(captured_at),MAX(captured_at) FROM planner_observations WHERE plant=? AND dataset=?',(plant,c.get('sourceDatasetId',ds))).fetchone()
|
||||
count=con.execute('SELECT COUNT(*),MIN(captured_at),MAX(captured_at),MAX(received_at) FROM planner_observations WHERE plant=? AND dataset=?',(plant,c.get('sourceDatasetId',ds))).fetchone()
|
||||
out.append({'datasetId':ds,'formula':c['formula'],'mappingSha256':c['mappingSha256'],'records':count[0],
|
||||
'firstCapture':iso(count[1]) if count[1] else None,'lastCapture':iso(count[2]) if count[2] else None,
|
||||
'lastReceived':iso(count[3]) if count[3] else None,
|
||||
'status':state[0] if state else 'awaiting_measurements','detail':json.loads(state[1]) if state else {},
|
||||
'minimumCoverage':c['minimumCoverage'],'maximumGapSeconds':c['maximumGapSeconds'],
|
||||
'observationDatasetId':c.get('sourceDatasetId',ds),
|
||||
|
||||
Reference in New Issue
Block a user