# MIT License # # Copyright (c) 2026 TableProof contributors # # Permission is hereby granted, free of charge, to any person obtaining a copy # of this software and associated documentation files (the "Software"), to deal # in the Software without restriction, including without limitation the rights # to use, copy, modify, merge, publish, distribute, sublicense, and/or sell # copies of the Software, and to permit persons to whom the Software is # furnished to do so, subject to the following conditions: # # The above copyright notice and this permission notice shall be included in all # copies or substantial portions of the Software. # # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE # SOFTWARE. # """Original offline audit of explicitly supplied scheduled-run evidence. Python 3.10+.""" import argparse from collections import Counter from datetime import datetime, timezone from decimal import Decimal import json import re from pathlib import Path MAX_BYTES = 1024 * 1024 STATUSES = {'completed', 'failed', 'canceled', 'running'} OUTCOMES = {'useful', 'no_change', 'unknown'} def unique_object(pairs): result = {} for key, value in pairs: if key in result: raise ValueError('duplicate JSON key') result[key] = value return result def stamp(value): if not isinstance(value, str) or not re.fullmatch(r'\d{4}-\d\d-\d\dT\d\d:\d\d:\d\d(?:\.\d{1,6})?(?:Z|[+-]\d\d:\d\d)', value): raise ValueError('timestamps require ISO date/time with seconds and explicit offset') result = datetime.fromisoformat(value.replace('Z', '+00:00')) return result.astimezone(timezone.utc) def identifier(value): if not isinstance(value, str) or not re.fullmatch(r'[A-Za-z0-9_.:-]{1,96}', value): raise ValueError('run/job IDs must be 1–96 ASCII identifier characters') return value def audit(data): if not isinstance(data, dict) or set(data) != {'window', 'runs'}: raise ValueError('expected exactly window and runs') window = data['window'] if not isinstance(window, dict) or set(window) != {'start', 'end'}: raise ValueError('expected window.start and window.end') start, end = stamp(window['start']), stamp(window['end']) if start >= end: raise ValueError('window start must precede end') runs = data['runs'] if not isinstance(runs, list) or len(runs) > 10000: raise ValueError('runs must be a list of at most 10000 records') seen, selected = set(), [] fields = {'run_id', 'job_id', 'started_at', 'status', 'model_called', 'notified', 'outcome', 'cost_usd'} for run in runs: if not isinstance(run, dict) or set(run) != fields: raise ValueError('each run must contain the eight documented fields') run_id, job_id = identifier(run['run_id']), identifier(run['job_id']) if run_id in seen: raise ValueError('duplicate run_id; supply exactly one current record per run') seen.add(run_id) when = stamp(run['started_at']) if not isinstance(run['status'], str) or not isinstance(run['outcome'], str) or run['status'] not in STATUSES or run['outcome'] not in OUTCOMES: raise ValueError('unsupported status/outcome') for field in ('model_called', 'notified'): if run[field] is not None and type(run[field]) is not bool: raise ValueError('model_called/notified require boolean or null') cost = run['cost_usd'] if cost is not None and (not isinstance(cost, str) or not re.fullmatch(r'(?:0|[1-9]\d{0,8})(?:\.\d{1,6})?', cost)): raise ValueError('cost_usd requires a nonnegative decimal string or null') if start <= when < end: selected.append(dict(run, job_id=job_id)) def summarize(rows): complete = [r for r in rows if r['status'] == 'completed'] known = [r for r in rows if r['cost_usd'] is not None] notified = sum(r['notified'] is True for r in complete) notification_known = sum(r['notified'] is not None for r in complete) return { 'runs': len(rows), 'statuses': dict(sorted(Counter(r['status'] for r in rows).items())), 'completed_runs': len(complete), 'completed_notified': notified, 'completed_silent': sum(r['notified'] is False for r in complete), 'completed_notification_unknown': len(complete) - notification_known, 'completed_notification_rate_known_denominator': ( format(Decimal(notified) / Decimal(notification_known), '.6f') if notification_known else None), 'completed_outcomes': dict(sorted(Counter(r['outcome'] for r in complete).items())), 'model_called': sum(r['model_called'] is True for r in rows), 'model_not_called': sum(r['model_called'] is False for r in rows), 'model_call_unknown': sum(r['model_called'] is None for r in rows), 'reported_cost_usd_known_subtotal': format(sum((Decimal(r['cost_usd']) for r in known), Decimal(0)), '.6f'), 'reported_cost_missing_runs': len(rows) - len(known), 'cost_coverage_complete_for_supplied_runs': len(known) == len(rows), } grouped = {} for run in selected: grouped.setdefault(run['job_id'], []).append(run) return { 'schema': 'tableproof-run-outcome-audit-1', 'window_utc': {'start': start.isoformat(), 'end_exclusive': end.isoformat()}, 'supplied_runs': len(runs), 'excluded_outside_start_window': len(runs) - len(selected), 'summary': summarize(selected), 'jobs': {job: summarize(grouped[job]) for job in sorted(grouped)}, 'limits': ['Supplied records only; no scheduler, native Dots or historical completeness verification.', 'Run status, model calls, outcomes and costs are supplied labels, not independently verified facts.', 'A silent completed check can be valuable. Notifications do not prove usefulness or delivery.', 'Costs are caller-reported USD labels, not invoices, receipts, savings or earned revenue.', 'No schedule is changed; safe frequency reduction and missed-event risk are not established.'], } def main(): parser = argparse.ArgumentParser(description=__doc__) parser.add_argument('input', help='Explicit local UTF-8 JSON input; do not include secrets or message content') args = parser.parse_args() try: with Path(args.input).open('rb') as handle: raw = handle.read(MAX_BYTES + 1) if len(raw) > MAX_BYTES: raise ValueError('input exceeds 1 MiB') data = json.loads(raw.decode('utf-8'), object_pairs_hook=unique_object, parse_constant=lambda _: (_ for _ in ()).throw(ValueError('nonfinite JSON number'))) print(json.dumps(audit(data), indent=2)) except (OSError, UnicodeError, ValueError, TypeError) as error: parser.exit(2, f'Invalid audit input: {type(error).__name__}\n') if __name__ == '__main__': main()