"""Offline comparison of complete HTTP-200 UTF-8 snapshots; no network/actions. Hashes validate internal consistency, not provenance. HTML projection is not a browser rendering. Unknown reads never replace a valid comparison baseline. """ import json,hashlib,datetime,difflib,re,sys from html.parser import HTMLParser from urllib.parse import urlsplit MAX=131072 class Projection(HTMLParser): def __init__(self):super().__init__(convert_charrefs=True);self.skip=[];self.parts=[] def handle_starttag(self,t,a): if t in ('script','style','template'):self.skip.append(t) if not self.skip and t in ('p','div','li','br','h1','h2','h3','tr','section'):self.parts.append('\n') def handle_endtag(self,t): if self.skip and self.skip[-1]==t:self.skip.pop() if not self.skip and t in ('p','div','li','h1','h2','h3','tr','section'):self.parts.append('\n') def handle_data(self,s): if not self.skip:self.parts.append(s) def validate(s): if not isinstance(s,dict):raise ValueError('snapshot must be an object') u=urlsplit(s.get('url','')) if u.scheme!='https' or not u.hostname or u.username or u.password or u.query or u.fragment or u.port not in (None,443):raise ValueError('unsupported URL profile') when=datetime.datetime.fromisoformat(s['captured_at']) if when.tzinfo is None or when.utcoffset()!=datetime.timedelta(0):raise ValueError('UTC timestamp required') if s.get('status')!=200 or type(s.get('status')) is not int:raise ValueError('complete HTTP200 required; 304 needs separately validated cache baseline') if s.get('complete') is not True:raise ValueError('incomplete capture') mime=s.get('content_type','').split(';',1)[0].strip().lower() if mime not in ('text/html','text/plain'):raise ValueError('unsupported content type') body=s.get('body'); if not isinstance(body,str) or not body.strip() or len(body.encode('utf-8'))>MAX:raise ValueError('empty/oversize/nontext body') digest=hashlib.sha256(body.encode('utf-8')).hexdigest() if s.get('body_sha256')!=digest:raise ValueError('body/hash mismatch') if mime=='text/html': parser=Projection();parser.feed(body);text='\n'.join(' '.join(line.split()) for line in ''.join(parser.parts).splitlines() if line.strip()) else:text='\n'.join(' '.join(line.split()) for line in body.splitlines() if line.strip()) if not text:raise ValueError('no comparable static text') return when,mime,text def compare(before,after): result={'state':'unknown','baseline_replacement_allowed':False,'limitations':['Internal hashes do not prove capture origin or authenticity.','Static HTML text projection is not browser rendering or semantic importance.','Two observations cannot establish changes between them or continuous uptime.']} try: bt,bm,btext=validate(before) except Exception as e:return dict(result,reason='invalid baseline: '+str(e)) result.update({'baseline_url':before['url'],'baseline_captured_at':before['captured_at'],'baseline_sha256':before['body_sha256']}) try:at,am,atext=validate(after) except Exception as e:return dict(result,reason='invalid observation: '+str(e)) if before['url']!=after['url']:return dict(result,reason='different exact URL') if bm!=am:return dict(result,reason='content-type profile changed') if at<=bt:return dict(result,reason='observation not later than baseline') raw_equal=before['body_sha256']==after['body_sha256'];text_equal=btext==atext state='raw_equal' if raw_equal else 'raw_changed_static_text_equal' if text_equal else 'static_text_changed' diff=list(difflib.unified_diff(btext.splitlines(),atext.splitlines(),fromfile='before-static-text',tofile='after-static-text',lineterm='')) return dict(result,state=state,baseline_replacement_allowed=True,observation_captured_at=after['captured_at'],observation_sha256=after['body_sha256'],seconds_between=(at-bt).total_seconds(),static_text_equal=text_equal,raw_equal=raw_equal,diff=diff[:100],diff_truncated=len(diff)>100) if __name__=='__main__': if len(sys.argv)!=3:raise SystemExit('Usage: python snapshot-compare.py before.json after.json') print(json.dumps(compare(json.load(open(sys.argv[1])),json.load(open(sys.argv[2]))),indent=2))