"""Offline capture inspection only; does not send requests.
python publish.py --private-output runner/data/curl-cffi-public-30-20260930
"""
import argparse,ast,hashlib,json
from pathlib import Path
from datetime import datetime,timezone
from urllib.parse import urljoin
HERE=Path(__file__).resolve().parent
sha=lambda data:hashlib.sha256(data).hexdigest()
p=argparse.ArgumentParser(description=__doc__);p.add_argument('--private-output',type=Path,required=True);p.add_argument('--canary',type=Path,required=True);args=p.parse_args()
manifest=json.loads((HERE.parent/'public-30-v1.json').read_text())
def nodes(path):
 tree=ast.parse(path.read_text());return [n for n in tree.body if getattr(n,'name',None) in ['PageText','evaluate','result_class'] or isinstance(n,ast.Assign) and any(isinstance(t,ast.Name) and t.id in ['BLOCK_PHRASES','GATE_PHRASES'] for t in n.targets)]
reference=nodes(HERE.parent/'scrapling-public-30-20260929/run.py')
assert [ast.dump(n,include_attributes=False) for n in nodes(HERE/'run.py')]==[ast.dump(n,include_attributes=False) for n in reference]
ns={};exec('from html.parser import HTMLParser\nimport re',ns);exec(compile(ast.Module(body=reference,type_ignores=[]),'frozen-checker','exec'),ns)

def outgoing(events):
 result=[]
 for e in events:
  if e['kind']!='request_headers':continue
  lines=e['data'].splitlines()
  result.append({'request_line':lines[0] if lines else None,'headers':[{'name':s.split(':',1)[0],'value':s.split(':',1)[1].strip()} for s in lines[1:] if ':' in s and s.split(':',1)[0].lower() not in ['cookie','authorization','proxy-authorization']],'cookie_values':'private/not published'})
 return result

def response_blocks(events):
 blocks=[];current=None
 for e in events:
  if e['kind']!='response_header':continue
  for line in e['data'].splitlines():
   if line.startswith('HTTP/'):
    current={'status':int(line.split()[1]),'headers':{}};blocks.append(current)
   elif current is not None and ':' in line:
    k,v=line.split(':',1);current['headers'][k.lower()]=v.strip()
 return blocks

def verify_attempt(a,t,folder):
 bodypath=folder/(a['target_id']+'.body');body=bodypath.read_bytes() if bodypath.exists() else b''
 assert ns['evaluate'](body,t)==a['checks'] and ns['result_class'](a)==a['classification']
 assert a['raw_response_sha256']==(sha(body) if bodypath.exists() else None) and a['response_bytes']==len(body)
 assert a['usable_content']==(a['classification']=='usable') and a['empty_content']==(a['checks']['text_chars']==0)
 assert a['attempt_process_exit']==0 and a['cleanup_surviving_owned_pids']==[] and a['request_started_at']
 readings=json.loads((folder/(a['target_id']+'-pss.json')).read_text())
 assert len(readings)==a['pss_samples'] and max((s['pss_mb'] for s in readings),default=None)==a['peak_sampled_tree_pss_mb']
 assert all(s['host_available_mb']>=3072 and s['pss_mb']<=1024 for s in readings)
 log=folder/(a['target_id']+'-transport.json');assert sha(log.read_bytes())==a['transport_log_sha256'];transport=json.loads(log.read_text())
 assert outgoing(transport['header_events'])==a['observed_request_headers']
 assert (transport['status'],transport['final_url'],transport['redirect_history'])==(a['status'],a['final_url'],a['redirect_history'])
 blocks=[b for b in response_blocks(transport['header_events']) if b['status']>=200]
 if blocks:
  assert blocks[-1]['status']==a['status']
  original=a['original_url'];hops=[]
  for b in blocks[:-1]:
   assert 300<=b['status']<400 and b['headers'].get('location')
   final=urljoin(original,b['headers']['location']);hops.append({'previous':original,'url':final,'status':b['status']});original=final
  assert hops==a['redirect_history']
 assert sha((folder/(a['target_id']+'-driver.log')).read_bytes())==a['driver_log_sha256']
 for block in a['observed_request_headers']:
  assert not any(h['name'].lower() in ['cookie','authorization','proxy-authorization','referer'] for h in block['headers'])
 for k in ['raw_response_private_path','screenshot_private_path']:a.pop(k,None)
 a['inspection_note']=('Complete client-decompressed HTTP body met all unchanged word groups, listing patterns and minimum text. No JavaScript, rendered visibility or complete field extraction was verified.' if a['usable_content'] else 'Frozen body-source checks matched soft-block phrases: '+', '.join(a['checks']['soft_block_checks']['matched']) if a['classification']=='soft_block' else 'HTTP 200 body had no checked source text; empty checked content is not a usable 200.' if a['status']==200 and a['empty_content'] else 'Original client-decompressed body was rechecked with unchanged rules: '+a['classification'].replace('_',' ')+'.')
 if a['target_id']=='amazon':
  assert b'RON' in body and all(c['matches']==0 for c in a['checks']['text_regex_checks'])
  a['inspection_note']+=' Saved source included RON price labels; the frozen listing-price pattern accepts only $, £ or € and found zero matches. The currency rule stays unchanged.'
 if a['target_id']=='example':a['inspection_note']+=' Checked source text contained documentation-example wording but not the required Example Domain phrase. The frozen parser and word rule stay unchanged.'
 if a['target_id']=='bestbuy':a['inspection_note']+=' Source text was a country-choice page, not the required laptop listings. No choice or link was followed.'
 if a['target_id']=='target':a['inspection_note']+=' Checked source text had header/footer content without required laptop listings; no rendered page or cause was inspected.'
 if a['target_id']=='webscrapingdev':a['inspection_note']+=' The reviews route returns source HTML in this HTTP mode; no browser/JavaScript was run to populate reviews.'
 return readings
run=json.loads((args.private_output/'results.json').read_text())
assert len(run['attempts'])==30 and run.get('finished_at') and not run.get('blocker')
assert [(a['target_id'],a['original_url']) for a in run['attempts']]==[(t['id'],t['url']) for t in manifest['targets']]
assert run['script_sha256']==sha((HERE/'run.py').read_bytes()) and run['manifest_sha256']==sha((HERE.parent/'public-30-v1.json').read_bytes())
assert run['requirements_sha256']==sha((HERE/'requirements.txt').read_bytes())
samples={a['target_id']:verify_attempt(a,t,args.private_output) for a,t in zip(run['attempts'],manifest['targets'])}
canary=json.loads((args.canary/'results.json').read_text());assert len(canary['attempts'])==1 and canary['script_sha256']==run['script_sha256'] and canary['config']==run['config']
c=canary['attempts'][0];t=next(t for t in manifest['targets'] if t['id']==c['target_id']);verify_attempt(c,t,args.canary)
startup={'separate_public_canary':c,'canary_started_at':canary['started_at'],'canary_finished_at':canary['finished_at'],'executed_command':canary['executed_command'],'runner_sha256':run['script_sha256'],'config_same_as_batch':True,'counts_relation':'One complete HTTP 200 canary body failed frozen reviews checks. It establishes a returned request, not usable review extraction. Excluded from the full 30-target batch; 31 explicit public GET calls total, plus redirects. No browser, local smoke or failed setup request.'}
(HERE/'startup-checks.json').write_text(json.dumps(startup,indent=2)+'\n')
run['summary']['http_200_soft_blocks']=sum(a['status']==200 and a['classification']=='soft_block' for a in run['attempts'])
run['summary']['http_200_empty_content']=sum(a['status']==200 and a['empty_content'] for a in run['attempts'])
run['publication_validation']={'checked_at':datetime.now(timezone.utc).isoformat(),'unchanged_validator_ast_verified':True,'raw_body_hashes_and_checks_verified':30,'pss_peaks_verified':30,'private_transport_log_hashes_and_headers_verified':30,'main_response_status_and_redirect_headers_verified':30,'driver_log_hashes_verified':30,'additional_target_requests':0,'surviving_owned_processes':0}
run['startup_checks_url']='/evidence/curl-cffi-public-30-20260929/startup-checks.json'
run['canary_relation']=startup['counts_relation']
(HERE/'pss-samples.json').write_text(json.dumps(samples,indent=2)+'\n');run['pss_samples_sha256']=sha((HERE/'pss-samples.json').read_bytes())
(HERE/'results.json').write_text(json.dumps(run,indent=2)+'\n');print('Rechecked all 30 private bodies/logs/PSS and separate canary:',run['summary'])
