"""Offline capture inspection; never sends target requests.
python publish.py --private-output runner/data/scrapy-public-30-20260930 --canary runner/data/scrapy-canary-20260930
"""
import argparse,ast,hashlib,json
from importlib.metadata import version
from pathlib import Path
from datetime import datetime,timezone
from urllib.parse import urljoin
HERE=Path(__file__).resolve().parent
sha=lambda data:hashlib.sha256(data).hexdigest()
p=argparse.ArgumentParser(description=__doc__);p.add_argument('--private-output',type=Path,required=True);p.add_argument('--canary',type=Path,required=True);args=p.parse_args()
manifest=json.loads((HERE.parent/'public-30-v1.json').read_text())
def nodes(path):
 tree=ast.parse(path.read_text());return [n for n in tree.body if getattr(n,'name',None) in ['PageText','evaluate','result_class'] or isinstance(n,ast.Assign) and any(isinstance(t,ast.Name) and t.id in ['BLOCK_PHRASES','GATE_PHRASES'] for t in n.targets)]
reference=nodes(HERE.parent/'scrapling-public-30-20260929/run.py')
assert [ast.dump(n,include_attributes=False) for n in nodes(HERE/'run.py')]==[ast.dump(n,include_attributes=False) for n in reference]
ns={};exec('from html.parser import HTMLParser\nimport re',ns);exec(compile(ast.Module(body=reference,type_ignores=[]),'unchanged-checker','exec'),ns)

def verify_attempt(a,t,folder):
 bodypath=folder/(a['target_id']+'.body');body=bodypath.read_bytes() if bodypath.exists() else b''
 assert ns['evaluate'](body,t)==a['checks'] and ns['result_class'](a)==a['classification']
 assert a['raw_response_sha256']==(sha(body) if bodypath.exists() else None) and a['response_bytes']==len(body)
 assert a['usable_content']==(a['classification']=='usable') and a['empty_content']==(a['checks']['text_chars']==0)
 assert a['attempt_process_exit']==0 and a['cleanup_surviving_owned_pids']==[] and a['request_started_at']
 readings=json.loads((folder/(a['target_id']+'-pss.json')).read_text())
 assert len(readings)==a['pss_samples'] and max((s['pss_mb'] for s in readings),default=None)==a['peak_sampled_tree_pss_mb']
 assert all(s['host_available_mb']>=3072 and s['pss_mb']<=1024 for s in readings)
 log=folder/(a['target_id']+'-transport.json');assert sha(log.read_bytes())==a['transport_log_sha256'];transport=json.loads(log.read_text())
 outgoing=[{'url':r['url'],'method':r['method'],'headers':[h for h in r['headers'] if h['name'].lower() not in ['cookie','authorization','proxy-authorization']]} for r in transport['requests']]
 assert outgoing==a['observed_request_headers']
 assert len(outgoing)==a['downloader_request_count'] and 1<=len(outgoing)<=4
 responses=transport['responses']
 assert [{k:r[k] for k in ['url','status','protocol','wire_body_bytes','wire_body_sha256']} for r in responses]==a['document_responses']
 for r in responses:
  wire=(folder/r['wire_file']).read_bytes();assert len(wire)==r['wire_body_bytes'] and sha(wire)==r['wire_body_sha256']
 if responses:
  assert responses[-1]['status']==a['status'] and responses[-1]['url']==a['final_url']
  hops=[]
  for i,r in enumerate(responses[:-1]):
   assert r['status'] in [301,302,303,307,308]
   location=next(h['values'][0] for h in r['headers'] if h['name'].lower()=='location')
   assert urljoin(r['url'],location)==responses[i+1]['url']
   hops.append({'previous':r['url'],'url':responses[i+1]['url'],'status':r['status']})
  assert hops==a['redirect_history']
 assert sha((folder/(a['target_id']+'-driver.log')).read_bytes())==a['driver_log_sha256']
 for block in outgoing:
  assert not any(h['name'].lower() in ['cookie','authorization','proxy-authorization'] for h in block['headers'])
  ua=next(h['values'][0] for h in block['headers'] if h['name'].lower()=='user-agent');assert ua=='Scrapy/2.19.0 (+https://scrapy.org)'
 assert a['effective_settings']['ROBOTSTXT_OBEY']==False and a['effective_settings']['COOKIES_ENABLED']==False and a['effective_settings']['RETRY_ENABLED']==False
 assert a['effective_settings']['HTTPERROR_ALLOW_ALL']==True and a['effective_settings']['REDIRECT_MAX_TIMES']==3 and a['effective_settings']['DOWNLOAD_VERIFY_CERTIFICATES']==True
 for k in ['raw_response_private_path','screenshot_private_path']:a.pop(k,None)
 a['request_header_observation_basis']='Downloader middleware before the native handler; implicit Host/connection/transport headers are not a complete wire dump. Cookies/auth values excluded.'
 a['inspection_note']=('Complete client-decompressed Spider response met every unchanged word group, listing pattern and minimum text. No rendered visibility or complete field extraction was verified.' if a['usable_content'] else 'Frozen response-source checks matched soft-block phrases: '+', '.join(a['checks']['soft_block_checks']['matched']) if a['classification']=='soft_block' else 'HTTP 200 had no checked source text; empty checked text is not usable content.' if a['status']==200 and a['empty_content'] else 'Saved response was rechecked with unchanged rules: '+a['classification'].replace('_',' ')+'.')
 if a['target_id']=='walmart':a['inspection_note']+=' Saved checked text says the requested URL was rejected. This observed rejection is not in the frozen soft-block phrases; classification remains missing_required_content.'
 if a['target_id']=='remoteok':a['inspection_note']+=' Three redirects were followed; the fourth response was HTTP 302. Scrapy raised IgnoreRequest: max redirections reached. The saved final early body is empty and marked incomplete.'
 if a['target_id']=='target':a['inspection_note']+=' Saved source text contains header/footer navigation without the required laptop listings; no rendered page was inspected.'
 if a['target_id']=='bestbuy':a['inspection_note']+=' Saved checked text asks the visitor to choose a country; it does not meet the frozen listing rules.'
 if a['target_id']=='amazon' and b'RON' in body and all(c['matches']==0 for c in a['checks']['text_regex_checks']):a['inspection_note']+=' Source contained RON price labels while the frozen pattern accepts only $, £ or €; no currency rule changed.'
 if a['target_id']=='example':a['inspection_note']+=' The frozen source parser/phrase rule is unchanged; only its actual checked text and matches are reported, without a browser inspection.'
 if a['target_id']=='webscrapingdev':a['inspection_note']+=' This HTTP-only pass did not execute JavaScript to populate the reviews route.'
 return readings
run=json.loads((args.private_output/'results.json').read_text())
assert len(run['attempts'])==30 and run.get('finished_at') and not run.get('blocker')
assert [(a['target_id'],a['original_url']) for a in run['attempts']]==[(t['id'],t['url']) for t in manifest['targets']]
assert run['script_sha256']==sha((HERE/'run.py').read_bytes()) and run['manifest_sha256']==sha((HERE.parent/'public-30-v1.json').read_bytes())
assert run['requirements_sha256']==sha((HERE/'requirements.txt').read_bytes())
pins=dict(line.split('==') for line in (HERE/'requirements.txt').read_text().strip().splitlines())
assert all(version(name)==pin for name,pin in pins.items())
run['installed_packages']=pins
run['additional_versioned_verification']=json.loads((HERE/'documentation.json').read_text())['twisted_connection_recovery']
samples={a['target_id']:verify_attempt(a,t,args.private_output) for a,t in zip(run['attempts'],manifest['targets'])}
canary=json.loads((args.canary/'results.json').read_text());assert len(canary['attempts'])==1 and canary['script_sha256']==run['script_sha256'] and canary['config']==run['config']
c=canary['attempts'][0];t=next(t for t in manifest['targets'] if t['id']==c['target_id']);verify_attempt(c,t,args.canary)
startup={'separate_public_canary':c,'canary_started_at':canary['started_at'],'canary_finished_at':canary['finished_at'],'executed_command':canary['executed_command'],'runner_sha256':run['script_sha256'],'config_same_as_batch':True,'counts_relation':'One complete HTTP 200 canary body failed frozen reviews checks. It establishes returned downloader plumbing, not usable review extraction. Excluded from the full 30-target batch; 31 explicit root Spider Requests total, plus ordinary redirects. No local smoke, browser or failed setup navigation.'}
(HERE/'startup-checks.json').write_text(json.dumps(startup,indent=2)+'\n')
run['summary']['http_200_soft_blocks']=sum(a['status']==200 and a['classification']=='soft_block' for a in run['attempts'])
run['summary']['http_200_empty_content']=sum(a['status']==200 and a['empty_content'] for a in run['attempts'])
run['policy_differences']=[note.replace('No retries, proxy,', 'Scrapy retry middleware disabled; stock Twisted cached-connection recovery not separately disabled or instrumented. No proxy,') for note in run['policy_differences']]
run['downloader_count_basis']='Observed downloader middleware Requests, including ordinary redirects; not an independently captured count of wire attempts. Stock Twisted cached-connection recovery may happen below this observer.'
run['summary']['downloader_requests_including_redirects']=sum(a['downloader_request_count'] for a in run['attempts'])
run['publication_validation']={'checked_at':datetime.now(timezone.utc).isoformat(),'unchanged_validator_ast_verified':True,'body_hashes_and_checks_verified':30,'pss_peaks_verified':30,'private_transport_logs_and_wire_hashes_verified':30,'response_status_and_redirects_verified':30,'driver_log_hashes_verified':30,'additional_target_requests':0,'surviving_owned_processes':0}
run['request_header_observation_basis']=c['request_header_observation_basis']
run['startup_checks_url']='/evidence/scrapy-public-30-20260929/startup-checks.json';run['canary_relation']=startup['counts_relation']
(HERE/'pss-samples.json').write_text(json.dumps(samples,indent=2)+'\n');run['pss_samples_sha256']=sha((HERE/'pss-samples.json').read_bytes())
(HERE/'results.json').write_text(json.dumps(run,indent=2)+'\n');print('Rechecked private captures/logs/PSS and canary:',run['summary'])
