"""One bounded direct navigation per frozen target, with ordinary redirects; full bodies stay in private runner/data.

Usage: python run.py --manifest public/evidence/public-30-v1.json \
    --output runner/data/wreq-public-30-redirects-20260929
Requires Python >=3.11 and wreq==0.12.3. Output must not already exist.
"""
import argparse
import asyncio
from collections import Counter
from datetime import datetime, timedelta, timezone
import hashlib
from html.parser import HTMLParser
from importlib.metadata import version
import json
import os
from pathlib import Path
import platform
import re
import shlex
import sys
from urllib.parse import urlsplit

import wreq
from wreq import redirect

CONFIG = {
    "client": "wreq.Client", "emulation": "Chrome134", "method": "GET",
    "proxy_mode": "none", "no_proxy": True, "redirects": "follow", "max_redirects": 3,
    "application_retries": 0, "timeout_seconds": 20, "max_body_bytes": 5 * 1024 * 1024,
    "concurrency": 1, "gap_seconds": 2, "cookie_store": False,
    "javascript": False, "login": False, "captcha_solving": False,
}
BLOCK_PHRASES = (
    "verify you are human", "robot or human", "robot check", "access denied",
    "pardon our interruption", "please complete the captcha", "enter the characters",
    "additional verification required", "unusual traffic", "checking your browser",
    "enable javascript and cookies to continue", "just a moment",
)
GATE_PHRASES = ("log in to continue", "sign in to continue", "sign in to linkedin", "consent required", "before you continue")


def utc():
    return datetime.now(timezone.utc).isoformat()


def sha(data):
    return hashlib.sha256(data).hexdigest()


class PageText(HTMLParser):
    """Source-text approximation, not rendered visibility or structured extraction."""
    SKIP = {"head", "script", "style", "template", "noscript"}

    def __init__(self):
        super().__init__(convert_charrefs=True)
        self.hidden = []
        self.parts = []

    def handle_starttag(self, tag, attrs):
        if tag in self.SKIP:
            self.hidden.append(tag)

    def handle_endtag(self, tag):
        if tag in self.hidden:
            self.hidden = self.hidden[:self.hidden.index(tag)]

    def handle_data(self, data):
        if not self.hidden:
            self.parts.append(data)


def evaluate(body, target):
    parser = PageText()
    parser.feed(body.decode("utf-8", errors="replace"))
    text = " ".join(" ".join(parser.parts).split())
    lower = text.lower()
    checks = [
        {"alternatives": group, "matched": [m for m in group if m.lower() in lower]}
        for group in target["required_text_groups"]
    ]
    regex_checks = [
        {**check, "matches": len(re.findall(check["pattern"], text, flags=re.I))}
        for check in target["text_regex_checks"]
    ]
    expected = (
        len(text) >= target["min_text_chars"]
        and all(check["matched"] for check in checks)
        and all(check["matches"] >= check["min_matches"] for check in regex_checks)
    )
    return {
        "text_chars": len(text), "required_text_checks": checks, "text_regex_checks": regex_checks,
        "min_text_chars": target["min_text_chars"], "has_required_content": bool(expected),
        "soft_block_checks": {"phrases_checked": list(BLOCK_PHRASES), "matched": [p for p in BLOCK_PHRASES if p in lower]},
        "gate_checks": {"phrases_checked": list(GATE_PHRASES), "matched": [p for p in GATE_PHRASES if p in lower]},
    }


def classify(row):
    if row["error"]:
        return row["error"]["classification"]
    status = row["status"]
    if 300 <= status < 400:
        return "unresolved_redirect"
    if status == 429:
        return "http_rate_limited"
    if status in (401, 403, 999):
        return "http_access_denied"
    if status >= 500:
        return "http_server_error"
    if status >= 400:
        return "http_client_error"
    if row["checks"]["soft_block_checks"]["matched"]:
        return "soft_block"
    if not row["checks"]["has_required_content"]:
        return "login_or_consent_gate" if row["checks"]["gate_checks"]["matched"] else "missing_required_content"
    return "usable" if status == 200 else "unexpected_http_status"


async def attempt(target, output):
    row = {"target_id": target["id"], "site": target["name"], "url": target["url"],
           "category": target["category"], "expected_required_content": target["expected_required_content"],
           "started_at": utc(), "status": None, "original_url": target["url"], "final_url": None, "response_url": None, "redirect_history": [], "redirect_outcome": "unknown", "error": None,
           "body_complete": False, "raw_response_sha256": None, "raw_response_private_path": None}
    body = bytearray()
    response = None
    client = wreq.Client(emulation=wreq.Emulation.Chrome134, no_proxy=True,
                         redirect=redirect.Policy.limited(CONFIG["max_redirects"]), timeout=timedelta(seconds=CONFIG["timeout_seconds"]),
                         cookie_store=False)
    start = asyncio.get_running_loop().time()
    try:
        async with asyncio.timeout(CONFIG["timeout_seconds"]):
            response = await client.get(target["url"], no_proxy=True)
            row.update(status=response.status.as_int(), response_url=str(response.url), final_url=str(response.url),
                       redirect_history=[{"status": hop.status, "url": hop.url, "previous": hop.previous} for hop in response.history])
            row["redirect_outcome"] = "followed" if row["redirect_history"] else "none"
            async with response.stream() as stream:
                async for chunk in stream:
                    if not isinstance(chunk, bytes):  # trailers are not body bytes
                        continue
                    room = CONFIG["max_body_bytes"] - len(body)
                    body.extend(chunk[:room])
                    if len(chunk) > room:
                        raise ValueError("Response body exceeded the 5 MiB cap")
            row["body_complete"] = True
    except Exception as exc:
        message = str(exc)
        category = "redirect_limit_or_loop" if isinstance(exc, wreq.exceptions.RedirectError) else "body_limit" if isinstance(exc, ValueError) else "timeout" if isinstance(exc, TimeoutError) or "timed out" in message.lower() else "transport_error"
        if response is None:
            row["redirect_outcome"] = "limit_or_loop_error" if category == "redirect_limit_or_loop" else "no_response"
        row["error"] = {"type": type(exc).__name__, "message": message, "classification": category}
    finally:
        row["latency_seconds"] = round(asyncio.get_running_loop().time() - start, 3)
        if response is not None:
            await response.close()
        client.close()
    row.update(finished_at=utc(), response_bytes=len(body), checks=evaluate(bytes(body), target))
    if response is not None:
        private_body = output / f"{target['id']}.body"
        private_body.write_bytes(body)
        row.update(raw_response_private_path=str(private_body), raw_response_sha256=sha(body))
    if row["status"] is not None and 300 <= row["status"] < 400:
        row["redirect_outcome"] = "unresolved"
    row["classification"] = classify(row)
    row["usable_content"] = row["classification"] == "usable" and row["body_complete"]
    (output / f"{target['id']}.json").write_text(json.dumps(row, indent=2) + "\n")
    print(f"{target['id']}: HTTP {row['status']} | {row['classification']} | {len(body)} bytes | {row['latency_seconds']}s", flush=True)
    return row


async def main(args):
    manifest_bytes = args.manifest.read_bytes()
    manifest = json.loads(manifest_bytes)
    targets = manifest["targets"]
    hosts = {urlsplit(t["url"]).hostname for t in targets}
    assert len(targets) == len(hosts) == len({t['id'] for t in targets}) == 30
    assert all(urlsplit(t["url"]).scheme == "https" for t in targets)
    assert version("wreq") == "0.12.3", "Use the pinned wreq version"
    assert all(CONFIG[k] == v for k, v in manifest["request_policy"].items()), "Use the shared manifest request policy"
    os.umask(0o077)
    args.output.mkdir(parents=True, exist_ok=False, mode=0o700)
    run = {"run_id": args.output.name, "tool": "wreq", "tool_version": version("wreq"),
           "python_version": platform.python_version(), "platform": platform.platform(),
           "network": "local host direct outbound; no configured/system proxy; public logged-out requests",
           "manifest_id": manifest["id"], "manifest_sha256": sha(manifest_bytes),
           "script_sha256": sha(Path(__file__).read_bytes()), "config": CONFIG,
           "executed_command": shlex.join([sys.executable, *sys.argv]), "runtime_executable": sys.executable,
           "request_policy_revision": manifest["policy_revision"],
           "body_hash_basis": "Client-decompressed body bytes as saved, not re-encoded text or wire bytes; null if no response. Partial bodies explicitly marked.",
           "started_at": utc(), "attempts": []}
    for index, target in enumerate(targets):
        if index:
            await asyncio.sleep(CONFIG["gap_seconds"])
        run["attempts"].append(await attempt(target, args.output))
        (args.output / "results.json").write_text(json.dumps(run, ensure_ascii=False, indent=2) + "\n")
    run.update(finished_at=utc(), summary={"attempted_distinct_sites": len(run["attempts"]),
               "usable_results": sum(a["usable_content"] for a in run["attempts"]),
               "classifications": dict(Counter(a["classification"] for a in run["attempts"]))})
    (args.output / "results.json").write_text(json.dumps(run, ensure_ascii=False, indent=2) + "\n")
    print(json.dumps(run["summary"]), flush=True)


if __name__ == "__main__":
    parser = argparse.ArgumentParser(description=__doc__)
    parser.add_argument("--manifest", type=Path, required=True)
    parser.add_argument("--output", type=Path, required=True)
    asyncio.run(main(parser.parse_args()))
