"""One fresh local Patchright browser per frozen URL; private captures only.
Requires Python >=3.11 and patchright==1.63.0 with an existing matching Chromium.
"""
import argparse
import asyncio
from collections import Counter
from datetime import datetime, timezone
import hashlib
from html.parser import HTMLParser
from importlib.metadata import version
import json
import os
from pathlib import Path
import platform
import re
import shlex
import signal
import subprocess
import sys
import time
from urllib.parse import urlsplit
from patchright.async_api import async_playwright

CONFIG = {
 "proxy_mode": "none", "login": False, "captcha_solving": False,
 "concurrency": 1, "gap_seconds": 2, "application_retries": 0,
 "headless": True, "fresh_browser_per_target": True, "accept_downloads": False,
 "navigation_timeout_seconds": 20, "attempt_timeout_seconds": 40,
 "wait_until": "domcontentloaded", "settle_seconds": 2,
 "max_capture_bytes": 5242880, "network_response_byte_cap": None,
 "redirects": "follow", "max_redirects": 20,
 "redirect_cap_basis": "Chromium URLRequest::kMaxRedirects at Browser.getVersion revision 971a7443b0c9b0a9b2860529b33331b76077ec62",
 "resources": "Normal Chromium images/styles/scripts/iframes/workers/subrequests; no resource interception",
 "cookies": "fresh in-memory context; no imported login or persistent profile",
 "launch_args": ["--no-proxy-server"], "pss_sample_interval_seconds": 0.1,
 "pss_scope": "Isolated attempt Python process and descendants, including Patchright driver and Chromium; excludes batch parent",
 "min_host_available_mb": 3072, "max_attempt_tree_pss_mb": 4096,
}

BLOCK_PHRASES = (
    "verify you are human", "robot or human", "robot check", "access denied",
    "pardon our interruption", "please complete the captcha", "enter the characters",
    "additional verification required", "unusual traffic", "checking your browser",
    "enable javascript and cookies to continue", "just a moment",
)
GATE_PHRASES = ("log in to continue", "sign in to continue", "sign in to linkedin", "consent required", "before you continue")


def utc():
    return datetime.now(timezone.utc).isoformat()


def sha(data):
    return hashlib.sha256(data).hexdigest()


class PageText(HTMLParser):
    """Source-text approximation, not rendered visibility or structured extraction."""
    SKIP = {"head", "script", "style", "template", "noscript"}

    def __init__(self):
        super().__init__(convert_charrefs=True)
        self.hidden = []
        self.parts = []

    def handle_starttag(self, tag, attrs):
        if tag in self.SKIP:
            self.hidden.append(tag)

    def handle_endtag(self, tag):
        if tag in self.hidden:
            self.hidden = self.hidden[:self.hidden.index(tag)]

    def handle_data(self, data):
        if not self.hidden:
            self.parts.append(data)


def evaluate(body, target):
    parser = PageText()
    parser.feed(body.decode("utf-8", errors="replace"))
    text = " ".join(" ".join(parser.parts).split())
    lower = text.lower()
    checks = [
        {"alternatives": group, "matched": [m for m in group if m.lower() in lower]}
        for group in target["required_text_groups"]
    ]
    regex_checks = [
        {**check, "matches": len(re.findall(check["pattern"], text, flags=re.I))}
        for check in target["text_regex_checks"]
    ]
    expected = (
        len(text) >= target["min_text_chars"]
        and all(check["matched"] for check in checks)
        and all(check["matches"] >= check["min_matches"] for check in regex_checks)
    )
    return {
        "text_chars": len(text), "required_text_checks": checks, "text_regex_checks": regex_checks,
        "min_text_chars": target["min_text_chars"], "has_required_content": bool(expected),
        "soft_block_checks": {"phrases_checked": list(BLOCK_PHRASES), "matched": [p for p in BLOCK_PHRASES if p in lower]},
        "gate_checks": {"phrases_checked": list(GATE_PHRASES), "matched": [p for p in GATE_PHRASES if p in lower]},
    }



def result_class(row):
    if row["error"]:
        return row["error"]["classification"]
    if row["navigation_error"]:
        return "navigation_error"
    if row["capture_error"]:
        return "capture_error"
    if row["status"] is None:
        return "status_unobserved"
    if row["status"] in (401, 403, 999):
        return "http_access_denied"
    if row["status"] == 429:
        return "http_rate_limited"
    if row["status"] >= 500:
        return "http_server_error"
    if row["status"] >= 400:
        return "http_client_error"
    if 300 <= row["status"] < 400:
        return "unresolved_redirect"
    if row["checks"]["soft_block_checks"]["matched"]:
        return "soft_block"
    if not row["checks"]["has_required_content"]:
        return "login_or_consent_gate" if row["checks"]["gate_checks"]["matched"] else "missing_required_content"
    return "usable" if row["status"] == 200 and row["capture_complete"] else "incomplete_capture"


def env_direct():
    env = os.environ.copy()
    for key in list(env):
        if key.lower() in ("http_proxy", "https_proxy", "all_proxy"):
            env.pop(key)
    env.update(PYTHONDONTWRITEBYTECODE="1")
    return env


def memory():
    readings = {line.split(":")[0]: int(line.split()[1]) / 1024
                for line in Path("/proc/meminfo").read_text().splitlines() if line.split(":")[0] in ("MemTotal", "MemAvailable", "SwapTotal", "SwapFree")}
    return {**readings, "unit": "MiB", "measured_at": utc()}


def base_row(target):
    return {"target_id": target["id"], "site": target["name"], "category": target["category"],
            "url": target["url"], "original_url": target["url"], "final_url": None,
            "expected_required_content": target["expected_required_content"], "started_at": utc(),
            "navigation_started_at": None, "status": None, "navigation_outcome": "not_started",
            "redirect_history": [], "redirect_outcome": "unobserved", "document_responses": [],
            "browser_version": None, "browser_cdp_version": None, "error": None,
            "navigation_error": None, "capture_error": None, "screenshot_error": None,
            "screenshot_private_path": None, "screenshot_sha256": None,
            "capture_complete": False, "raw_html_sha256": None, "raw_html_private_path": None,
            "html_bytes": 0, "navigation_seconds": None, "attempt_seconds": None}


async def single(args, target):
    row = base_row(target)
    capture = b""
    row_path = args.output / f"{target['id']}.json"
    def save():
        row_path.write_text(json.dumps(row, indent=2) + "\n")
    save()
    start = time.perf_counter()
    try:
        async with async_playwright() as p:
            browser = await p.chromium.launch(headless=True, executable_path=str(args.binary),
                args=CONFIG["launch_args"], timeout=10000)
            try:
                row["browser_version"] = browser.version
                session = await browser.new_browser_cdp_session()
                row["browser_cdp_version"] = await session.send("Browser.getVersion")
                context = await browser.new_context(accept_downloads=False)
                page = await context.new_page()
                def received(response):
                    try:
                        if response.request.is_navigation_request() and response.request.frame == page.main_frame:
                            row["document_responses"].append({"url": response.url, "status": response.status})
                    except Exception:
                        pass
                page.on("response", received)
                row["navigation_started_at"] = utc()
                save()
                nav_start = time.perf_counter()
                response = None
                try:
                    response = await page.goto(target["url"], wait_until=CONFIG["wait_until"], timeout=20000)
                    row["navigation_outcome"] = "returned"
                except Exception as exc:
                    row["navigation_outcome"] = "error"
                    row["navigation_error"] = {"type": type(exc).__name__, "message": str(exc).splitlines()[0][:400]}
                row["navigation_seconds"] = round(time.perf_counter() - nav_start, 3)
                await asyncio.sleep(CONFIG["settle_seconds"])
                row["final_url"] = page.url
                if response is not None:
                    row["status"] = response.status
                    hops = []
                    request = response.request
                    while request.redirected_from is not None:
                        previous = request.redirected_from
                        prior = await previous.response()
                        hops.append({"previous": previous.url, "url": request.url, "status": prior.status if prior else None})
                        request = previous
                    row["redirect_history"] = list(reversed(hops))
                if row["document_responses"]:
                    row["status"] = row["document_responses"][-1]["status"]
                row["redirect_outcome"] = "followed" if row["redirect_history"] else (
                    "none" if row["final_url"] == target["url"] else "final_url_changed_history_unavailable")
                try:
                    html = await asyncio.wait_for(page.content(), timeout=5)
                    capture = html.encode("utf-8")
                    if len(capture) > CONFIG["max_capture_bytes"]:
                        capture = capture[:CONFIG["max_capture_bytes"]]
                        row["capture_error"] = {"type": "DOMSizeLimit", "message": "Serialized DOM exceeded 5 MiB; saved prefix only"}
                    else:
                        row["capture_complete"] = True
                    dest = args.output / f"{target['id']}.html"
                    dest.write_bytes(capture)
                    row.update(raw_html_private_path=str(dest), raw_html_sha256=sha(capture), html_bytes=len(capture))
                except Exception as exc:
                    row["capture_error"] = {"type": type(exc).__name__, "message": str(exc).splitlines()[0][:400]}
                row["latency_seconds"] = round(time.perf_counter() - nav_start, 3)
                save()
                try:
                    dest = args.output / f"{target['id']}.png"
                    await page.screenshot(path=str(dest), full_page=False, timeout=5000)
                    row.update(screenshot_private_path=str(dest), screenshot_sha256=sha(dest.read_bytes()))
                except Exception as exc:
                    row["screenshot_error"] = {"type": type(exc).__name__, "message": str(exc).splitlines()[0][:400]}
            finally:
                await browser.close()
    except Exception as exc:
        row["error"] = {"type": type(exc).__name__, "message": str(exc).splitlines()[0][:400], "classification": "browser_or_driver_error"}
    row.update(finished_at=utc(), attempt_seconds=round(time.perf_counter() - start, 3), checks=evaluate(capture, target))
    row.setdefault("latency_seconds", row["attempt_seconds"])
    row["empty_content"] = row["checks"]["text_chars"] == 0
    row["classification"] = result_class(row)
    row["usable_content"] = row["classification"] == "usable"
    save()


def tree_members(root):
    pending, seen, members = [root], set(), {}
    while pending:
        pid = pending.pop()
        if pid in seen:
            continue
        seen.add(pid)
        try:
            for task in Path(f"/proc/{pid}/task").iterdir():
                pending.extend(int(p) for p in (task / "children").read_text().split())
            stat = Path(f"/proc/{pid}/stat").read_text().rsplit(")", 1)[1].split()
            members[pid] = stat[19]  # start time; protects cleanup from PID reuse
        except (OSError, ValueError, IndexError):
            continue
    return members


def pss_members(members):
    total, read = 0, 0
    for pid in members:
        try:
            match = re.search(r"^Pss:\s+(\d+)", Path(f"/proc/{pid}/smaps_rollup").read_text(), re.M)
            if match:
                total += int(match.group(1))
                read += 1
        except OSError:
            continue
    return total / 1024 if read else None


def signal_owned(members, sig):
    # Chromium may use its own process group. Signal only recorded descendants.
    for pid, birth in reversed(list(members.items())):
        try:
            current = Path(f"/proc/{pid}/stat").read_text().rsplit(")", 1)[1].split()[19]
            if current == birth:
                os.kill(pid, sig)
        except (OSError, IndexError):
            pass


async def batch(args):
    manifest_bytes = args.manifest.read_bytes()
    manifest = json.loads(manifest_bytes)
    targets = manifest["targets"]
    assert len(targets) == len({t["id"] for t in targets}) == len({urlsplit(t["url"]).hostname for t in targets}) == 30
    assert version("patchright") == "1.63.0"
    initial_ram = memory()
    if initial_ram["MemAvailable"] < CONFIG["min_host_available_mb"]:
        raise RuntimeError("Unsafe RAM blocker before batch: less than 3 GiB available")
    os.umask(0o077)
    args.output.mkdir(parents=True, exist_ok=False, mode=0o700)
    run = {"run_id": args.output.name, "tool": "patchright", "tool_version": version("patchright"),
           "browser_executable": str(args.binary), "binary_sha256": sha(args.binary.read_bytes()),
           "executable_version": subprocess.check_output([str(args.binary), "--version"], text=True).strip(),
           "python_version": platform.python_version(), "platform": platform.platform(),
           "network": "Same local host direct outbound; proxy environment removed and --no-proxy-server; different UTC window from prior tools",
           "manifest_id": manifest["id"], "manifest_sha256": sha(manifest_bytes),
           "script_sha256": sha(Path(__file__).read_bytes()), "config": CONFIG,
           "executed_command": shlex.join([sys.executable, *sys.argv]), "started_at": utc(),
           "host_ram_preflight": initial_ram, "attempts": [],
           "body_hash_basis": "UTF-8 serialized DOM, not HTTP response bytes. Full HTML/screenshots/logs stay private in ignored runner/data.",
           "pss_caveat": "100 ms samples may miss brief peaks or changing processes. Includes isolated Python, Patchright driver, Chromium startup, screenshots and teardown; excludes batch parent. Sampled totals, not exact peaks or pure browser memory.",
           "policy_differences": ["Normal Chromium redirects: native URLRequest cap 20, versus wreq 3 and Lightpanda 10; no page.goto max_redirects option.",
             "Full Chromium resources and JS with fresh in-session cookies versus wreq HTTP-only without cookie store; Lightpanda skips optional image/style/iframe/worker loads.",
             "20-second navigation then two-second settle and DOM serialization, like Lightpanda, but full Chromium resources differ; screenshots/teardown are outside capture latency.",
             "40-second isolated-attempt watchdog plus RAM guards versus Lightpanda 30 seconds and wreq 20-second HTTP transfer.",
             "5 MiB DOM capture cap only; no per-network-response byte cap, unlike Lightpanda per-response cap and wreq final-body cap.",
             "Normal Chromium network/resource defaults; Lightpanda separately blocks private-network requests. No proxy, login, CAPTCHA solving, retry, custom user agent, persistent profile or resource interception.",
             "Default headless nonpersistent Chromium with Patchright patches; differs from project's headed persistent Chrome recommendation and earlier Amazon-specific waits."]}
    def save_run():
        run["summary"] = {"attempted_distinct_sites": len(run["attempts"]), "usable_results": sum(a["usable_content"] for a in run["attempts"]),
                          "classifications": dict(Counter(a["classification"] for a in run["attempts"]))}
        (args.output / "results.json").write_text(json.dumps(run, indent=2) + "\n")
    for index, target in enumerate(targets):
        if index:
            await asyncio.sleep(CONFIG["gap_seconds"])
        before_ram = memory()
        if before_ram["MemAvailable"] < CONFIG["min_host_available_mb"]:
            run["blocker"] = "Unsafe host RAM before next navigation"
            save_run()
            raise RuntimeError(run["blocker"])
        command = [sys.executable, str(Path(__file__).resolve()), "--binary", str(args.binary),
                   "--manifest", str(args.manifest), "--output", str(args.output), "--single", target["id"]]
        with open(args.output / f"{target['id']}-driver.log", "wb") as log:
            proc = await asyncio.create_subprocess_exec(*command, stdout=log, stderr=log, env=env_direct(), start_new_session=True)
            started = time.perf_counter()
            samples, guard, owned = [], None, {}
            while proc.returncode is None:
                members = tree_members(proc.pid)
                owned.update(members)
                reading = pss_members(members)
                ram = memory()
                if reading is not None:
                    samples.append({"elapsed_seconds": round(time.perf_counter() - started, 3), "pss_mb": round(reading, 3), "host_available_mb": round(ram["MemAvailable"], 3)})
                if ram["MemAvailable"] < CONFIG["min_host_available_mb"] or (reading or 0) > CONFIG["max_attempt_tree_pss_mb"]:
                    guard = "unsafe_memory_guard"
                elif time.perf_counter() - started > CONFIG["attempt_timeout_seconds"]:
                    guard = "attempt_timeout"
                if guard:
                    signal_owned(owned, signal.SIGTERM)
                    await asyncio.sleep(0.2)
                    signal_owned(owned, signal.SIGKILL)
                    break
                await asyncio.sleep(CONFIG["pss_sample_interval_seconds"])
            await proc.wait()
            signal_owned(owned, signal.SIGKILL)
        row_path = args.output / f"{target['id']}.json"
        row = json.loads(row_path.read_text()) if row_path.exists() else base_row(target)
        if guard or proc.returncode != 0:
            row["error"] = {"type": "AttemptGuard" if guard else "AttemptProcessExit", "message": guard or f"Attempt process exit {proc.returncode}", "classification": guard or "attempt_process_error"}
        saved_body = args.output / f"{target['id']}.html"
        row.setdefault("checks", evaluate(saved_body.read_bytes() if saved_body.exists() else b"", target))
        row["empty_content"] = row["checks"]["text_chars"] == 0
        row.setdefault("latency_seconds", round(time.perf_counter() - started, 3))
        row.setdefault("finished_at", utc())
        row.update(peak_sampled_tree_pss_mb=max((s["pss_mb"] for s in samples), default=None), pss_samples=len(samples),
                   attempt_process_exit=proc.returncode, host_ram_before=before_ram)
        row["classification"] = result_class(row)
        row["usable_content"] = row["classification"] == "usable"
        (args.output / f"{target['id']}-pss.json").write_text(json.dumps(samples) + "\n")
        row_path.write_text(json.dumps(row, indent=2) + "\n")
        run["attempts"].append(row)
        save_run()
        print(f"{target['id']}: HTTP {row['status']} | {row['classification']} | {row['html_bytes']} DOM bytes | PSS {row['peak_sampled_tree_pss_mb']} MiB", flush=True)
        if row["navigation_started_at"] is None or guard == "unsafe_memory_guard":
            run["blocker"] = "Setup or unsafe memory blocker: " + str(row["error"])
            save_run()
            raise RuntimeError(run["blocker"])
    run["finished_at"] = utc()
    save_run()
    print(json.dumps(run["summary"]), flush=True)


if __name__ == "__main__":
    parser = argparse.ArgumentParser(description=__doc__)
    parser.add_argument("--binary", type=Path, required=True)
    parser.add_argument("--manifest", type=Path, required=True)
    parser.add_argument("--output", type=Path, required=True)
    parser.add_argument("--single")
    args = parser.parse_args()
    args.binary = args.binary.resolve()
    if args.single:
        target = next(t for t in json.loads(args.manifest.read_text())["targets"] if t["id"] == args.single)
        asyncio.run(single(args, target))
    else:
        asyncio.run(batch(args))
