"""One fresh local Playwright browser per frozen URL; private captures only.
Use exact requirements.txt and the same pinned Chrome for Testing as Patchright. No stealth patches.
"""
import argparse
import asyncio
from collections import Counter
from datetime import datetime, timezone
import hashlib
from html.parser import HTMLParser
from importlib.metadata import version
import json
import os
from pathlib import Path
import platform
import re
import shlex
import signal
import subprocess
import sys
import time
from urllib.parse import urlsplit
from playwright.async_api import async_playwright

CONFIG = {
 "proxy_mode": "none", "login": False, "captcha_solving": False,
 "concurrency": 1, "gap_seconds": 2, "application_retries": 0,
 "headless": True, "fresh_browser_per_target": True, "accept_downloads": False,
 "navigation_timeout_seconds": 20, "attempt_timeout_seconds": 40,
 "wait_until": "domcontentloaded", "settle_seconds": 2,
 "max_capture_bytes": 5242880, "network_response_byte_cap": None,
 "redirects": "follow", "max_redirects": 20,
 "redirect_cap_basis": "Chromium URLRequest::kMaxRedirects at Browser.getVersion revision 971a7443b0c9b0a9b2860529b33331b76077ec62",
 "resources": "Normal Chromium images/styles/scripts/iframes/workers/subrequests; no resource interception",
 "cookies": "fresh in-memory context; no imported login or persistent profile",
 "launch_args": ["--no-proxy-server"], "pss_sample_interval_seconds": 0.1,
 "pss_scope": "Isolated attempt Python process and descendants, including Playwright driver and Chromium; excludes batch parent",
 "slice_timeout_seconds": 1500, "no_stealth_patches": True,
 "min_host_available_mb": 3072, "max_attempt_tree_pss_mb": 4096,
}

BLOCK_PHRASES = (
    "verify you are human", "robot or human", "robot check", "access denied",
    "pardon our interruption", "please complete the captcha", "enter the characters",
    "additional verification required", "unusual traffic", "checking your browser",
    "enable javascript and cookies to continue", "just a moment",
)
GATE_PHRASES = ("log in to continue", "sign in to continue", "sign in to linkedin", "consent required", "before you continue")


def utc():
    return datetime.now(timezone.utc).isoformat()


def sha(data):
    return hashlib.sha256(data).hexdigest()


class PageText(HTMLParser):
    """Source-text approximation, not rendered visibility or structured extraction."""
    SKIP = {"head", "script", "style", "template", "noscript"}

    def __init__(self):
        super().__init__(convert_charrefs=True)
        self.hidden = []
        self.parts = []

    def handle_starttag(self, tag, attrs):
        if tag in self.SKIP:
            self.hidden.append(tag)

    def handle_endtag(self, tag):
        if tag in self.hidden:
            self.hidden = self.hidden[:self.hidden.index(tag)]

    def handle_data(self, data):
        if not self.hidden:
            self.parts.append(data)


def evaluate(body, target):
    parser = PageText()
    parser.feed(body.decode("utf-8", errors="replace"))
    text = " ".join(" ".join(parser.parts).split())
    lower = text.lower()
    checks = [
        {"alternatives": group, "matched": [m for m in group if m.lower() in lower]}
        for group in target["required_text_groups"]
    ]
    regex_checks = [
        {**check, "matches": len(re.findall(check["pattern"], text, flags=re.I))}
        for check in target["text_regex_checks"]
    ]
    expected = (
        len(text) >= target["min_text_chars"]
        and all(check["matched"] for check in checks)
        and all(check["matches"] >= check["min_matches"] for check in regex_checks)
    )
    return {
        "text_chars": len(text), "required_text_checks": checks, "text_regex_checks": regex_checks,
        "min_text_chars": target["min_text_chars"], "has_required_content": bool(expected),
        "soft_block_checks": {"phrases_checked": list(BLOCK_PHRASES), "matched": [p for p in BLOCK_PHRASES if p in lower]},
        "gate_checks": {"phrases_checked": list(GATE_PHRASES), "matched": [p for p in GATE_PHRASES if p in lower]},
    }



def result_class(row):
    if row["error"]:
        return row["error"]["classification"]
    if row["navigation_error"]:
        return "navigation_error"
    if row["capture_error"]:
        return "capture_error"
    if row["status"] is None:
        return "status_unobserved"
    if row["status"] in (401, 403, 999):
        return "http_access_denied"
    if row["status"] == 429:
        return "http_rate_limited"
    if row["status"] >= 500:
        return "http_server_error"
    if row["status"] >= 400:
        return "http_client_error"
    if 300 <= row["status"] < 400:
        return "unresolved_redirect"
    if row["checks"]["soft_block_checks"]["matched"]:
        return "soft_block"
    if not row["checks"]["has_required_content"]:
        return "login_or_consent_gate" if row["checks"]["gate_checks"]["matched"] else "missing_required_content"
    return "usable" if row["status"] == 200 and row["capture_complete"] else "incomplete_capture"


def env_direct():
    env = os.environ.copy()
    for key in list(env):
        if key.lower() in ("http_proxy", "https_proxy", "all_proxy"):
            env.pop(key)
    env.update(PYTHONDONTWRITEBYTECODE="1")
    return env


def memory():
    readings = {line.split(":")[0]: int(line.split()[1]) / 1024
                for line in Path("/proc/meminfo").read_text().splitlines() if line.split(":")[0] in ("MemTotal", "MemAvailable", "SwapTotal", "SwapFree")}
    return {**readings, "unit": "MiB", "measured_at": utc()}


def base_row(target):
    return {"target_id": target["id"], "site": target["name"], "category": target["category"],
            "url": target["url"], "original_url": target["url"], "final_url": None,
            "expected_required_content": target["expected_required_content"], "started_at": utc(),
            "navigation_started_at": None, "status": None, "navigation_outcome": "not_started",
            "redirect_history": [], "redirect_outcome": "unobserved", "document_responses": [],
            "browser_version": None, "browser_cdp_version": None, "error": None,
            "navigation_error": None, "capture_error": None, "screenshot_error": None,
            "screenshot_private_path": None, "screenshot_sha256": None,
            "capture_complete": False, "raw_html_sha256": None, "raw_html_private_path": None,
            "html_bytes": 0, "navigation_seconds": None, "attempt_seconds": None}


async def single(args, target):
    row = base_row(target)
    capture = b""
    events = []
    row_path = args.output / f"{target['id']}.json"
    def save():
        row_path.write_text(json.dumps(row, indent=2) + "\n")
    save()
    start = time.perf_counter()
    try:
        async with async_playwright() as p:
            browser = await p.chromium.launch(headless=True, executable_path=str(args.binary),
                args=CONFIG["launch_args"], timeout=10000)
            try:
                row["browser_version"] = browser.version
                session = await browser.new_browser_cdp_session()
                row["browser_cdp_version"] = await session.send("Browser.getVersion")
                try:
                    row["effective_browser_arguments"] = (await session.send("Browser.getBrowserCommandLine"))["arguments"]
                    row["browser_arguments_basis"] = "Browser.getBrowserCommandLine via CDP before public navigation"
                except Exception as exc:
                    row["effective_browser_arguments"] = None
                    row["browser_arguments_error"] = {"type":type(exc).__name__,"message":str(exc).splitlines()[0][:400]}
                for pid,birth in tree_members(os.getpid()).items():
                    try:
                        process_path=Path(f"/proc/{pid}")
                        arguments=(process_path/"cmdline").read_bytes().decode("utf-8",errors="replace").split("\0")
                        arguments=[a for a in arguments if a]
                        if (process_path/"exe").resolve()==args.binary and not any(a.startswith("--type=") for a in arguments):
                            row["effective_browser_arguments"]=arguments
                            row["browser_arguments_basis"]="Owned descendant main Chrome /proc cmdline read before public navigation"
                            break
                    except (OSError,ValueError):pass
                context = await browser.new_context(accept_downloads=False)
                page = await context.new_page()
                row["initial_cookie_count"] = len(await context.cookies())
                row["browser_observation"] = await page.evaluate("({user_agent:navigator.userAgent,webdriver:navigator.webdriver,language:navigator.language,platform:navigator.platform,viewport:{width:innerWidth,height:innerHeight}})")
                def log_event(kind,value):
                    if len(events)<1000: events.append({"utc":utc(),"kind":kind,"value":value})
                page.on("console",lambda m:log_event("console",{"type":m.type,"text":m.text[:1000]}))
                page.on("pageerror",lambda e:log_event("pageerror",str(e)[:1000]))
                page.on("requestfailed",lambda r:log_event("requestfailed",{"url":r.url,"error":r.failure}))
                def received(response):
                    try:
                        if response.request.is_navigation_request() and response.request.frame == page.main_frame:
                            observed={"url": response.url, "status": response.status}
                            row["document_responses"].append(observed)
                            log_event("main_response",observed)
                    except Exception:
                        pass
                page.on("response", received)
                row["navigation_started_at"] = utc()
                save()
                nav_start = time.perf_counter()
                response = None
                try:
                    response = await page.goto(target["url"], wait_until=CONFIG["wait_until"], timeout=20000)
                    row["navigation_outcome"] = "returned"
                except Exception as exc:
                    row["navigation_outcome"] = "error"
                    row["navigation_error"] = {"type": type(exc).__name__, "message": str(exc).splitlines()[0][:400]}
                row["navigation_seconds"] = round(time.perf_counter() - nav_start, 3)
                await asyncio.sleep(CONFIG["settle_seconds"])
                row["final_url"] = page.url
                if response is not None:
                    row["status"] = response.status
                    hops = []
                    request = response.request
                    while request.redirected_from is not None:
                        previous = request.redirected_from
                        prior = await previous.response()
                        hops.append({"previous": previous.url, "url": request.url, "status": prior.status if prior else None})
                        request = previous
                    row["redirect_history"] = list(reversed(hops))
                if row["document_responses"]:
                    row["status"] = row["document_responses"][-1]["status"]
                row["redirect_outcome"] = "followed" if row["redirect_history"] else (
                    "none" if row["final_url"] == target["url"] else "final_url_changed_history_unavailable")
                try:
                    html = await asyncio.wait_for(page.content(), timeout=5)
                    capture = html.encode("utf-8")
                    if len(capture) > CONFIG["max_capture_bytes"]:
                        capture = capture[:CONFIG["max_capture_bytes"]]
                        row["capture_error"] = {"type": "DOMSizeLimit", "message": "Serialized DOM exceeded 5 MiB; saved prefix only"}
                    else:
                        row["capture_complete"] = True
                    dest = args.output / f"{target['id']}.html"
                    dest.write_bytes(capture)
                    row.update(raw_html_private_path=str(dest), raw_html_sha256=sha(capture), html_bytes=len(capture))
                except Exception as exc:
                    row["capture_error"] = {"type": type(exc).__name__, "message": str(exc).splitlines()[0][:400]}
                row["latency_seconds"] = round(time.perf_counter() - nav_start, 3)
                save()
                try:
                    dest = args.output / f"{target['id']}.png"
                    await page.screenshot(path=str(dest), full_page=False, timeout=5000)
                    row.update(screenshot_private_path=str(dest), screenshot_sha256=sha(dest.read_bytes()))
                except Exception as exc:
                    row["screenshot_error"] = {"type": type(exc).__name__, "message": str(exc).splitlines()[0][:400]}
            finally:
                await browser.close()
    except Exception as exc:
        row["error"] = {"type": type(exc).__name__, "message": str(exc).splitlines()[0][:400], "classification": "browser_or_driver_error"}
    event_path=args.output / f"{target['id']}-browser-events.json"
    event_path.write_text(json.dumps(events,indent=2)+"\n")
    row["browser_event_log_sha256"]=sha(event_path.read_bytes())
    row["browser_event_log_entries"]=len(events)
    row.update(finished_at=utc(), attempt_seconds=round(time.perf_counter() - start, 3), checks=evaluate(capture, target))
    row.setdefault("latency_seconds", row["attempt_seconds"])
    row["empty_content"] = row["checks"]["text_chars"] == 0
    row["classification"] = result_class(row)
    row["usable_content"] = row["classification"] == "usable"
    save()


def tree_members(root):
    pending, seen, members = [root], set(), {}
    while pending:
        pid = pending.pop()
        if pid in seen:
            continue
        seen.add(pid)
        try:
            for task in Path(f"/proc/{pid}/task").iterdir():
                pending.extend(int(p) for p in (task / "children").read_text().split())
            stat = Path(f"/proc/{pid}/stat").read_text().rsplit(")", 1)[1].split()
            members[pid] = stat[19]  # start time; protects cleanup from PID reuse
        except (OSError, ValueError, IndexError):
            continue
    return members


def pss_members(members):
    total, read = 0, 0
    for pid in members:
        try:
            match = re.search(r"^Pss:\s+(\d+)", Path(f"/proc/{pid}/smaps_rollup").read_text(), re.M)
            if match:
                total += int(match.group(1))
                read += 1
        except OSError:
            continue
    return total / 1024 if read else None


def signal_owned(members, sig):
    # Chromium may use its own process group. Signal only recorded descendants.
    for pid, birth in reversed(list(members.items())):
        try:
            current = Path(f"/proc/{pid}/stat").read_text().rsplit(")", 1)[1].split()[19]
            if current == birth:
                os.kill(pid, sig)
        except (OSError, IndexError):
            pass


async def batch(args):
    manifest_bytes = args.manifest.read_bytes()
    manifest = json.loads(manifest_bytes)
    targets = manifest["targets"]
    assert len(targets) == len({t["id"] for t in targets}) == len({urlsplit(t["url"]).hostname for t in targets}) == 30
    assert version("playwright") == "1.63.0"
    assert sha(args.binary.read_bytes()) == "8c599d43aec53f2460a31ae2f4af6bd863f8258b34ff519564bc5d4726bfaa1e"
    if args.target: targets=[t for t in targets if t["id"]==args.target]
    slice_start=time.perf_counter()
    initial_ram = memory()
    if initial_ram["MemAvailable"] < CONFIG["min_host_available_mb"]:
        raise RuntimeError("Unsafe RAM blocker before batch: less than 3 GiB available")
    os.umask(0o077)
    args.output.mkdir(parents=True, exist_ok=False, mode=0o700)
    run = {"run_id": args.output.name, "tool": "playwright", "tool_version": version("playwright"),
           "mode":"Plain Playwright / headless Chrome", "requirements_sha256":sha((Path(__file__).parent/"requirements.txt").read_bytes()),
           "documentation":json.loads((Path(__file__).parent/"documentation.json").read_text()),
           "browser_executable": str(args.binary), "binary_sha256": sha(args.binary.read_bytes()),
           "executable_version": subprocess.check_output([str(args.binary), "--version"], text=True).strip(),
           "python_version": platform.python_version(), "platform": platform.platform(),
           "network": "Same local host direct outbound; proxy environment removed and --no-proxy-server; different UTC window from prior tools",
           "manifest_id": manifest["id"], "manifest_sha256": sha(manifest_bytes),
           "script_sha256": sha(Path(__file__).read_bytes()), "config": CONFIG,
           "executed_command": shlex.join([sys.executable, *sys.argv]), "started_at": utc(),
           "host_ram_preflight": initial_ram, "attempts": [],
           "body_hash_basis": "UTF-8 serialized DOM, not HTTP response bytes. Full HTML/screenshots/logs stay private in ignored runner/data.",
           "pss_caveat": "100 ms samples may miss brief peaks or changing processes. Includes isolated Python, Playwright driver, Chromium startup, screenshots and teardown; excludes batch parent. Sampled totals, not exact peaks or pure browser memory.",
           "policy_differences": [
             "Same pinned Chrome for Testing 153.0.8010.12 binary and native redirect cap 20 as earlier Patchright 1.63.0; same host, later UTC window. Different Python/Node driver implementation: plain upstream Playwright has no Patchright patches.",
             "Same explicit launch options: headless=True, executable_path, args=[--no-proxy-server], launch timeout 10s; new_context accept_downloads=False; default ordinary JS/resources/cookies/UA and TLS policy retained. No ignore_default_args, stealth module, init script, custom UA, fingerprint override or interception.",
             "Plain pinned Playwright defaults include --no-sandbox but omit --enable-automation. CDP getBrowserCommandLine can therefore refuse; an own-descendant /proc main Chrome command line is also read. Navigator UA/webdriver observations are recorded on local about:blank before navigation. No default flags changed; current DeepWiki sandbox summary conflicts with pinned source.",
             "20-second DOMContentLoaded wait then two-second settle and 5-second DOM serialization guard match earlier Patchright. Latency through DOM capture excludes startup, screenshot and teardown; no application repeat of page.goto. Native browser resource retries/reconnections are not instrumented.",
             "40-second whole-worker watchdog, 3 GiB available host RAM floor, 4 GiB tree PSS cap, 100 ms samples and two-second sequential gaps match Patchright; added 1500-second slice deadline checked before each target.",
             "5 MiB DOM cap applies after serialization, not to HTTP response/network resource bytes. Native HTTP redirects and client-side final URL changes are reported separately; no goto max_redirects parameter.",
             "Viewport PNG after DOM capture; screenshot timeout/failure does not change unchanged frozen classifier. Source-text parser misses iframe/shadow-root content and does not resolve CSS visibility. All raw DOM, screenshots and browser/driver logs private.",
             "One dated navigation per target is not a reliability estimate or causal test of stealth, detection, headers or driver features; no global winner."]}

    def save_run():
        run["summary"] = {"attempted_distinct_sites": len(run["attempts"]), "usable_results": sum(a["usable_content"] for a in run["attempts"]),
                          "classifications": dict(Counter(a["classification"] for a in run["attempts"]))}
        (args.output / "results.json").write_text(json.dumps(run, indent=2) + "\n")
    for index, target in enumerate(targets):
        if time.perf_counter()-slice_start>CONFIG["slice_timeout_seconds"]:
            run["blocker"]="Bounded slice deadline reached before next target";save_run();raise RuntimeError(run["blocker"])
        if index:
            await asyncio.sleep(CONFIG["gap_seconds"])
        before_ram = memory()
        if before_ram["MemAvailable"] < CONFIG["min_host_available_mb"]:
            run["blocker"] = "Unsafe host RAM before next navigation"
            save_run()
            raise RuntimeError(run["blocker"])
        command = [sys.executable, str(Path(__file__).resolve()), "--binary", str(args.binary),
                   "--manifest", str(args.manifest), "--output", str(args.output), "--single", target["id"]]
        with open(args.output / f"{target['id']}-driver.log", "wb") as log:
            proc = await asyncio.create_subprocess_exec(*command, stdout=log, stderr=log, env=env_direct(), start_new_session=True)
            started = time.perf_counter()
            samples, guard, owned = [], None, {}
            while proc.returncode is None:
                members = tree_members(proc.pid)
                owned.update(members)
                reading = pss_members(members)
                ram = memory()
                if reading is not None:
                    samples.append({"elapsed_seconds": round(time.perf_counter() - started, 3), "pss_mb": round(reading, 3), "host_available_mb": round(ram["MemAvailable"], 3)})
                if ram["MemAvailable"] < CONFIG["min_host_available_mb"] or (reading or 0) > CONFIG["max_attempt_tree_pss_mb"]:
                    guard = "unsafe_memory_guard"
                elif time.perf_counter() - started > CONFIG["attempt_timeout_seconds"]:
                    guard = "attempt_timeout"
                if guard:
                    signal_owned(owned, signal.SIGTERM)
                    await asyncio.sleep(0.2)
                    signal_owned(owned, signal.SIGKILL)
                    break
                await asyncio.sleep(CONFIG["pss_sample_interval_seconds"])
            await proc.wait()
            signal_owned(owned, signal.SIGKILL)
        row_path = args.output / f"{target['id']}.json"
        row = json.loads(row_path.read_text()) if row_path.exists() else base_row(target)
        if guard or proc.returncode != 0:
            row["error"] = {"type": "AttemptGuard" if guard else "AttemptProcessExit", "message": guard or f"Attempt process exit {proc.returncode}", "classification": guard or "attempt_process_error"}
        saved_body = args.output / f"{target['id']}.html"
        row.setdefault("checks", evaluate(saved_body.read_bytes() if saved_body.exists() else b"", target))
        row["empty_content"] = row["checks"]["text_chars"] == 0
        row.setdefault("latency_seconds", round(time.perf_counter() - started, 3))
        row.setdefault("finished_at", utc())
        surviving=[]
        for pid,birth in owned.items():
            try:
                fields=Path(f"/proc/{pid}/stat").read_text().rsplit(")",1)[1].split()
                if fields[19]==birth and fields[0]!="Z":surviving.append(pid)
            except (OSError,IndexError):pass
        row["cleanup_surviving_owned_pids"]=surviving
        row["driver_log_sha256"]=sha((args.output/f"{target['id']}-driver.log").read_bytes())
        row.update(peak_sampled_tree_pss_mb=max((s["pss_mb"] for s in samples), default=None), pss_samples=len(samples),
                   attempt_process_exit=proc.returncode, host_ram_before=before_ram)
        row["classification"] = result_class(row)
        row["usable_content"] = row["classification"] == "usable"
        (args.output / f"{target['id']}-pss.json").write_text(json.dumps(samples) + "\n")
        row_path.write_text(json.dumps(row, indent=2) + "\n")
        run["attempts"].append(row)
        save_run()
        print(f"{target['id']}: HTTP {row['status']} | {row['classification']} | {row['html_bytes']} DOM bytes | PSS {row['peak_sampled_tree_pss_mb']} MiB", flush=True)
        if row["navigation_started_at"] is None or guard == "unsafe_memory_guard":
            run["blocker"] = "Setup or unsafe memory blocker: " + str(row["error"])
            save_run()
            raise RuntimeError(run["blocker"])
    run["finished_at"] = utc()
    save_run()
    print(json.dumps(run["summary"]), flush=True)


if __name__ == "__main__":
    parser = argparse.ArgumentParser(description=__doc__)
    parser.add_argument("--binary", type=Path, required=True)
    parser.add_argument("--manifest", type=Path, required=True)
    parser.add_argument("--output", type=Path, required=True)
    parser.add_argument("--single")
    parser.add_argument("--target",choices=[t["id"] for t in json.loads(Path("runner/corpora/public-30-v1.json").read_text())["targets"]])
    args = parser.parse_args()
    args.binary = args.binary.resolve()
    args.manifest=args.manifest.resolve();args.output=args.output.resolve()
    if args.single:
        target = next(t for t in json.loads(args.manifest.read_text())["targets"] if t["id"] == args.single)
        asyncio.run(single(args, target))
    else:
        asyncio.run(batch(args))
