"""One fresh local Camoufox browser per frozen URL; private captures only.
Requires Python >=3.11 and camoufox==0.5.6 and playwright==1.62.0 with an existing Firefox bundle and cached default uBlock Origin addon.
"""
import argparse
import asyncio
from collections import Counter
from datetime import datetime, timezone
import hashlib
from html.parser import HTMLParser
from importlib.metadata import version
import json
import os
from pathlib import Path
import platform
import re
import shlex
import signal
import subprocess
import sys
import time
from urllib.parse import urlsplit
from playwright.async_api import async_playwright
from camoufox.async_api import AsyncNewBrowser
from camoufox.utils import launch_options
from camoufox.addons import ADDONS_DIR

CONFIG = {
 "proxy_mode": "none", "login": False, "captcha_solving": False,
 "concurrency": 1, "gap_seconds": 2, "application_retries": 0,
 "headless": True, "fresh_browser_per_target": True, "accept_downloads": False,
 "navigation_timeout_seconds": 20, "attempt_timeout_seconds": 40,
 "wait_until": "domcontentloaded", "settle_seconds": 2,
 "max_capture_bytes": 5242880, "network_response_byte_cap": None,
 "redirects": "follow", "max_redirects": 3,
 "redirect_cap_basis": "Explicit Firefox network.http.redirection-limit=3; bundled greprefs.js native default is 20",
 "resources": "Normal Firefox JS/resources with default cached uBlock Origin 1.75.0; addon can filter requests",
 "cookies": "fresh in-memory context; no imported login or persistent profile",
 "geoip": False, "humanize": False, "os": "linux",
 "firefox_user_prefs": {"network.proxy.type": 0, "network.http.redirection-limit": 3}, "pss_sample_interval_seconds": 0.1,
 "pss_scope": "Isolated attempt Python process and descendants, including Camoufox driver and Firefox; excludes batch parent",
 "min_host_available_mb": 3072, "max_attempt_tree_pss_mb": 4096,
}

BLOCK_PHRASES = (
    "verify you are human", "robot or human", "robot check", "access denied",
    "pardon our interruption", "please complete the captcha", "enter the characters",
    "additional verification required", "unusual traffic", "checking your browser",
    "enable javascript and cookies to continue", "just a moment",
)
GATE_PHRASES = ("log in to continue", "sign in to continue", "sign in to linkedin", "consent required", "before you continue")


def utc():
    return datetime.now(timezone.utc).isoformat()


def sha(data):
    return hashlib.sha256(data).hexdigest()


class PageText(HTMLParser):
    """Source-text approximation, not rendered visibility or structured extraction."""
    SKIP = {"head", "script", "style", "template", "noscript"}

    def __init__(self):
        super().__init__(convert_charrefs=True)
        self.hidden = []
        self.parts = []

    def handle_starttag(self, tag, attrs):
        if tag in self.SKIP:
            self.hidden.append(tag)

    def handle_endtag(self, tag):
        if tag in self.hidden:
            self.hidden = self.hidden[:self.hidden.index(tag)]

    def handle_data(self, data):
        if not self.hidden:
            self.parts.append(data)


def evaluate(body, target):
    parser = PageText()
    parser.feed(body.decode("utf-8", errors="replace"))
    text = " ".join(" ".join(parser.parts).split())
    lower = text.lower()
    checks = [
        {"alternatives": group, "matched": [m for m in group if m.lower() in lower]}
        for group in target["required_text_groups"]
    ]
    regex_checks = [
        {**check, "matches": len(re.findall(check["pattern"], text, flags=re.I))}
        for check in target["text_regex_checks"]
    ]
    expected = (
        len(text) >= target["min_text_chars"]
        and all(check["matched"] for check in checks)
        and all(check["matches"] >= check["min_matches"] for check in regex_checks)
    )
    return {
        "text_chars": len(text), "required_text_checks": checks, "text_regex_checks": regex_checks,
        "min_text_chars": target["min_text_chars"], "has_required_content": bool(expected),
        "soft_block_checks": {"phrases_checked": list(BLOCK_PHRASES), "matched": [p for p in BLOCK_PHRASES if p in lower]},
        "gate_checks": {"phrases_checked": list(GATE_PHRASES), "matched": [p for p in GATE_PHRASES if p in lower]},
    }



def result_class(row):
    if row["error"]:
        return row["error"]["classification"]
    if row["navigation_error"]:
        return "navigation_error"
    if row["capture_error"]:
        return "capture_error"
    if row["status"] is None:
        return "status_unobserved"
    if row["status"] in (401, 403, 999):
        return "http_access_denied"
    if row["status"] == 429:
        return "http_rate_limited"
    if row["status"] >= 500:
        return "http_server_error"
    if row["status"] >= 400:
        return "http_client_error"
    if 300 <= row["status"] < 400:
        return "unresolved_redirect"
    if row["checks"]["soft_block_checks"]["matched"]:
        return "soft_block"
    if not row["checks"]["has_required_content"]:
        return "login_or_consent_gate" if row["checks"]["gate_checks"]["matched"] else "missing_required_content"
    return "usable" if row["status"] == 200 and row["capture_complete"] else "incomplete_capture"


def env_direct():
    env = os.environ.copy()
    for key in list(env):
        if key.lower() in ("http_proxy", "https_proxy", "all_proxy"):
            env.pop(key)
    env.update(PYTHONDONTWRITEBYTECODE="1")
    return env


def memory():
    readings = {line.split(":")[0]: int(line.split()[1]) / 1024
                for line in Path("/proc/meminfo").read_text().splitlines() if line.split(":")[0] in ("MemTotal", "MemAvailable", "SwapTotal", "SwapFree")}
    return {**readings, "unit": "MiB", "measured_at": utc()}


def base_row(target):
    return {"target_id": target["id"], "site": target["name"], "category": target["category"],
            "url": target["url"], "original_url": target["url"], "final_url": None,
            "expected_required_content": target["expected_required_content"], "started_at": utc(),
            "navigation_started_at": None, "status": None, "navigation_outcome": "not_started",
            "redirect_history": [], "redirect_outcome": "unobserved", "document_responses": [],
            "browser_version": None, "generated_launch_config": None, "error": None,
            "navigation_error": None, "capture_error": None, "screenshot_error": None,
            "screenshot_private_path": None, "screenshot_sha256": None,
            "capture_complete": False, "raw_html_sha256": None, "raw_html_private_path": None,
            "html_bytes": 0, "navigation_seconds": None, "attempt_seconds": None}


async def single(args, target):
    row = base_row(target)
    capture = b""
    row_path = args.output / f"{target['id']}.json"
    def save():
        row_path.write_text(json.dumps(row, indent=2) + "\n")
    save()
    start = time.perf_counter()
    try:
        async with async_playwright() as p:
            opts = launch_options(headless=True, os="linux", geoip=False, humanize=False,
                executable_path=str(args.binary), firefox_user_prefs=CONFIG["firefox_user_prefs"], env=env_direct())
            config_chunks = sorted((k,v) for k,v in opts["env"].items() if k.startswith("CAMOU_CONFIG_"))
            row["generated_launch_config"] = {
                "synthetic_fingerprint": json.loads("".join(v for k,v in sorted(config_chunks, key=lambda kv:int(kv[0].rsplit("_",1)[1])))),
                "options": {k:v for k,v in opts.items() if k != "env"},
                "env_note": "Only generated CAMOU_CONFIG metadata is published; inherited environment is excluded"}
            save()
            browser = await AsyncNewBrowser(p, from_options={**opts, "timeout": 10000})
            try:
                row["browser_version"] = browser.version
                context = await browser.new_context(accept_downloads=False)
                page = await context.new_page()
                def received(response):
                    try:
                        if response.request.is_navigation_request() and response.request.frame == page.main_frame:
                            row["document_responses"].append({"url": response.url, "status": response.status})
                    except Exception:
                        pass
                page.on("response", received)
                row["navigation_started_at"] = utc()
                save()
                nav_start = time.perf_counter()
                response = None
                try:
                    response = await page.goto(target["url"], wait_until=CONFIG["wait_until"], timeout=20000)
                    row["navigation_outcome"] = "returned"
                except Exception as exc:
                    row["navigation_outcome"] = "error"
                    row["navigation_error"] = {"type": type(exc).__name__, "message": str(exc).splitlines()[0][:400]}
                row["navigation_seconds"] = round(time.perf_counter() - nav_start, 3)
                await asyncio.sleep(CONFIG["settle_seconds"])
                row["final_url"] = page.url
                if response is not None:
                    row["status"] = response.status
                    hops = []
                    request = response.request
                    while request.redirected_from is not None:
                        previous = request.redirected_from
                        prior = await previous.response()
                        hops.append({"previous": previous.url, "url": request.url, "status": prior.status if prior else None})
                        request = previous
                    row["redirect_history"] = list(reversed(hops))
                if row["document_responses"]:
                    row["status"] = row["document_responses"][-1]["status"]
                row["redirect_outcome"] = "followed" if row["redirect_history"] else (
                    "none" if row["final_url"] == target["url"] else "final_url_changed_history_unavailable")
                try:
                    html = await asyncio.wait_for(page.content(), timeout=5)
                    capture = html.encode("utf-8")
                    if len(capture) > CONFIG["max_capture_bytes"]:
                        capture = capture[:CONFIG["max_capture_bytes"]]
                        row["capture_error"] = {"type": "DOMSizeLimit", "message": "Serialized DOM exceeded 5 MiB; saved prefix only"}
                    else:
                        row["capture_complete"] = True
                    dest = args.output / f"{target['id']}.html"
                    dest.write_bytes(capture)
                    row.update(raw_html_private_path=str(dest), raw_html_sha256=sha(capture), html_bytes=len(capture))
                except Exception as exc:
                    row["capture_error"] = {"type": type(exc).__name__, "message": str(exc).splitlines()[0][:400]}
                row["latency_seconds"] = round(time.perf_counter() - nav_start, 3)
                save()
                try:
                    dest = args.output / f"{target['id']}.png"
                    await page.screenshot(path=str(dest), full_page=False, timeout=5000)
                    row.update(screenshot_private_path=str(dest), screenshot_sha256=sha(dest.read_bytes()))
                except Exception as exc:
                    row["screenshot_error"] = {"type": type(exc).__name__, "message": str(exc).splitlines()[0][:400]}
            finally:
                await browser.close()
    except Exception as exc:
        row["error"] = {"type": type(exc).__name__, "message": str(exc).splitlines()[0][:400], "classification": "browser_or_driver_error"}
    row.update(finished_at=utc(), attempt_seconds=round(time.perf_counter() - start, 3), checks=evaluate(capture, target))
    row.setdefault("latency_seconds", row["attempt_seconds"])
    row["empty_content"] = row["checks"]["text_chars"] == 0
    row["classification"] = result_class(row)
    row["usable_content"] = row["classification"] == "usable"
    save()


def tree_members(root):
    pending, seen, members = [root], set(), {}
    while pending:
        pid = pending.pop()
        if pid in seen:
            continue
        seen.add(pid)
        try:
            for task in Path(f"/proc/{pid}/task").iterdir():
                pending.extend(int(p) for p in (task / "children").read_text().split())
            stat = Path(f"/proc/{pid}/stat").read_text().rsplit(")", 1)[1].split()
            members[pid] = stat[19]  # start time; protects cleanup from PID reuse
        except (OSError, ValueError, IndexError):
            continue
    return members


def pss_members(members):
    total, read = 0, 0
    for pid in members:
        try:
            match = re.search(r"^Pss:\s+(\d+)", Path(f"/proc/{pid}/smaps_rollup").read_text(), re.M)
            if match:
                total += int(match.group(1))
                read += 1
        except OSError:
            continue
    return total / 1024 if read else None


def signal_owned(members, sig):
    # Firefox may use its own process group. Signal only recorded descendants.
    for pid, birth in reversed(list(members.items())):
        try:
            current = Path(f"/proc/{pid}/stat").read_text().rsplit(")", 1)[1].split()[19]
            if current == birth:
                os.kill(pid, sig)
        except (OSError, IndexError):
            pass


async def batch(args):
    manifest_bytes = args.manifest.read_bytes()
    manifest = json.loads(manifest_bytes)
    targets = manifest["targets"]
    assert len(targets) == len({t["id"] for t in targets}) == len({urlsplit(t["url"]).hostname for t in targets}) == 30
    assert version("camoufox") == "0.5.6"
    assert version("playwright") == "1.62.0"
    addon_manifest = ADDONS_DIR / "UBO" / "manifest.json"
    assert addon_manifest.exists(), "Default addon missing; stop instead of downloading during target attempts"
    initial_ram = memory()
    if initial_ram["MemAvailable"] < CONFIG["min_host_available_mb"]:
        raise RuntimeError("Unsafe RAM blocker before batch: less than 3 GiB available")
    os.umask(0o077)
    args.output.mkdir(parents=True, exist_ok=False, mode=0o700)
    run = {"run_id": args.output.name, "tool": "camoufox", "tool_version": version("camoufox"), "playwright_version": version("playwright"),
           "browserforge_version": version("browserforge"),
           "default_addon": {"name": "uBlock Origin", "version": json.loads(addon_manifest.read_text())["version"], "manifest_sha256": sha(addon_manifest.read_bytes())},
           "browser_bundle_metadata": json.loads((args.binary.parent / "version.json").read_text()),
           "browser_executable": str(args.binary), "binary_sha256": sha(args.binary.read_bytes()),
           "executable_version": "152.0.4-beta.31 (application.ini and version.json)",
           "python_version": platform.python_version(), "platform": platform.platform(),
           "network": "Same local host direct outbound; proxy environment removed and network.proxy.type=0; different UTC window from prior tools",
           "manifest_id": manifest["id"], "manifest_sha256": sha(manifest_bytes),
           "script_sha256": sha(Path(__file__).read_bytes()), "config": CONFIG,
           "executed_command": shlex.join([sys.executable, *sys.argv]), "started_at": utc(),
           "host_ram_preflight": initial_ram, "attempts": [],
           "body_hash_basis": "UTF-8 serialized DOM, not HTTP response bytes. Full HTML/screenshots/logs stay private in ignored runner/data.",
           "pss_caveat": "100 ms samples may miss brief peaks or changing processes. Includes isolated Python, Camoufox driver, Firefox startup, screenshots and teardown; excludes batch parent. Sampled totals, not exact peaks or pure browser memory.",
           "policy_differences": [
             "Normal Firefox redirects with explicit network.http.redirection-limit=3, matching wreq; Lightpanda native cap 10 and Patchright Chromium native cap 20 differ.",
             "Fresh Firefox JS/resources plus default cached uBlock Origin 1.75.0, unlike Patchright Chromium without this addon, Lightpanda reduced resources and wreq HTTP-only. Addon filtering and subrequests are not separately measured.",
             "20-second DOMContentLoaded navigation then two-second settle and DOM serialization; optional screenshot has a separate five-second deadline. 40-second whole-attempt watchdog includes startup and teardown.",
             "5 MiB serialized DOM cap only, with no per-response network byte cap; differs from wreq final-body cap and Lightpanda per-response cap.",
             "Generated BrowserForge Linux fingerprint and random geometry per fresh launch; exact generated synthetic config recorded per target. No GeoIP lookup, humanization, login, proxy, CAPTCHA solving, persistent profile, custom waits or application retry.",
             "100 ms process-tree PSS includes Python, Playwright driver, Firefox, addon, screenshots and teardown; sampled observations can miss peaks. Host available RAM must remain at least 3 GiB and attempt PSS below 4 GiB; swap is nearly full at preflight.",
             "Different UTC window from earlier tools on the same host direct network; transient responses, browser engine, fingerprint, cookies, addon and resource policies can all affect splits. One observation per site cannot establish a winner."]}

    def save_run():
        run["summary"] = {"attempted_distinct_sites": len(run["attempts"]), "usable_results": sum(a["usable_content"] for a in run["attempts"]),
                          "classifications": dict(Counter(a["classification"] for a in run["attempts"]))}
        (args.output / "results.json").write_text(json.dumps(run, indent=2) + "\n")
    for index, target in enumerate(targets):
        if index:
            await asyncio.sleep(CONFIG["gap_seconds"])
        before_ram = memory()
        if before_ram["MemAvailable"] < CONFIG["min_host_available_mb"]:
            run["blocker"] = "Unsafe host RAM before next navigation"
            save_run()
            raise RuntimeError(run["blocker"])
        command = [sys.executable, str(Path(__file__).resolve()), "--binary", str(args.binary),
                   "--manifest", str(args.manifest), "--output", str(args.output), "--single", target["id"]]
        with open(args.output / f"{target['id']}-driver.log", "wb") as log:
            proc = await asyncio.create_subprocess_exec(*command, stdout=log, stderr=log, env=env_direct(), start_new_session=True)
            started = time.perf_counter()
            samples, guard, owned = [], None, {}
            while proc.returncode is None:
                members = tree_members(proc.pid)
                owned.update(members)
                reading = pss_members(members)
                ram = memory()
                if reading is not None:
                    samples.append({"elapsed_seconds": round(time.perf_counter() - started, 3), "pss_mb": round(reading, 3), "host_available_mb": round(ram["MemAvailable"], 3)})
                if ram["MemAvailable"] < CONFIG["min_host_available_mb"] or (reading or 0) > CONFIG["max_attempt_tree_pss_mb"]:
                    guard = "unsafe_memory_guard"
                elif time.perf_counter() - started > CONFIG["attempt_timeout_seconds"]:
                    guard = "attempt_timeout"
                if guard:
                    signal_owned(owned, signal.SIGTERM)
                    await asyncio.sleep(0.2)
                    signal_owned(owned, signal.SIGKILL)
                    break
                await asyncio.sleep(CONFIG["pss_sample_interval_seconds"])
            await proc.wait()
            signal_owned(owned, signal.SIGKILL)
        row_path = args.output / f"{target['id']}.json"
        row = json.loads(row_path.read_text()) if row_path.exists() else base_row(target)
        if guard or proc.returncode != 0:
            row["error"] = {"type": "AttemptGuard" if guard else "AttemptProcessExit", "message": guard or f"Attempt process exit {proc.returncode}", "classification": guard or "attempt_process_error"}
        saved_body = args.output / f"{target['id']}.html"
        row.setdefault("checks", evaluate(saved_body.read_bytes() if saved_body.exists() else b"", target))
        row["empty_content"] = row["checks"]["text_chars"] == 0
        row.setdefault("latency_seconds", round(time.perf_counter() - started, 3))
        row.setdefault("finished_at", utc())
        row.update(peak_sampled_tree_pss_mb=max((s["pss_mb"] for s in samples), default=None), pss_samples=len(samples),
                   attempt_process_exit=proc.returncode, host_ram_before=before_ram)
        row["classification"] = result_class(row)
        row["usable_content"] = row["classification"] == "usable"
        (args.output / f"{target['id']}-pss.json").write_text(json.dumps(samples) + "\n")
        row_path.write_text(json.dumps(row, indent=2) + "\n")
        run["attempts"].append(row)
        save_run()
        print(f"{target['id']}: HTTP {row['status']} | {row['classification']} | {row['html_bytes']} DOM bytes | PSS {row['peak_sampled_tree_pss_mb']} MiB", flush=True)
        if row["navigation_started_at"] is None or guard == "unsafe_memory_guard":
            run["blocker"] = "Setup or unsafe memory blocker: " + str(row["error"])
            save_run()
            raise RuntimeError(run["blocker"])
    run["finished_at"] = utc()
    save_run()
    print(json.dumps(run["summary"]), flush=True)


if __name__ == "__main__":
    parser = argparse.ArgumentParser(description=__doc__)
    parser.add_argument("--binary", type=Path, required=True)
    parser.add_argument("--manifest", type=Path, required=True)
    parser.add_argument("--output", type=Path, required=True)
    parser.add_argument("--single")
    args = parser.parse_args()
    args.binary = args.binary.resolve()
    if args.single:
        target = next(t for t in json.loads(args.manifest.read_text())["targets"] if t["id"] == args.single)
        asyncio.run(single(args, target))
    else:
        asyncio.run(batch(args))
