"""One isolated Scrapling HTTP Fetcher.get per frozen URL; private captures only.
Requires Python >=3.11, scrapling[fetchers]==0.4.15 and curl_cffi==0.16.3. No browser launches.
"""
import argparse
import asyncio
from collections import Counter
from datetime import datetime, timezone
import hashlib
from html.parser import HTMLParser
from importlib.metadata import version
import json
import os
from pathlib import Path
import platform
import re
import shlex
import signal
import subprocess
import sys
import time
from urllib.parse import urlsplit
from scrapling.fetchers import Fetcher

CONFIG = {
 "mode": "Fetcher.get HTTP only", "proxy_mode": "none", "login": False, "captcha_solving": False,
 "concurrency": 1, "gap_seconds": 2, "application_retries": 0,
 "fresh_process_and_session_per_target": True, "impersonate": "chrome150", "stealthy_headers": True,
 "follow_redirects": True, "max_redirects": 3, "timeout": 20, "attempt_timeout_seconds": 30,
 "max_capture_bytes": 5242880, "network_response_byte_cap": None,
 "cookies": "Fresh temporary curl session; redirect cookies possible; no imported cookies or cross-target persistence",
 "resources": "HTTP response only; no browser, JavaScript, images, stylesheets, iframes or workers",
 "verify": True, "http3": False, "selector_config": {"adaptive": False},
 "pss_sample_interval_seconds": 0.1, "min_host_available_mb": 3072, "max_attempt_tree_pss_mb": 1024,
 "pss_scope": "Isolated Python attempt and descendants, including HTTP client and selector parse; excludes batch parent",
}

BLOCK_PHRASES = (
    "verify you are human", "robot or human", "robot check", "access denied",
    "pardon our interruption", "please complete the captcha", "enter the characters",
    "additional verification required", "unusual traffic", "checking your browser",
    "enable javascript and cookies to continue", "just a moment",
)
GATE_PHRASES = ("log in to continue", "sign in to continue", "sign in to linkedin", "consent required", "before you continue")


def utc():
    return datetime.now(timezone.utc).isoformat()


def sha(data):
    return hashlib.sha256(data).hexdigest()


class PageText(HTMLParser):
    """Source-text approximation, not rendered visibility or structured extraction."""
    SKIP = {"head", "script", "style", "template", "noscript"}

    def __init__(self):
        super().__init__(convert_charrefs=True)
        self.hidden = []
        self.parts = []

    def handle_starttag(self, tag, attrs):
        if tag in self.SKIP:
            self.hidden.append(tag)

    def handle_endtag(self, tag):
        if tag in self.hidden:
            self.hidden = self.hidden[:self.hidden.index(tag)]

    def handle_data(self, data):
        if not self.hidden:
            self.parts.append(data)


def evaluate(body, target):
    parser = PageText()
    parser.feed(body.decode("utf-8", errors="replace"))
    text = " ".join(" ".join(parser.parts).split())
    lower = text.lower()
    checks = [
        {"alternatives": group, "matched": [m for m in group if m.lower() in lower]}
        for group in target["required_text_groups"]
    ]
    regex_checks = [
        {**check, "matches": len(re.findall(check["pattern"], text, flags=re.I))}
        for check in target["text_regex_checks"]
    ]
    expected = (
        len(text) >= target["min_text_chars"]
        and all(check["matched"] for check in checks)
        and all(check["matches"] >= check["min_matches"] for check in regex_checks)
    )
    return {
        "text_chars": len(text), "required_text_checks": checks, "text_regex_checks": regex_checks,
        "min_text_chars": target["min_text_chars"], "has_required_content": bool(expected),
        "soft_block_checks": {"phrases_checked": list(BLOCK_PHRASES), "matched": [p for p in BLOCK_PHRASES if p in lower]},
        "gate_checks": {"phrases_checked": list(GATE_PHRASES), "matched": [p for p in GATE_PHRASES if p in lower]},
    }



def result_class(row):
    if row["error"]:
        return row["error"]["classification"]
    if row["navigation_error"]:
        return "navigation_error"
    if row["capture_error"]:
        return "capture_error"
    if row["status"] is None:
        return "status_unobserved"
    if row["status"] in (401, 403, 999):
        return "http_access_denied"
    if row["status"] == 429:
        return "http_rate_limited"
    if row["status"] >= 500:
        return "http_server_error"
    if row["status"] >= 400:
        return "http_client_error"
    if 300 <= row["status"] < 400:
        return "unresolved_redirect"
    if row["checks"]["soft_block_checks"]["matched"]:
        return "soft_block"
    if not row["checks"]["has_required_content"]:
        return "login_or_consent_gate" if row["checks"]["gate_checks"]["matched"] else "missing_required_content"
    return "usable" if row["status"] == 200 and row["capture_complete"] else "incomplete_capture"


def env_direct():
    env = os.environ.copy()
    for key in list(env):
        if key.lower() in ("http_proxy", "https_proxy", "all_proxy"):
            env.pop(key)
    env.update(PYTHONDONTWRITEBYTECODE="1")
    return env


def memory():
    readings = {line.split(":")[0]: int(line.split()[1]) / 1024
                for line in Path("/proc/meminfo").read_text().splitlines() if line.split(":")[0] in ("MemTotal", "MemAvailable", "SwapTotal", "SwapFree")}
    return {**readings, "unit": "MiB", "measured_at": utc()}


def base_row(target):
    return {"target_id": target["id"], "site": target["name"], "category": target["category"],
            "url": target["url"], "original_url": target["url"], "final_url": None,
            "expected_required_content": target["expected_required_content"], "started_at": utc(),
            "request_started_at": None, "status": None, "request_outcome": "not_started",
            "redirect_history": [], "redirect_outcome": "unobserved", "document_responses": [],
            "explicit_request_headers": {}, "response_metadata": {}, "error": None,
            "navigation_error": None, "capture_error": None, "screenshot_error": None,
            "screenshot_private_path": None, "screenshot_sha256": None,
            "capture_complete": False, "raw_response_sha256": None, "raw_response_private_path": None,
            "response_bytes": 0, "downloaded_body_bytes": 0, "body_complete": False, "navigation_seconds": None, "attempt_seconds": None}


async def single(args, target):
    # Run synchronously in this isolated child; the parent enforces whole-attempt guards.
    row, capture = base_row(target), b""
    row_path = args.output / f"{target['id']}.json"
    def save():
        row_path.write_text(json.dumps(row, indent=2) + "\n")
    start = time.perf_counter()
    row["request_started_at"] = utc()
    save()
    response = None
    request_start = time.perf_counter()
    try:
        response = Fetcher.get(target["url"], timeout=CONFIG["timeout"], retries=0,
            follow_redirects=True, max_redirects=3, impersonate="chrome150", stealthy_headers=True,
            proxy=None, proxies={}, cookies={}, discard_cookies=False, verify=True, http3=False,
            selector_config={"adaptive": False})
        row["request_outcome"] = "returned"
        row["capture_complete"] = True
    except Exception as exc:
        message = str(exc).splitlines()[0][:500]
        code = getattr(exc, "code", None)
        category = "redirect_limit_or_loop" if code == 47 else "timeout" if code == 28 or "timed out" in message.lower() else "transport_error"
        row["request_outcome"] = "error"
        row["error"] = {"type": type(exc).__name__, "message": message, "classification": category, "curl_code": code}
        response = getattr(exc, "response", None)
    row["latency_seconds"] = round(time.perf_counter() - request_start, 3)
    if response is not None:
        row["status"] = getattr(response, "status", None) or getattr(response, "status_code", None)
        row["final_url"] = str(response.url)
        capture = getattr(response, "body", None)
        if capture is None:
            capture = response.content
        row["downloaded_body_bytes"] = len(capture)
        history = list(response.history)
        row["redirect_history"] = [{"previous": h.url,
            "url": history[i+1].url if i+1 < len(history) else row["final_url"],
            "status": getattr(h, "status", None) or getattr(h, "status_code", None)} for i,h in enumerate(history)]
        row["redirect_outcome"] = "followed" if history else "none" if row["final_url"] == target["url"] else "final_url_changed_history_unavailable"
        if row["error"] and row["error"]["classification"] == "redirect_limit_or_loop":
            row["redirect_outcome"] = "limit_or_loop_error"
        row["response_metadata"] = {"content_type": response.headers.get("content-type"),
            "content_encoding": response.headers.get("content-encoding"), "cookie_values": "private/not published"}
        explicit = getattr(response, "request_headers", {})
        row["explicit_request_headers"] = {k:v for k,v in explicit.items() if k.lower() not in ("cookie", "authorization", "proxy-authorization")}
        if len(capture) > CONFIG["max_capture_bytes"]:
            capture = capture[:CONFIG["max_capture_bytes"]]
            row["capture_complete"] = False
            row["capture_error"] = {"type": "BodyCaptureLimit", "message": "Downloaded body exceeded 5 MiB; saved prefix only. Transfer itself is bounded by time/RAM, not bytes."}
        dest = args.output / f"{target['id']}.body"
        dest.write_bytes(capture)
        row.update(raw_response_private_path=str(dest), raw_response_sha256=sha(capture), response_bytes=len(capture))
    else:
        row["redirect_outcome"] = "no_response"
    row.update(finished_at=utc(), attempt_seconds=round(time.perf_counter()-start,3), checks=evaluate(capture,target))
    row["empty_content"] = row["checks"]["text_chars"] == 0
    row["body_complete"] = row["capture_complete"]
    row["classification"] = result_class(row)
    row["usable_content"] = row["classification"] == "usable"
    save()


def tree_members(root):
    pending, seen, members = [root], set(), {}
    while pending:
        pid = pending.pop()
        if pid in seen:
            continue
        seen.add(pid)
        try:
            for task in Path(f"/proc/{pid}/task").iterdir():
                pending.extend(int(p) for p in (task / "children").read_text().split())
            stat = Path(f"/proc/{pid}/stat").read_text().rsplit(")", 1)[1].split()
            members[pid] = stat[19]  # start time; protects cleanup from PID reuse
        except (OSError, ValueError, IndexError):
            continue
    return members


def pss_members(members):
    total, read = 0, 0
    for pid in members:
        try:
            match = re.search(r"^Pss:\s+(\d+)", Path(f"/proc/{pid}/smaps_rollup").read_text(), re.M)
            if match:
                total += int(match.group(1))
                read += 1
        except OSError:
            continue
    return total / 1024 if read else None


def signal_owned(members, sig):
    # Firefox may use its own process group. Signal only recorded descendants.
    for pid, birth in reversed(list(members.items())):
        try:
            current = Path(f"/proc/{pid}/stat").read_text().rsplit(")", 1)[1].split()[19]
            if current == birth:
                os.kill(pid, sig)
        except (OSError, IndexError):
            pass


async def batch(args):
    manifest_bytes = args.manifest.read_bytes()
    manifest = json.loads(manifest_bytes)
    targets = manifest["targets"]
    assert len(targets) == len({t["id"] for t in targets}) == len({urlsplit(t["url"]).hostname for t in targets}) == 30
    assert version("scrapling") == "0.4.15"
    assert version("curl_cffi") == "0.16.3"
    initial_ram = memory()
    if initial_ram["MemAvailable"] < CONFIG["min_host_available_mb"]:
        raise RuntimeError("Unsafe RAM blocker before batch: less than 3 GiB available")
    os.umask(0o077)
    args.output.mkdir(parents=True, exist_ok=False, mode=0o700)
    run = {"run_id": args.output.name, "tool": "scrapling", "tool_version": version("scrapling"),
           "curl_cffi_version": version("curl_cffi"), "mode": CONFIG["mode"],
           "python_version": platform.python_version(), "platform": platform.platform(),
           "network": "Same local host direct outbound; http_proxy/https_proxy/all_proxy removed case-insensitively; no configured proxies. Different UTC window from prior tools.",
           "manifest_id": manifest["id"], "manifest_sha256": sha(manifest_bytes),
           "script_sha256": sha(Path(__file__).read_bytes()), "config": CONFIG,
           "executed_command": shlex.join([sys.executable, *sys.argv]), "started_at": utc(),
           "host_ram_preflight": initial_ram, "attempts": [],
           "mode_selection": "Documented basic Fetcher.get is a practical bounded HTTP-only slice comparable in workload to wreq. It does not evaluate Scrapling browser modes, crawling or adaptive extraction; no per-target switching.",
           "body_hash_basis": "Original client-decompressed HTTP bytes, not serialized browser DOM. Partial/absent captures explicit. Full bodies/logs private in ignored runner/data.",
           "pss_caveat": "100 ms samples may miss peaks or changing processes. Includes isolated Python import, Scrapling/curl HTTP request and selector parse; excludes batch parent. Different workload from browser PSS.",
           "policy_differences": [
              "Fresh temporary Fetcher curl session per isolated process; redirect cookies possible, no cross-target state. wreq cookie storage is disabled; browser baselines have fresh in-session cookies.",
              "Normal redirects followed, cap 3 like corrected wreq/Camoufox; Lightpanda cap 10 and Patchright cap 20. Curl redirect history has statuses/URLs but not intermediate response bodies.",
              "Chrome150 TLS/HTTP impersonation and default stealthy_headers (Google Referer); wreq emulates Chrome134. Implicit curl-generated browser headers are not exposed in Scrapling request_headers; only explicit headers recorded.",
              "No JavaScript or browser resource loads. 20-second transfer deadline; latency also includes Scrapling selector construction. 30-second whole-attempt watchdog plus RAM/PSS guards includes interpreter/import work.",
              "5 MiB saved-body cap after fetch returns, not a network byte cap. Download/parse memory bounded by whole-attempt time and sampled 1 GiB process-tree guard; host available RAM must remain above 3 GiB.",
              "Proxy variables removed; Scrapling per-request API does not accept curl Session trust_env. TLS verify=True is separate from proxy environment behavior. No login, CAPTCHA solving, paid service, browser mode or application retry.",
              "One dated observation per target; time/network responses, TLS/header fingerprint, cookie and JS policies differ. No winner, reliability estimate or causal claim."]}
    def save_run():
        run["summary"] = {"attempted_distinct_sites": len(run["attempts"]), "usable_results": sum(a["usable_content"] for a in run["attempts"]),
                          "classifications": dict(Counter(a["classification"] for a in run["attempts"]))}
        (args.output / "results.json").write_text(json.dumps(run, indent=2) + "\n")
    for index, target in enumerate(targets):
        if index:
            await asyncio.sleep(CONFIG["gap_seconds"])
        before_ram = memory()
        if before_ram["MemAvailable"] < CONFIG["min_host_available_mb"]:
            run["blocker"] = "Unsafe host RAM before next navigation"
            save_run()
            raise RuntimeError(run["blocker"])
        command = [sys.executable, str(Path(__file__).resolve()), "--manifest", str(args.manifest), "--output", str(args.output), "--single", target["id"]]
        with open(args.output / f"{target['id']}-driver.log", "wb") as log:
            proc = await asyncio.create_subprocess_exec(*command, stdout=log, stderr=log, env=env_direct(), start_new_session=True)
            started = time.perf_counter()
            samples, guard, owned = [], None, {}
            while proc.returncode is None:
                members = tree_members(proc.pid)
                owned.update(members)
                reading = pss_members(members)
                ram = memory()
                if reading is not None:
                    samples.append({"elapsed_seconds": round(time.perf_counter() - started, 3), "pss_mb": round(reading, 3), "host_available_mb": round(ram["MemAvailable"], 3)})
                if ram["MemAvailable"] < CONFIG["min_host_available_mb"] or (reading or 0) > CONFIG["max_attempt_tree_pss_mb"]:
                    guard = "unsafe_memory_guard"
                elif time.perf_counter() - started > CONFIG["attempt_timeout_seconds"]:
                    guard = "attempt_timeout"
                if guard:
                    signal_owned(owned, signal.SIGTERM)
                    await asyncio.sleep(0.2)
                    signal_owned(owned, signal.SIGKILL)
                    break
                await asyncio.sleep(CONFIG["pss_sample_interval_seconds"])
            await proc.wait()
            signal_owned(owned, signal.SIGKILL)
        row_path = args.output / f"{target['id']}.json"
        row = json.loads(row_path.read_text()) if row_path.exists() else base_row(target)
        if guard or proc.returncode != 0:
            row["error"] = {"type": "AttemptGuard" if guard else "AttemptProcessExit", "message": guard or f"Attempt process exit {proc.returncode}", "classification": guard or "attempt_process_error"}
        saved_body = args.output / f"{target['id']}.body"
        row.setdefault("checks", evaluate(saved_body.read_bytes() if saved_body.exists() else b"", target))
        row["empty_content"] = row["checks"]["text_chars"] == 0
        row.setdefault("latency_seconds", round(time.perf_counter() - started, 3))
        row.setdefault("finished_at", utc())
        row.update(peak_sampled_tree_pss_mb=max((s["pss_mb"] for s in samples), default=None), pss_samples=len(samples),
                   attempt_process_exit=proc.returncode, host_ram_before=before_ram)
        row["classification"] = result_class(row)
        row["usable_content"] = row["classification"] == "usable"
        (args.output / f"{target['id']}-pss.json").write_text(json.dumps(samples) + "\n")
        row_path.write_text(json.dumps(row, indent=2) + "\n")
        run["attempts"].append(row)
        save_run()
        print(f"{target['id']}: HTTP {row['status']} | {row['classification']} | {row['response_bytes']} body bytes | PSS {row['peak_sampled_tree_pss_mb']} MiB", flush=True)
        if row["request_started_at"] is None or guard == "unsafe_memory_guard":
            run["blocker"] = "Setup or unsafe memory blocker: " + str(row["error"])
            save_run()
            raise RuntimeError(run["blocker"])
    run["finished_at"] = utc()
    save_run()
    print(json.dumps(run["summary"]), flush=True)


if __name__ == "__main__":
    parser = argparse.ArgumentParser(description=__doc__)
    parser.add_argument("--manifest", type=Path, required=True)
    parser.add_argument("--output", type=Path, required=True)
    parser.add_argument("--single")
    args = parser.parse_args()
    if args.single:
        target = next(t for t in json.loads(args.manifest.read_text())["targets"] if t["id"] == args.single)
        asyncio.run(single(args, target))
    else:
        asyncio.run(batch(args))
