"""Run the two public Lightpanda checks and save inspectable output."""
import hashlib
import json
import os
import platform
import re
import subprocess
import sys
import time
from datetime import datetime, timezone
from html.parser import HTMLParser
from pathlib import Path

HERE = Path(__file__).resolve().parent
TARGETS = (
    ("wikipedia", "https://en.wikipedia.org/wiki/Web_scraping", []),
    ("quotes", "https://quotes.toscrape.com/js/", ["--wait-selector", ".quote"]),
)
ANSI = re.compile(rb"\x1b\[[0-9;]*m")


class Markers(HTMLParser):
    def __init__(self):
        super().__init__()
        self.article = 0
        self.quotes = 0
        self.authors = 0
        self.author_names = []
        self.in_author = False

    def handle_starttag(self, tag, attrs):
        attrs = dict(attrs)
        classes = attrs.get("class", "").split()
        self.article += attrs.get("id") == "mw-content-text"
        self.quotes += tag == "div" and "quote" in classes
        if tag == "small" and "author" in classes:
            self.authors += 1
            self.in_author = True

    def handle_data(self, data):
        if self.in_author:
            self.author_names.append(data.strip())

    def handle_endtag(self, tag):
        if tag == "small":
            self.in_author = False


def main(binary_name):
    binary = Path(binary_name).resolve()
    version = subprocess.check_output([str(binary), "version"], text=True).strip()
    sha256 = hashlib.sha256(binary.read_bytes()).hexdigest()
    env = os.environ.copy()
    for key in ("HTTP_PROXY", "HTTPS_PROXY", "ALL_PROXY", "http_proxy", "https_proxy", "all_proxy"):
        env.pop(key, None)
    records = []
    for name, url, wait in TARGETS:
        command = ["/usr/bin/time", "-f", "wall_seconds=%e\\nmax_rss_kb=%M\\nuser_seconds=%U\\nsystem_seconds=%S\\nexit_code=%x",
                   "-o", str(HERE / f"{name}-resource.txt"), "timeout", "30s", str(binary), "fetch", "--obey-robots",
                   "--json", "--dump", "html", *wait, "--log-format", "logfmt", "--log-level", "info", url]
        started = datetime.now(timezone.utc).isoformat()
        before = time.perf_counter()
        proc = subprocess.run(command, capture_output=True, env=env, timeout=35)
        (HERE / f"{name}-log.txt").write_bytes(ANSI.sub(b"", proc.stderr))
        try:
            response = json.loads(proc.stdout)
        except (json.JSONDecodeError, UnicodeDecodeError):
            response = {"http_status": None, "error": "CLI output was not valid JSON", "content": ""}
            (HERE / f"{name}-cli-output.txt").write_bytes(proc.stdout)
        html = response.get("content") or ""
        (HERE / f"{name}-rendered.txt").write_text(html, encoding="utf-8")
        readings = dict(line.split("=", 1) for line in (HERE / f"{name}-resource.txt").read_text().splitlines() if "=" in line)
        markers = Markers()
        markers.feed(html)
        expected = (markers.article > 0 and "Web scraping is" in html) if name == "wikipedia" else (markers.quotes > 0 and "Albert Einstein" in markers.author_names)
        record = {
            "target": name, "url": url, "started_at": started,
            "binary_version": version, "binary_sha256": sha256, "binary_path": str(binary),
            "runtime": platform.platform(), "glibc_version": platform.libc_ver()[1],
            "reproduction_python_version": platform.python_version(), "command": command,
            "proxy_mode": "none; common proxy environment variables removed", "login": False,
            "http_status": response.get("http_status"), "navigation_error": response.get("error"), "exit_code": proc.returncode,
            "elapsed_seconds": round(time.perf_counter() - before, 3),
            "process_wall_seconds": float(readings["wall_seconds"]), "process_peak_rss_kb": int(readings["max_rss_kb"]),
            "rendered_html_bytes": len(html.encode()), "article_container_count": markers.article,
            "rendered_quote_count": markers.quotes, "rendered_author_count": markers.authors,
            "expected_content_found": bool(expected), "usable_content": proc.returncode == 0 and response.get("http_status") == 200 and bool(expected),
            "rendered_capture": f"{name}-rendered.txt", "log": f"{name}-log.txt", "resource_use": f"{name}-resource.txt",
        }
        records.append(record)
        print(f"{name}: HTTP {record['http_status']}, usable={record['usable_content']}, error={record['navigation_error']}")
    (HERE / "results.json").write_text(json.dumps({"runs": records}, indent=2) + "\n", encoding="utf-8")


if __name__ == "__main__":
    main(sys.argv[1])
