{
  "id": "public-30-v1",
  "version": "2026-09-29",
  "description": "Frozen 30 distinct public websites for matched tool attempts. Organization pages only on social networks; one URL per registrable domain.",
  "validator_basis": "HTML text excluding head, script, style, template and noscript; no JavaScript execution",
  "targets": [
    {
      "id": "kohls",
      "name": "Kohl’s",
      "category": "commerce",
      "url": "https://www.kohls.com/catalog/mens-tshirts-clothing.jsp?CN=Gender:Mens+Product:T-Shirts+Department:Clothing",
      "expected_required_content": "Men’s T-shirt product listing with prices",
      "required_text_groups": [
        [
          "t-shirt",
          "tees"
        ],
        [
          "men"
        ]
      ],
      "min_text_chars": 300,
      "text_regex_checks": [
        {
          "pattern": "[$£€]\\s?\\d[\\d,.]*",
          "min_matches": 3
        }
      ]
    },
    {
      "id": "amazon",
      "name": "Amazon",
      "category": "commerce",
      "url": "https://www.amazon.com/s?k=laptop",
      "expected_required_content": "Laptop search results with product prices",
      "required_text_groups": [
        [
          "laptop"
        ],
        [
          "results"
        ]
      ],
      "min_text_chars": 500,
      "text_regex_checks": [
        {
          "pattern": "[$£€]\\s?\\d[\\d,.]*",
          "min_matches": 3
        }
      ]
    },
    {
      "id": "walmart",
      "name": "Walmart",
      "category": "commerce",
      "url": "https://www.walmart.com/search?q=laptop",
      "expected_required_content": "Laptop search results with product prices",
      "required_text_groups": [
        [
          "laptop"
        ]
      ],
      "min_text_chars": 500,
      "text_regex_checks": [
        {
          "pattern": "[$£€]\\s?\\d[\\d,.]*",
          "min_matches": 3
        }
      ]
    },
    {
      "id": "ebay",
      "name": "eBay",
      "category": "commerce",
      "url": "https://www.ebay.com/sch/i.html?_nkw=laptop",
      "expected_required_content": "Laptop marketplace listings with prices",
      "required_text_groups": [
        [
          "laptop"
        ],
        [
          "results"
        ]
      ],
      "min_text_chars": 500,
      "text_regex_checks": [
        {
          "pattern": "[$£€]\\s?\\d[\\d,.]*",
          "min_matches": 3
        }
      ]
    },
    {
      "id": "target",
      "name": "Target",
      "category": "commerce",
      "url": "https://www.target.com/s?searchTerm=laptop",
      "expected_required_content": "Laptop search listing with prices",
      "required_text_groups": [
        [
          "laptop"
        ]
      ],
      "min_text_chars": 500,
      "text_regex_checks": [
        {
          "pattern": "[$£€]\\s?\\d[\\d,.]*",
          "min_matches": 3
        }
      ]
    },
    {
      "id": "etsy",
      "name": "Etsy",
      "category": "commerce",
      "url": "https://www.etsy.com/search?q=ceramic+mug",
      "expected_required_content": "Ceramic mug listings with prices",
      "required_text_groups": [
        [
          "ceramic"
        ],
        [
          "mug"
        ]
      ],
      "min_text_chars": 500,
      "text_regex_checks": [
        {
          "pattern": "[$£€]\\s?\\d[\\d,.]*",
          "min_matches": 3
        }
      ]
    },
    {
      "id": "bestbuy",
      "name": "Best Buy",
      "category": "commerce",
      "url": "https://www.bestbuy.com/site/searchpage.jsp?st=laptop",
      "expected_required_content": "Laptop product listing with prices",
      "required_text_groups": [
        [
          "laptop"
        ]
      ],
      "min_text_chars": 500,
      "text_regex_checks": [
        {
          "pattern": "[$£€]\\s?\\d[\\d,.]*",
          "min_matches": 3
        }
      ]
    },
    {
      "id": "linkedin",
      "name": "LinkedIn",
      "category": "organization",
      "url": "https://www.linkedin.com/company/nasa/",
      "expected_required_content": "Public NASA organization description and industry/topic",
      "required_text_groups": [
        [
          "nasa",
          "national aeronautics"
        ],
        [
          "space",
          "research",
          "aviation"
        ]
      ],
      "min_text_chars": 500,
      "text_regex_checks": []
    },
    {
      "id": "instagram",
      "name": "Instagram",
      "category": "organization",
      "url": "https://www.instagram.com/nasa/",
      "expected_required_content": "NASA organization profile with follower and post information",
      "required_text_groups": [
        [
          "nasa"
        ],
        [
          "followers"
        ],
        [
          "posts"
        ]
      ],
      "min_text_chars": 150,
      "text_regex_checks": []
    },
    {
      "id": "tiktok",
      "name": "TikTok",
      "category": "organization",
      "url": "https://www.tiktok.com/@nasa",
      "expected_required_content": "NASA organization profile with video and audience information",
      "required_text_groups": [
        [
          "nasa"
        ],
        [
          "followers"
        ],
        [
          "videos",
          "likes"
        ]
      ],
      "min_text_chars": 150,
      "text_regex_checks": []
    },
    {
      "id": "g2",
      "name": "G2",
      "category": "reviews",
      "url": "https://www.g2.com/products/asana/reviews",
      "expected_required_content": "Asana review listing with review and rating information",
      "required_text_groups": [
        [
          "asana"
        ],
        [
          "reviews"
        ],
        [
          "rating",
          "rated",
          "stars"
        ]
      ],
      "min_text_chars": 500,
      "text_regex_checks": []
    },
    {
      "id": "trustpilot",
      "name": "Trustpilot",
      "category": "reviews",
      "url": "https://www.trustpilot.com/review/www.amazon.com",
      "expected_required_content": "Amazon company review listing with ratings",
      "required_text_groups": [
        [
          "amazon"
        ],
        [
          "reviews"
        ],
        [
          "trustscore",
          "rating",
          "rated"
        ]
      ],
      "min_text_chars": 500,
      "text_regex_checks": []
    },
    {
      "id": "tripadvisor",
      "name": "Tripadvisor",
      "category": "reviews",
      "url": "https://www.tripadvisor.com/Restaurants-g187147-Paris_Ile_de_France.html",
      "expected_required_content": "Paris restaurant listing with reviews",
      "required_text_groups": [
        [
          "paris"
        ],
        [
          "restaurants"
        ],
        [
          "reviews"
        ]
      ],
      "min_text_chars": 500,
      "text_regex_checks": []
    },
    {
      "id": "indeed",
      "name": "Indeed",
      "category": "jobs",
      "url": "https://www.indeed.com/jobs?q=python&l=Remote",
      "expected_required_content": "Remote Python job listings",
      "required_text_groups": [
        [
          "python"
        ],
        [
          "remote"
        ],
        [
          "jobs"
        ]
      ],
      "min_text_chars": 500,
      "text_regex_checks": [
        {
          "pattern": "\\b(?:developer|engineer|programmer)\\b",
          "min_matches": 3
        }
      ]
    },
    {
      "id": "remoteok",
      "name": "Remote OK",
      "category": "jobs",
      "url": "https://remoteok.com/remote-dev-jobs",
      "expected_required_content": "Remote developer job listings",
      "required_text_groups": [
        [
          "remote"
        ],
        [
          "developer",
          "engineer"
        ],
        [
          "jobs"
        ]
      ],
      "min_text_chars": 500,
      "text_regex_checks": [
        {
          "pattern": "\\b(?:developer|engineer|programmer)\\b",
          "min_matches": 3
        }
      ]
    },
    {
      "id": "weworkremotely",
      "name": "We Work Remotely",
      "category": "jobs",
      "url": "https://weworkremotely.com/categories/remote-programming-jobs",
      "expected_required_content": "Remote programming jobs with actual role descriptions",
      "required_text_groups": [
        [
          "remote"
        ],
        [
          "programming"
        ],
        [
          "developer",
          "engineer"
        ]
      ],
      "min_text_chars": 500,
      "text_regex_checks": [
        {
          "pattern": "\\b(?:developer|engineer|programmer)\\b",
          "min_matches": 3
        }
      ]
    },
    {
      "id": "wikipedia",
      "name": "Wikipedia",
      "category": "content",
      "url": "https://en.wikipedia.org/wiki/Web_scraping",
      "expected_required_content": "Web scraping article with substantive explanation",
      "required_text_groups": [
        [
          "web scraping"
        ],
        [
          "web crawler",
          "data extraction"
        ]
      ],
      "min_text_chars": 3000,
      "text_regex_checks": []
    },
    {
      "id": "hackernews",
      "name": "Hacker News",
      "category": "content",
      "url": "https://news.ycombinator.com/",
      "expected_required_content": "News story listing with points and comment links",
      "required_text_groups": [
        [
          "hacker news"
        ],
        [
          "points"
        ],
        [
          "comments"
        ]
      ],
      "min_text_chars": 1000,
      "text_regex_checks": []
    },
    {
      "id": "bbc",
      "name": "BBC",
      "category": "content",
      "url": "https://www.bbc.com/news",
      "expected_required_content": "Public news headlines and topic navigation",
      "required_text_groups": [
        [
          "bbc"
        ],
        [
          "news"
        ],
        [
          "world",
          "business"
        ]
      ],
      "min_text_chars": 1000,
      "text_regex_checks": []
    },
    {
      "id": "arxiv",
      "name": "arXiv",
      "category": "content",
      "url": "https://arxiv.org/list/cs/recent",
      "expected_required_content": "Recent Computer Science papers with subject metadata",
      "required_text_groups": [
        [
          "computer science"
        ],
        [
          "subjects:"
        ]
      ],
      "min_text_chars": 1500,
      "text_regex_checks": []
    },
    {
      "id": "gutenberg",
      "name": "Project Gutenberg",
      "category": "content",
      "url": "https://www.gutenberg.org/ebooks/1342",
      "expected_required_content": "Pride and Prejudice book description with author",
      "required_text_groups": [
        [
          "pride and prejudice"
        ],
        [
          "jane austen"
        ]
      ],
      "min_text_chars": 500,
      "text_regex_checks": []
    },
    {
      "id": "govuk",
      "name": "GOV.UK",
      "category": "content",
      "url": "https://www.gov.uk/bank-holidays",
      "expected_required_content": "UK bank holiday dates for England and Wales",
      "required_text_groups": [
        [
          "uk bank holidays"
        ],
        [
          "england and wales"
        ],
        [
          "christmas day"
        ]
      ],
      "min_text_chars": 500,
      "text_regex_checks": []
    },
    {
      "id": "python",
      "name": "Python documentation",
      "category": "content",
      "url": "https://docs.python.org/3/library/urllib.request.html",
      "expected_required_content": "urllib.request documentation with API explanation",
      "required_text_groups": [
        [
          "urllib.request"
        ],
        [
          "urlopen"
        ]
      ],
      "min_text_chars": 3000,
      "text_regex_checks": []
    },
    {
      "id": "quotes",
      "name": "Quotes to Scrape",
      "category": "javascript",
      "url": "https://quotes.toscrape.com/js/",
      "expected_required_content": "Quote text and author after JavaScript rendering",
      "required_text_groups": [
        [
          "the world as we have created it"
        ],
        [
          "albert einstein"
        ]
      ],
      "min_text_chars": 300,
      "text_regex_checks": []
    },
    {
      "id": "scrapingcourse",
      "name": "ScrapingCourse",
      "category": "javascript",
      "url": "https://www.scrapingcourse.com/javascript-rendering",
      "expected_required_content": "JavaScript product listing with named products and prices",
      "required_text_groups": [
        [
          "chaz kangeroo hoodie",
          "teton pullover hoodie"
        ]
      ],
      "min_text_chars": 300,
      "text_regex_checks": [
        {
          "pattern": "[$£€]\\s?\\d[\\d,.]*",
          "min_matches": 3
        }
      ]
    },
    {
      "id": "webscrapingdev",
      "name": "Web Scraping Dev",
      "category": "javascript",
      "url": "https://web-scraping.dev/reviews",
      "expected_required_content": "Review text loaded by the JavaScript review interface",
      "required_text_groups": [
        [
          "unique flavor and great energy boost",
          "excellent energy drink for gamers"
        ]
      ],
      "min_text_chars": 300,
      "text_regex_checks": []
    },
    {
      "id": "example",
      "name": "Example Domain",
      "category": "control",
      "url": "https://example.com/",
      "expected_required_content": "Example Domain heading and documentation explanation",
      "required_text_groups": [
        [
          "example domain"
        ],
        [
          "documentation examples"
        ]
      ],
      "min_text_chars": 100,
      "text_regex_checks": []
    },
    {
      "id": "httpbingo",
      "name": "HTTPBingo",
      "category": "control",
      "url": "https://httpbingo.org/html",
      "expected_required_content": "Moby-Dick HTML sample",
      "required_text_groups": [
        [
          "herman melville"
        ],
        [
          "moby-dick"
        ]
      ],
      "min_text_chars": 300,
      "text_regex_checks": []
    },
    {
      "id": "scrapethissite",
      "name": "Scrape This Site",
      "category": "control",
      "url": "https://www.scrapethissite.com/pages/simple/",
      "expected_required_content": "Country records including Andorra and population",
      "required_text_groups": [
        [
          "countries of the world"
        ],
        [
          "andorra"
        ],
        [
          "population"
        ]
      ],
      "min_text_chars": 1000,
      "text_regex_checks": []
    },
    {
      "id": "iana",
      "name": "IANA",
      "category": "control",
      "url": "https://www.iana.org/domains/reserved",
      "expected_required_content": "Reserved-domain reference and example domains",
      "required_text_groups": [
        [
          "iana-managed reserved domains"
        ],
        [
          "example.com"
        ]
      ],
      "min_text_chars": 500,
      "text_regex_checks": []
    }
  ],
  "request_policy": {
    "method": "GET",
    "redirects": "follow",
    "max_redirects": 3,
    "timeout_seconds": 20,
    "max_body_bytes": 5242880,
    "application_retries": 0,
    "gap_seconds": 2
  },
  "policy_revision": "2026-09-29-redirects-3"
}
