"""For the sites that returned a 200 for a missing page, request a second random
missing path. If both answers look the same (same final path pattern, title and
size), a crawler can learn the site's "not found" fingerprint once and recognise it.
Run on October 9, 2026, from Mumbai, with httpx 0.28.
Run with: uv run --with httpx --with beautifulsoup4 python calibrate.py"""
import asyncio
import json
import secrets
import warnings

import httpx
from bs4 import XMLParsedAsHTMLWarning

from test import UA, summary

warnings.filterwarnings("ignore", category=XMLParsedAsHTMLWarning)
R = json.load(open("results.json"))
soft = [r for r in R if "error" not in r["home"] and r["home"].get("status") == 200 and r["missing"].get("status") == 200]


async def main():
    out = []
    async with httpx.AsyncClient(headers={"User-Agent": UA}, follow_redirects=True, timeout=20) as client:
        for r in soft:
            url = str(httpx.URL(r["missing_url"]).join(f"/doccraft-test-{secrets.token_hex(6)}"))
            try:
                out.append({"domain": r["domain"], "second_url": url, "second": summary(await client.get(url))})
            except Exception as e:
                out.append({"domain": r["domain"], "second_url": url, "second": {"error": type(e).__name__}})
            await asyncio.sleep(1)
    json.dump(out, open("calibrate.json", "w"), indent=1)
    print(len(out), "sites")


asyncio.run(main())
