"""What do top sites return for a page that doesn't exist? For each homepage that
returned content in the 1,000-site test, request the homepage and one random path
that can't exist, and record the status, redirects, title and visible text.
Run on October 9, 2026, from Mumbai, with httpx 0.28.
Run with: uv run --with httpx --with beautifulsoup4 python test.py"""
import asyncio
import csv
import json
import secrets

import httpx
from bs4 import BeautifulSoup

UA = "DoccraftResearchBot/1.0 (https://doccraft.dev; hello@doccraft.dev) httpx/0.28"
OK = {"in the plain HTML", "in the page source (JSON)", "partly in the HTML", "needs JavaScript"}
rows = [r for r in csv.DictReader(open("../browser-test-2026/results.csv")) if r["group"] in OK]


def summary(r):
    soup = BeautifulSoup(r.text, "html.parser")
    title = soup.title.get_text(" ", strip=True) if soup.title else ""
    scripts = sum(len(s.get_text()) for s in soup("script"))
    for tag in soup(["script", "style", "noscript", "template"]):
        tag.decompose()
    text = " ".join(soup.get_text(" ").split())
    return {
        "status": r.status_code,
        "final_url": str(r.url),
        "redirects": [h.status_code for h in r.history],
        "title": title[:200],
        "words": len(text.split()),
        "script_chars": scripts,
        "text": text[:3000],
    }


async def check(client, sem, row):
    home = row["http_final"] or f"https://{row['domain']}/"
    missing = str(httpx.URL(home).join(f"/doccraft-test-{secrets.token_hex(6)}"))
    out = {"domain": row["domain"], "missing_url": missing}
    async with sem:
        for kind, url in [("home", home), ("missing", missing)]:
            try:
                out[kind] = summary(await client.get(url))
            except Exception as e:
                out[kind] = {"error": type(e).__name__}
            await asyncio.sleep(1)
    return out


async def main():
    sem = asyncio.Semaphore(10)
    async with httpx.AsyncClient(headers={"User-Agent": UA}, follow_redirects=True, timeout=20) as client:
        results = await asyncio.gather(*(check(client, sem, r) for r in rows))
    json.dump(results, open("results.json", "w"), indent=1)
    print(len(results), "sites")


if __name__ == "__main__":
    asyncio.run(main())
