"""Do top sites support conditional requests? For each homepage that returned content in
the 1,000-site test, and for one inner page from each site's sitemap, send 2 normal
requests and 2 conditional ones (If-None-Match and If-Modified-Since), and record the
validators, the status codes and whether the visible text changed.
Run on October 8, 2026, from Mumbai, with httpx 0.28.
Run with: uv run --with httpx --with beautifulsoup4 python test.py"""
import asyncio
import csv
import hashlib
import json
import re
import warnings

import httpx
from bs4 import BeautifulSoup, XMLParsedAsHTMLWarning

warnings.filterwarnings("ignore", category=XMLParsedAsHTMLWarning)

UA = "DoccraftResearchBot/1.0 (https://doccraft.dev; hello@doccraft.dev) httpx/0.28"
OK = {"in the plain HTML", "in the page source (JSON)", "partly in the HTML", "needs JavaScript"}
rows = [r for r in csv.DictReader(open("../browser-test-2026/results.csv")) if r["group"] in OK]


def text_hash(html):
    """A hash of the visible text, so tokens in scripts or attributes don't count as changes."""
    soup = BeautifulSoup(html, "html.parser")
    for tag in soup(["script", "style", "noscript", "template"]):
        tag.decompose()
    return hashlib.sha1(" ".join(soup.get_text(" ").split()).encode()).hexdigest()


async def inner_page(client, home):
    """The first page URL in the site's sitemap, following one sitemap index if needed."""
    robots = await client.get(httpx.URL(home).join("/robots.txt"))
    maps = re.findall(r"(?im)^\s*sitemap:\s*(\S+)", robots.text) if robots.status_code == 200 else []
    maps = maps or [str(httpx.URL(home).join("/sitemap.xml"))]
    for depth in range(2):
        r = await client.get(maps[0])
        if r.status_code != 200:
            return None
        locs = re.findall(r"<loc>\s*([^<\s]+)\s*</loc>", r.text)
        if not locs:
            return None
        if "<sitemapindex" in r.text[:2000]:
            maps = locs
            continue
        pages = [u for u in locs if httpx.URL(u).path.strip("/")]
        return pages[0].replace("&amp;", "&") if pages else None
    return None


async def probe(client, url):
    out = {"url": url}
    a = await client.get(url)
    await asyncio.sleep(1)
    b = await client.get(url)
    out.update(
        status=a.status_code,
        content_type=a.headers.get("content-type"),
        bytes=a.num_bytes_downloaded,  # bytes on the wire, compressed if the server compressed them
        etag_1=a.headers.get("etag"), etag_2=b.headers.get("etag"),
        last_modified_1=a.headers.get("last-modified"), last_modified_2=b.headers.get("last-modified"),
        date=b.headers.get("date"),
        cache_control=b.headers.get("cache-control"),
        same_bytes=a.content == b.content,
        same_text=text_hash(a.text) == text_hash(b.text),
    )
    if out["etag_2"]:
        await asyncio.sleep(1)
        c = await client.get(url, headers={"If-None-Match": out["etag_2"]})
        out["if_none_match_status"], out["if_none_match_bytes"] = c.status_code, c.num_bytes_downloaded
    if out["last_modified_2"]:
        await asyncio.sleep(1)
        d = await client.get(url, headers={"If-Modified-Since": out["last_modified_2"]})
        out["if_modified_since_status"], out["if_modified_since_bytes"] = d.status_code, d.num_bytes_downloaded
    return out


async def check(client, sem, row):
    home = row["http_final"] or f"https://{row['domain']}/"
    out = {"domain": row["domain"]}
    async with sem:
        for kind, get_url in [("homepage", lambda: home), ("inner", None)]:
            try:
                url = get_url() if get_url else await inner_page(client, home)
                out[kind] = await probe(client, url) if url else None
            except Exception as e:
                out[kind] = {"error": type(e).__name__}
    return out


async def main():
    sem = asyncio.Semaphore(10)
    async with httpx.AsyncClient(headers={"User-Agent": UA}, follow_redirects=True, timeout=20) as client:
        results = await asyncio.gather(*(check(client, sem, r) for r in rows))
    json.dump(results, open("results.json", "w"), indent=1)
    print(len(results), "sites")


asyncio.run(main())
