"""Requests each homepage that returned a normal page in the October 3, 2026 check (ai-instructions-2026)
with 5 user agents, one request each, 1 second apart. Saves status, bytes and visible word count.
Run on October 3, 2026, from Mumbai, with httpx 0.28."""
import asyncio, json, time
import httpx
from bs4 import BeautifulSoup
sites=[r["domain"] for r in json.load(open("../ai-instructions-2026/results.json")) if r.get("status") == 200]
CH=("Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 "
    "(KHTML, like Gecko) Chrome/140.0.0.0 Safari/537.36")
UAS={"library default":None,"copied from Chrome":CH,"made-up name":"MyScraper/1.0",
     "Chrome + email":f"{CH} (you@example.com)","descriptive":"MyScraperBot/1.0 (https://example.com/bot; you@example.com) httpx/0.28"}
def words(h):
    s=BeautifulSoup(h[:3_000_000],"html.parser")
    for t in s(["script","style","noscript"]): t.decompose()
    return len(s.get_text(" ").split())
async def one(c,d,sem):
    out={"domain":d}
    async with sem:
        for name,ua in UAS.items():
            h={"User-Agent":ua} if ua else {}
            try:
                r=await c.get(f"https://{d}/",headers=h,follow_redirects=True,timeout=25)
                out[name]=[r.status_code,len(r.content),words(r.text)]
            except Exception as e: out[name]=[type(e).__name__,0,0]
            await asyncio.sleep(1.0)
    return out
async def main():
    sem=asyncio.Semaphore(16)
    async with httpx.AsyncClient() as c:
        res=await asyncio.gather(*[one(c,d,sem) for d in sites])
    json.dump(res,open("ua_full.json","w"),indent=1)
t=time.time(); asyncio.run(main()); print(f"{time.time()-t:.0f}s")
