"""Checks the top 1,000 homepages (Tranco V349N, minus sites whose robots.txt disallows crawling)
for AI-directed HTTP headers (X-AI, X-LLM and similar) and common phrases addressed to AI models.
One GET request per site. Run on October 3, 2026, from Mumbai."""
import asyncio, csv, json, re, sys, time
import httpx
from bs4 import BeautifulSoup, Comment
rows=[r for r in csv.DictReader(open("../browser-test-2026/results.csv"))
      if r["group"]!="excluded: robots.txt disallows"]
UA="MyScraperBot/1.0 (https://example.com/bot; you@example.com) httpx/0.28"
HDR=re.compile(r"^x-(ai|llm|chatgpt|openai|bot-instr|ai-[a-z-]+)$", re.I)
PHR=re.compile(r"(ignore (all |any )?(previous|prior|above) instructions|disregard (all |any )?(previous|prior) instructions|you are (an? )?(ai|language model|llm)|if you are an? (ai|llm|language model|large language model)|attention (ai|llm)|note to (ai|llm)s?|instructions? for (ai|llm)s?|ai agents?:)", re.I)
async def one(c, r, sem):
    async with sem:
        d=r["domain"]; out={"domain":d,"rank":r["rank"]}
        try:
            resp=await c.get(f"https://{d}/", headers={"User-Agent":UA}, follow_redirects=True, timeout=20)
            out["status"]=resp.status_code
            out["headers"]={k:v[:300] for k,v in resp.headers.items() if HDR.match(k)}
            html=resp.text[:3_000_000]
            soup=BeautifulSoup(html,"html.parser")
            hits=[]
            for t in soup.select('script[type="application/ld+json"]'):
                for m in PHR.finditer(t.string or ""): hits.append(("json-ld", (t.string or "")[max(0,m.start()-80):m.end()+120]))
            for cm in soup.find_all(string=lambda x: isinstance(x, Comment)):
                for m in PHR.finditer(cm): hits.append(("comment", cm[max(0,m.start()-80):m.end()+120]))
            for mt in soup.find_all("meta"):
                v=" ".join(str(x) for x in mt.attrs.values())
                for m in PHR.finditer(v): hits.append(("meta", v[:250]))
            for m in PHR.finditer(soup.get_text(" ")): hits.append(("text", soup.get_text(" ")[max(0,m.start()-100):m.end()+120]))
            out["hits"]=hits[:6]
        except Exception as e:
            out["error"]=type(e).__name__
        return out
async def main():
    sem=asyncio.Semaphore(24)
    async with httpx.AsyncClient(http2=False) as c:
        res=await asyncio.gather(*[one(c,r,sem) for r in rows])
    json.dump(res, open("inj_scan.json","w"), indent=1)
    ok=[r for r in res if "status" in r]
    print("sites", len(rows), "responded", len(ok))
    print("header hits:", [(r["domain"], r["headers"]) for r in res if r.get("headers")])
    print("content hits:", len([r for r in res if r.get("hits")]))
    for r in res:
        if r.get("hits"): print(r["domain"], r["hits"][:2])
t=time.time(); asyncio.run(main()); print(f"{time.time()-t:.0f}s")
