"""How many websites need a browser? Fetch each homepage twice, once with a plain
HTTP client and once in headless Chromium, and compare the text each one gets.

Polite by design: one robots.txt request, one HTTP request and one browser load per
site, public homepages only, sites that disallow "/" for all agents are skipped, and
nothing tries to get past a block (a block is recorded as a result).

Usage: uv run --with httpx --with selectolax --with protego --with playwright python3 study.py [N]
Results: results.jsonl (one line per site; reruns skip sites already done).
"""
import asyncio
import codecs
import json
import re
import sys
import time
from pathlib import Path

import httpx
from playwright.async_api import async_playwright
from protego import Protego
from selectolax.parser import HTMLParser

HERE = Path(__file__).parent
OUT = HERE / "results.jsonl"
N = int(sys.argv[1]) if len(sys.argv) > 1 else 1000

UA = ("Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 "
      "(KHTML, like Gecko) Chrome/140.0.0.0 Safari/537.36")
HEADERS = {
    "User-Agent": UA,
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "en-US,en;q=0.9",
}

BLOCK_MARKERS = [
    "just a moment", "cf-chl", "challenge-platform", "attention required",
    "px-captcha", "captcha-delivery", "datadome", "incapsula", "_incapsula_resource",
    "pardon our interruption", "verify you are a human", "are you a robot",
    "unusual traffic", "access denied", "request blocked", "bot detection",
    "please enable js and disable any ad blocker", "enable javascript and cookies to continue",
]
DATA_MARKERS = {
    "next_data": "__NEXT_DATA__",
    "nuxt": "__NUXT__",
    "json_ld": "application/ld+json",
    "initial_state": "__INITIAL_STATE__",
    "apollo_state": "__APOLLO_STATE__",
    "preloaded_state": "__PRELOADED_STATE__",
    "remix": "__remixContext",
    "rsc_flight": "self.__next_f",
}

CJK = re.compile(r"[぀-ヿ㐀-鿿가-힯]")


def tokens(text: str) -> set[str]:
    """Words of 3+ letters, plus character pairs for Chinese, Japanese and Korean."""
    out = set()
    for w in re.findall(r"\w+", text.lower()):
        if CJK.search(w):
            out.update(w[i:i + 2] for i in range(len(w) - 1))
        elif len(w) >= 3 and not w.isdigit():
            out.add(w)
    return out


def visible_text(html: str) -> str:
    tree = HTMLParser(html)
    for tag in tree.css("script, style, noscript, template, svg"):
        tag.decompose()
    body = tree.body or tree.root
    return body.text(separator=" ") if body else ""


def all_text(html: str) -> str:
    """Everything in the page source, including data inside <script> tags."""
    text = re.sub(r"<[^>]+>", " ", html)
    try:  # JSON in pages often escapes non-ASCII text as \\uXXXX
        text += " " + codecs.decode(re.sub(r"\\(?!u[0-9a-fA-F]{4})", " ", text), "unicode_escape", "ignore")
    except Exception:
        pass
    return text


def block_signals(status, html: str) -> list[str]:
    low = html[:200_000].lower()
    hits = [m for m in BLOCK_MARKERS if m in low]
    if status in (401, 403, 405, 406, 429, 503):
        hits.append(f"status {status}")
    return hits


async def robots_allows(client: httpx.AsyncClient, domain: str) -> tuple[bool, str]:
    try:
        r = await client.get(f"https://{domain}/robots.txt", timeout=10)
        if r.status_code >= 400:
            return True, f"robots {r.status_code}"
        # Protego reads robots.txt the way Google does (longest match wins);
        # Python's built-in parser wrongly treats "Disallow: /?" as "Disallow: /".
        return Protego.parse(r.text).can_fetch(f"https://{domain}/", "*"), "robots ok"
    except Exception as e:
        return True, f"robots error {type(e).__name__}"


async def http_fetch(client: httpx.AsyncClient, domain: str) -> dict:
    last = None
    for url in (f"https://{domain}/", f"https://www.{domain}/"):
        t = time.perf_counter()
        try:
            r = await client.get(url, timeout=20)
            html = r.text
            return {
                "url": url, "final_url": str(r.url), "status": r.status_code,
                "seconds": round(time.perf_counter() - t, 2), "bytes": len(r.content),
                "html": html,
            }
        except Exception as e:
            last = f"{type(e).__name__}: {str(e)[:120]}"
    return {"error": last}


async def browser_fetch(context, url: str) -> dict:
    page = await context.new_page()
    t = time.perf_counter()
    try:
        resp = await page.goto(url, wait_until="domcontentloaded", timeout=30_000)
        try:
            await page.wait_for_load_state("networkidle", timeout=8_000)
        except Exception:
            pass
        await page.wait_for_timeout(1000)
        text = await page.evaluate("document.body ? document.body.innerText : ''")
        html = await page.content()
        return {
            "final_url": page.url, "status": resp.status if resp else None,
            "seconds": round(time.perf_counter() - t, 2), "text": text, "html": html,
        }
    except Exception as e:
        return {"error": f"{type(e).__name__}: {str(e)[:120]}", "seconds": round(time.perf_counter() - t, 2)}
    finally:
        await page.close()


async def study(rank: int, domain: str, client, browser, sem, lock):
    async with sem:
        row = {"rank": rank, "domain": domain}
        allowed, note = await robots_allows(client, domain)
        row["robots"] = note
        if not allowed:
            row["result"] = "skipped_robots"
        else:
            h = await http_fetch(client, domain)
            if "error" in h:
                row["http_error"] = h["error"]
                url = f"https://{domain}/"
            else:
                url = h["url"]
                row.update(http_status=h["status"], http_seconds=h["seconds"], http_bytes=h["bytes"],
                           http_final=h["final_url"], http_blocks=block_signals(h["status"], h["html"]))
            context = await browser.new_context(user_agent=UA, locale="en-US", viewport={"width": 1366, "height": 900})
            b = await browser_fetch(context, url)
            await context.close()
            if "error" in b:
                row["browser_error"] = b["error"]
            else:
                row.update(browser_status=b["status"], browser_seconds=b["seconds"], browser_final=b["final_url"],
                           browser_blocks=block_signals(b["status"], b["html"]))
                R = tokens(b["text"])
                row["rendered_words"] = len(R)
                if "html" in h:
                    Hv, Ha = tokens(visible_text(h["html"])), tokens(all_text(h["html"]))
                    row["http_visible_words"] = len(Hv)
                    if R:
                        row["cov_visible"] = round(len(R & Hv) / len(R), 3)
                        row["cov_source"] = round(len(R & Ha) / len(R), 3)
                    row["data_markers"] = [k for k, v in DATA_MARKERS.items() if v in h["html"]]
                    row["browser_data_markers"] = [k for k, v in DATA_MARKERS.items() if v in b["html"]]
        async with lock:
            with OUT.open("a") as f:
                f.write(json.dumps(row) + "\n")
        print(rank, domain, row.get("cov_visible"), row.get("http_status"), row.get("result", ""), flush=True)


async def main():
    done = set()
    if OUT.exists():
        done = {json.loads(l)["domain"] for l in OUT.read_text().splitlines() if l.strip()}
    sites = []
    for line in (HERE / "top-1m.csv").open():
        rank, domain = line.strip().split(",")
        if int(rank) > N:
            break
        if domain not in done:
            sites.append((int(rank), domain))
    print(f"{len(sites)} sites to test ({len(done)} already done)", flush=True)
    limits = httpx.Limits(max_connections=40)
    async with httpx.AsyncClient(headers=HEADERS, follow_redirects=True, limits=limits, http2=False) as client:
        async with async_playwright() as p:
            browser = await p.chromium.launch()
            sem, lock = asyncio.Semaphore(8), asyncio.Lock()
            await asyncio.gather(*(study(r, d, client, browser, sem, lock) for r, d in sites))
            await browser.close()


if __name__ == "__main__":
    asyncio.run(main())
