"""Summarise results.jsonl: how many homepages need a browser?"""
import json
import statistics
import sys
from collections import Counter
from pathlib import Path

rows = [json.loads(l) for l in (Path(__file__).parent / "results.jsonl").read_text().splitlines() if l.strip()]


import re
host = lambda u: re.sub(r"^https?://(www\\.)?", "", (u or "")).split("/")[0]
HI, LO = 0.6, 0.3  # share of the page's rendered words found without a browser


def group(r):
    if r.get("result") == "skipped_robots":
        return "excluded: robots.txt disallows"
    if "browser_error" in r and "http_status" not in r:
        return "excluded: unreachable"
    if "browser_error" in r:
        return "excluded: browser failed to load"
    browser_blocked = bool(r.get("browser_blocks")) and r.get("rendered_words", 0) < 150
    http_ok = r.get("http_status") == 200 and not r.get("http_blocks") and r.get("http_visible_words", 0) >= 50
    if browser_blocked and http_ok:
        return "blocked for the browser only"
    if (r.get("browser_status") or 0) >= 400 or r.get("rendered_words", 0) < 50:
        if r.get("browser_blocks") and r.get("rendered_words", 0) < 50:
            return "blocked for both"
        return "excluded: no real homepage"
    if browser_blocked:
        return "blocked for both"
    if "http_status" not in r or (r.get("http_status") or 0) >= 400 or r.get("http_blocks"):
        return "blocked for plain HTTP only"
    # AWS WAF answers bots with a 202 challenge page; other tiny empty pages are challenges too.
    # An empty stub under 2.5 KB on the same site was a challenge page each time we checked (all Amazon stores).
    if r.get("http_status") == 202 or (r.get("http_visible_words") == 0 and r.get("http_bytes", 10**9) < 2500
                                       and host(r.get("http_final")) == host(r.get("browser_final"))):
        return "blocked for plain HTTP only"
    if host(r.get("http_final")) != host(r.get("browser_final")):
        return "excluded: browser redirected elsewhere"
    if r.get("cov_visible", 0) >= HI:
        return "in the plain HTML"
    if r.get("cov_source", 0) >= HI:
        return "in the page source (JSON)"
    if r.get("cov_source", 0) < LO:
        return "needs JavaScript"
    return "partly in the HTML"


for r in rows:
    r["group"] = group(r)

c = Counter(r["group"] for r in rows)
print(f"sites: {len(rows)}   thresholds: {HI} / {LO}")
for k, v in sorted(c.items(), key=lambda kv: -kv[1]):
    print(f"  {v:4d}  {k}")

studied = [r for r in rows if not r["group"].startswith("excluded")]
n = len(studied)
print(f"\nstudied (reachable homepages with real content): {n}")
for k in ["in the plain HTML", "in the page source (JSON)", "partly in the HTML", "needs JavaScript", "blocked for plain HTTP only", "blocked for the browser only", "blocked for both"]:
    v = sum(r["group"] == k for r in studied)
    print(f"  {v:4d}  {v / n:6.1%}  {k}")

ok = [r for r in studied if r["group"] in ("in the plain HTML", "in the page source (JSON)", "partly in the HTML", "needs JavaScript")]
print("\nspeed, reachable pages with content:")
print("  median HTTP seconds:   ", statistics.median(r["http_seconds"] for r in ok))
print("  median browser seconds:", statistics.median(r["browser_seconds"] for r in ok))

marks = Counter(m for r in ok for m in r.get("data_markers", []))
print("\nembedded data found in the plain HTML (share of", len(ok), "pages):")
for k, v in marks.most_common():
    print(f"  {v:4d}  {v / len(ok):6.1%}  {k}")

print("\nblock signals seen by plain HTTP (among 'blocked for plain HTTP only'):")
sig = Counter(s for r in studied if r["group"] == "blocked for plain HTTP only" for s in (r.get("http_blocks") or ["no response: " + str(r.get("http_error", ""))[:30]]))
for k, v in sig.most_common(12):
    print(f"  {v:4d}  {k}")

if "--list" in sys.argv:
    for g in ["needs JavaScript", "partly in the HTML", "in the page source (JSON)", "blocked for plain HTTP only", "blocked for the browser only", "blocked for both", "excluded: browser redirected elsewhere"]:
        print(f"\n{g}:", ", ".join(f"{r['domain']}" for r in sorted(rows, key=lambda r: r["rank"]) if r["group"] == g))
