"""Take the first 2,000 HTML pages from one Common Crawl WARC file, and the WET text
Common Crawl made from each of them. Run on October 7, 2026, from Mumbai.
Run with: uv run --with warcio --with requests python sample.py"""
import base64
import gzip
import json

import requests
from warcio.archiveiterator import ArchiveIterator

BASE = "https://data.commoncrawl.org/crawl-data/CC-MAIN-2026-39/segments/1788492699475.64"
NAME = "CC-MAIN-20260904131603-20260904161603-00000"
WANT = 2000

pages = {}
with requests.get(f"{BASE}/warc/{NAME}.warc.gz", stream=True, timeout=120) as r:
    for rec in ArchiveIterator(r.raw):
        if rec.rec_type != "response" or rec.http_headers is None:
            continue
        if rec.http_headers.get_statuscode() != "200":
            continue
        if "text/html" not in (rec.http_headers.get_header("Content-Type") or ""):
            continue
        html = rec.content_stream().read()
        pages[rec.rec_headers.get_header("WARC-Record-ID")] = {
            "url": rec.rec_headers.get_header("WARC-Target-URI"),
            "html_b64": base64.b64encode(html).decode(),  # raw bytes, so each extractor detects the encoding itself
        }
        if len(pages) == WANT:
            break

found = 0
with requests.get(f"{BASE}/wet/{NAME}.warc.wet.gz", stream=True, timeout=120) as r:
    for rec in ArchiveIterator(r.raw):
        if rec.rec_type != "conversion":
            continue
        page = pages.get(rec.rec_headers.get_header("WARC-Refers-To"))
        if page is not None:
            page["wet"] = rec.content_stream().read().decode("utf-8", "replace")
            page["languages"] = rec.rec_headers.get_header("WARC-Identified-Content-Language")
            found += 1
            if found == len(pages):
                break

with gzip.open("sample.jsonl.gz", "wt") as f:
    for page in pages.values():
        f.write(json.dumps(page) + "\n")
print(len(pages), "pages,", found, "with WET text")
