import gzip
import json

import httpx
import trafilatura
from resiliparse.extract.html2text import extract_plain_text
from resiliparse.parse.encoding import bytes_to_str, detect_encoding

CRAWL = "CC-MAIN-2026-39"
url = "airshipdaily.com/blog/on-the-fashion-fence-canvas-doc-martens"

# 1. Ask the Common Crawl index where the page is stored.
index = httpx.get(f"https://index.commoncrawl.org/{CRAWL}-index", params={"url": url, "output": "json"}, timeout=60)
hit = json.loads(index.text.splitlines()[0])

# 2. Download only that record from the WARC file, with a range request.
start, length = int(hit["offset"]), int(hit["length"])
record = httpx.get(
    f"https://data.commoncrawl.org/{hit['filename']}",
    headers={"Range": f"bytes={start}-{start + length - 1}"},
    timeout=60,
).content
html = gzip.decompress(record).split(b"\r\n\r\n", 2)[2]  # skip the WARC and HTTP headers

# 3. Extract the main text both ways.
traf = trafilatura.extract(html, favor_precision=True, include_comments=False, deduplicate=True)
resi = extract_plain_text(bytes_to_str(html, detect_encoding(html)), main_content=True)
for name, text in [("trafilatura", traf), ("resiliparse", resi)]:
    print(f"{name:12} {len(text.split()):4} words | {text.strip().splitlines()[0][:60]}")
