"""Classify a few real URLs: learn each site's homepage and its answer to a random
missing URL once, then classify the page you asked for.
Run with: uv run --with httpx --with beautifulsoup4 python check.py"""
import secrets

import httpx

from classify import classify
from test import UA, summary

urls = [
    "https://stripe.com/pricing",                         # a real page
    "https://store.steampowered.com/app/9999999999/",     # a game that doesn't exist
    "https://kick.com/no-such-channel-4821",              # a channel that doesn't exist
    "https://www.instagram.com/no-such-profile-4821/",    # a profile that doesn't exist
]

with httpx.Client(headers={"User-Agent": UA}, follow_redirects=True, timeout=20) as client:
    for url in urls:
        site = httpx.URL(url).join("/")
        home = summary(client.get(site))
        missing = summary(client.get(site.join(f"/missing-{secrets.token_hex(6)}")))
        page = summary(client.get(url))
        state = classify(page, url, home, missing)
        print(f"{page['status']} -> {state:<10} {url}")
