"""Classify a scraped response into a terminal state: success, not_found, block,
login_required, client_error, server_error or unclear. Heuristics first, then a per-site
"not found" fingerprint learned from one random URL."""
import re
from urllib.parse import urlparse

NOT_FOUND_TEXT = re.compile(
    r"not found|\b404\b|doesn.t exist|does not exist|no longer (exists|available)|nothing (was )?found",
    re.I,
)
BLOCK_TEXT = re.compile(
    r"captcha|robot or human|are you a robot|verify (that )?you are human|access denied|client challenge|just a moment",
    re.I,
)
LOGIN_PATH = re.compile(r"/(login|signin|sign-in|auth|oauth2?)\b", re.I)


def path(url):
    return urlparse(url).path.rstrip("/")


def looks_alike(a, b):
    """Same title and a similar amount of text: the same page template."""
    return a["title"] == b["title"] and abs(a["words"] - b["words"]) <= max(5, 0.15 * max(a["words"], b["words"]))


def classify(resp, requested_url, home=None, missing=None):
    """resp, home and missing are page summaries: status, final_url, title, words, script_chars, text.
    home is the site's homepage; missing is its answer to a random URL that can't exist."""
    status, top = resp["status"], resp["title"] + " " + resp["text"][:600]
    if status in (404, 410):
        return "not_found"
    if status >= 500:
        return "server_error"
    if status in (403, 429) or BLOCK_TEXT.search(top) or (status == 202 and resp["words"] < 50):
        return "block"  # a 202 with an almost empty page is a challenge (AWS WAF answers bots this way)
    if 400 <= status < 500:
        return "client_error"
    if LOGIN_PATH.search(path(resp["final_url"])) and not LOGIN_PATH.search(path(requested_url)):
        return "login_required"
    landed = path(resp["final_url"])
    if home and landed != path(requested_url) and landed in ("", path(home["final_url"])):
        return "not_found"  # a deep URL that redirected to the homepage
    if NOT_FOUND_TEXT.search(top):
        return "not_found"
    # The site's "not found" page, learned from one random URL. Skip it when that URL just
    # redirected to the homepage: the redirect check above already covers those sites.
    learned = missing and missing["status"] == 200 and not (home and path(missing["final_url"]) == path(home["final_url"]))
    if learned and looks_alike(resp, missing):
        # If the homepage looks the same too, the site serves the same HTML for every URL,
        # and the HTML alone can't say whether this page exists.
        return "unclear" if home and looks_alike(home, missing) else "not_found"
    return "success"
