onesvm-browser-server/bench/site-matrix/classify.py
chii 983259836d chore: init workspace with onesvm-dev-md + casa-commander
docs: 联网搜索服务架构方案全套(plan-final/design-arch/选型决策/整合导览/MCP文档/部署预设/联调手册)
bench: 5 方案 + 代理 + 站点矩阵本机实测工程(无密钥)
部署目标:primary mgr1 先行测试(待批准后执行)
2026-09-01 15:19:52 +08:00

131 lines
4.8 KiB
Python

#!/usr/bin/env python3
"""Shared intercept / title / extractability heuristics. No secrets."""
from __future__ import annotations
import re
from urllib.parse import urlparse
BLOCK_TITLE = re.compile(
r"(just a moment|attention required|access denied|access to this page has been denied|"
r"verify you are human|are you a human|robot check|pardon our interruption|"
r"please enable cookies|security check|sorry.? something went wrong|"
r"click the button below|continue shopping|unusual traffic)",
re.I,
)
def _blob(*parts: str) -> str:
return "\n".join(p or "" for p in parts).lower()
def classify_intercept(
*,
http_status: int | None,
final_url: str,
start_url: str,
title: str,
text: str,
html_head: str,
headers: dict | None = None,
error: str | None = None,
redirect_hops: int = 0,
) -> str:
headers = {str(k).lower(): str(v) for k, v in (headers or {}).items()}
blob = _blob(title, text, html_head, " ".join(headers.values()), error or "")
hdrs = " ".join(f"{k}:{v}" for k, v in headers.items())
if BLOCK_TITLE.search(title or "") or "cf-challenge" in blob or "cf-turnstile" in blob:
if "cloudflare" in blob or "cf-ray" in hdrs.lower() or "cf-mitigated" in hdrs.lower():
return "cloudflare"
if "cloudflare" in blob and ("just a moment" in blob or "attention required" in blob):
return "cloudflare"
if "cf-ray" in hdrs.lower() and (http_status in (403, 503, 429) or "just a moment" in blob):
return "cloudflare"
if any(x in blob for x in ("recaptcha", "hcaptcha", "cf-turnstile", "geetest", "验证码")):
if any(x in blob for x in ("captcha", "turnstile", "verify you are", "人机验证", "安全验证")):
return "captcha"
if re.search(r"\bcaptcha\b", blob) and any(
x in blob for x in ("robot", "verify", "human", "enter the characters", "opfcaptcha")
):
return "captcha"
if any(
x in blob
for x in (
"opfcaptcha",
"validatecaptcha",
"amazon.com/errors",
"click the button below to continue shopping",
"datadome",
"captcha-delivery",
"akamai",
"access denied",
"request blocked",
"robot check",
)
):
return "waf"
if http_status in (403, 429) and any(x in blob for x in ("waf", "forbidden", "blocked", "denied")):
return "waf"
if http_status == 403 and ("reddit" in (start_url or "") or "www.reddit.com" in (final_url or "")):
return "waf"
if any(x in blob for x in ("paywall", "subscribe to continue", "become a member to", "metered paywall")):
return "paywall"
start_host = (urlparse(start_url).hostname or "").lower().removeprefix("www.")
final_host = (urlparse(final_url or start_url).hostname or "").lower().removeprefix("www.")
path = (urlparse(final_url or "").path or "").lower()
loginish = any(x in path for x in ("/login", "/signin", "/checkpoint", "/account/login", "/passport"))
if loginish and (http_status in (200, 302, 301, 303, 307, 308) or redirect_hops):
# login wall after hop is still a redirect-class block for our matrix
return "redirect"
if start_host and final_host and start_host != final_host:
# stay on same brand (bbc.com -> bbc.co.uk etc.) is not always a block
if not (start_host in final_host or final_host in start_host):
return "redirect"
if http_status in (301, 302, 303, 307, 308) and redirect_hops >= 1:
return "redirect"
if error and any(x in error.lower() for x in ("timed out", "timeout", "timedout")):
return "empty"
if not http_status or http_status == 0:
return "empty"
if http_status >= 400:
# unclassified error page with no vendor fingerprint
if len((text or "").strip()) < 80:
return "empty"
return "waf" if http_status in (401, 403, 406, 429, 451, 503) else "empty"
text_len = len((text or "").strip())
if text_len < 80:
return "empty"
return "none"
def title_ok(title: str, needles: list[str], intercept: str) -> bool:
if intercept in {"cloudflare", "waf", "captcha"}:
return False
t = (title or "").lower()
if not t.strip():
return False
if BLOCK_TITLE.search(title or ""):
return False
for n in needles or []:
if n.lower() in t:
return True
# X/Twitter homepage often titles exactly "X"
if t.strip() in {"x", "x.com"}:
return True
return False
def extractable(intercept: str, text_len: int, title_is_ok: bool) -> bool:
if intercept != "none":
return False
if text_len >= 200:
return True
if text_len >= 80 and title_is_ok:
return True
return False