#!/usr/bin/env python3
"""Assert one trafilatura T2 JSON body. stdout: JSON verdict."""
from __future__ import annotations
import json
import re
import sys
NAV = re.compile(r"网站地图|设为首页|加入收藏|政务微信|客户端下载|无障碍浏览", re.I)
CRUMB = re.compile(r"首页\s*[>|/]\s*政策")
HTML_DIV = re.compile(r"
dict:
out = {
"assert_ok": False,
"fail_class": "",
"note": "",
"title": "",
"char_count": 0,
"truncated": None,
"ok_field": None,
"has_shisiwu": False,
"has_2035": False,
"has_cjk4": False,
"nav_hit": False,
"crumb_count": 0,
"html_div": 0,
"html_script": 0,
}
try:
data = json.loads(raw)
except json.JSONDecodeError:
out["fail_class"] = "html_dump"
out["note"] = "body_not_json"
return out
out["ok_field"] = data.get("ok")
out["title"] = (data.get("title") or "")[:200]
md = data.get("markdown") or ""
out["char_count"] = data.get("char_count") if data.get("char_count") is not None else len(md)
out["truncated"] = data.get("truncated")
notes = []
if data.get("fail_class"):
out["fail_class"] = data["fail_class"]
notes.append(str(data.get("error") or data["fail_class"]))
if not data.get("ok"):
out["fail_class"] = out["fail_class"] or "fetch_fail"
out["note"] = "; ".join(notes) or "ok!=true"
return out
if not md or len(md) < 800:
out["fail_class"] = "empty_extract"
notes.append(f"markdown_len={len(md)}")
if not (data.get("title") or "").strip():
notes.append("empty_title")
out["fail_class"] = out["fail_class"] or "empty_extract"
out["nav_hit"] = bool(NAV.search(md))
out["crumb_count"] = len(CRUMB.findall(md))
if out["nav_hit"] or out["crumb_count"] >= 3:
out["fail_class"] = out["fail_class"] or "nav_residue"
notes.append("nav_residue")
out["html_div"] = len(HTML_DIV.findall(md))
out["html_script"] = len(HTML_SCRIPT.findall(md))
if out["html_div"] or out["html_script"]:
out["fail_class"] = out["fail_class"] or "html_dump"
notes.append("html_dump")
out["has_shisiwu"] = "十四五" in md
out["has_2035"] = "2035" in md
out["has_cjk4"] = bool(title_cjk4) and title_cjk4 in md
use_fallback = (not expect_url) or expect_url.rstrip("/") == FALLBACK_URL.rstrip("/")
if use_fallback:
if not (out["has_shisiwu"] and out["has_2035"]):
notes.append("missing_十四五_or_2035")
if not out["fail_class"]:
out["fail_class"] = "empty_extract"
elif title_cjk4 and not out["has_cjk4"]:
notes.append(f"missing_cjk4={title_cjk4}")
if not out["fail_class"]:
out["fail_class"] = "empty_extract"
if data.get("truncated") is True:
max_chars = 20000
if int(out["char_count"] or 0) != max_chars:
notes.append(f"truncated_char_count={out['char_count']}")
if not out["fail_class"]:
out["fail_class"] = "empty_extract"
out["assert_ok"] = out["fail_class"] == ""
out["note"] = "; ".join(notes)
return out
def main() -> None:
path = sys.argv[1]
expect_url = sys.argv[2] if len(sys.argv) > 2 else ""
title_cjk4 = sys.argv[3] if len(sys.argv) > 3 else ""
raw = open(path, "r", encoding="utf-8", errors="replace").read()
print(json.dumps(eval_body(raw, expect_url, title_cjk4), ensure_ascii=False))
if __name__ == "__main__":
main()