#!/usr/bin/env python3 # -*- coding: utf-8 -*- """ 步骤 2:主报告构建脚本。读取数据库 → LLM 分析 → 渲染 HTML。 用法:: # 全流程(config.yaml 留空则 LLM 自动识别产品名和行业) ../310py/bin/python build_report.py # 指定产品名(覆盖 config.yaml),自定义输出路径 ../310py/bin/python build_report.py --product "Bikini Trimmer" --output output/bikini-trimmer.html # 跳过 LLM 调用(仅渲染模板骨架,用于验证模板和数据管道) ../310py/bin/python build_report.py --no-llm # 调试:将 LLM 分析结果保存为 JSON(方便人工检查/修正后重新渲染) ../310py/bin/python build_report.py --save-data 依赖:需先运行 run_pipeline.py 产出 SQLite 数据库。 """ from __future__ import annotations import argparse import json import logging import re import sys from pathlib import Path from typing import Any, Dict, List import yaml from data_loader import DataLoader, asin_link_html, build_amazon_url, MarketStats from llm_analyzer import ( discover_personas, discover_themes_both, analyze_kano_jtbd_keywords_parallel, analyze_matrix, analyze_all_rootcauses, get_llm_workers, configure_report_llm, ) from report_utils import ( enrich_theme_keywords, recalc_neg_priorities, compute_persona_pcts, assign_persona_clusters, validate_persona_physiological_labels, prepare_persona_catalog_for_llm, pick_persona_quotes, fix_jtbd_fields, filter_matrix_rows, enrich_matrix_scene_evidence, build_matrix_table_rows_html, enrich_rootcause_quotes, build_executive_summary, enhanced_market_judgment, fix_kano_items, normalize_kano_reverse_items, normalize_persona_dimension, normalize_rootcauses, build_asin_labels, build_asin_short_codes, build_asin_tables_html, build_footer_asin_links, get_layout_config, persona_display_meta, build_asin_theme_insights, asin_link_with_label, quote_cn_summary, sort_personas_by_evidence, sort_rootcauses_by_evidence, jtbd_cell_html, build_product_stopwords, build_kano_grid_html, build_neg_theme_summary_note, build_neg_theme_table_rows, build_pos_theme_table_rows, build_sentiment_keyword_groups, prepare_keyword_display, build_keyword_tables_html, build_quote_blocks_html, ) from echarts_builder import ( get_echarts_script, build_all_charts, calc_theme_freq, calc_per_asin_theme_freq, get_chart_layout, build_theme_mini_charts_html, ) logger = logging.getLogger("voc.build_report") logging.basicConfig(level=logging.INFO, format="%(asctime)s [%(levelname)s] %(message)s") _PLACEHOLDER_RE = re.compile(r"\{\{[A-Z0-9_]+\}\}") SCRIPT_DIR = Path(__file__).resolve().parent PROJECT_ROOT = SCRIPT_DIR.parent CONFIG_FILE = SCRIPT_DIR / "config.yaml" TEMPLATE_FILE = SCRIPT_DIR / "template.html" def product_file_slug(product: str, industry: str = "") -> str: """产品名 → 安全文件名 slug(仅产品名,不含行业/副标题)。""" s = (product or "voc-report").strip() if industry: ind = industry.strip() if ind and ind in s: for pat in ( f"({ind})", f"({ind})", f" - {ind}", f" · {ind}", f"|{ind}", f"|{ind}", f"/ {ind}", ): if pat in s: s = s.split(pat, 1)[0].strip() if s.endswith(ind): s = s[: -len(ind)].strip(" -/|·()()") # 「产品A / 产品B」类双描述只取主产品名(第一段) for sep in ("/", "|", "|"): if sep in s: s = s.split(sep, 1)[0].strip() break s = s.lower() s = re.sub(r'[/\\:*?"<>|]+', "-", s) s = re.sub(r"\s+", "-", s) s = re.sub(r"-+", "-", s).strip("-") return s or "voc-report" def load_config() -> dict: with CONFIG_FILE.open(encoding="utf-8") as f: return yaml.safe_load(f) or {} def finalize_template(html: str) -> str: """检查并清理未替换的 {{PLACEHOLDER}},避免报告页面露出模板标签。""" remaining = sorted(set(_PLACEHOLDER_RE.findall(html))) if remaining: logger.error( "报告模板存在未替换占位符 (%s 个): %s — 请更新 render_html 或重新渲染", len(remaining), ", ".join(remaining[:12]), ) html = _PLACEHOLDER_RE.sub("", html) return html def build_report_data(loader: DataLoader, cfg: dict) -> Dict[str, Any]: """构建所有报告数据。""" configure_report_llm(cfg) data: Dict[str, Any] = {} llm_workers = get_llm_workers(cfg) product = cfg.get("product_name", "主产品") industry = cfg.get("industry", "当前品类") logger.info("LLM 并发 workers=%s | 报告模式: max thinking", llm_workers) # ── 基础统计 ── stats = loader.load_basic_stats() data["stats"] = stats # ── 聚类数据 ── persona_sample_reviews = int(cfg.get("persona_sample_reviews", 5)) cluster_data = loader.build_cluster_prompt_data( persona_sample_reviews=persona_sample_reviews, ) data["cluster_data"] = cluster_data all_clusters = loader.load_cluster_data() reviews = loader.load_reviews() extractions = loader.load_comment_extractions() data["extractions"] = extractions persona_quote_max = int(cfg.get("persona_quote_max", 3)) rootcause_quote_max = int(cfg.get("rootcause_quote_max", 4)) physio_min = int(cfg.get("persona_physio_min_reviews", 5)) min_hit_count = int(cfg.get("persona_min_hit_count", 5)) data["persona_min_hit_count"] = min_hit_count prepare_persona_catalog_for_llm( cluster_data["persona_cluster_catalog"], all_clusters, reviews, min_physio_reviews=physio_min, ) # ── Persona ── personas = discover_personas(cluster_data, product_name=product, industry=industry) personas = assign_persona_clusters(personas, all_clusters) personas = validate_persona_physiological_labels( personas, cluster_data, reviews, all_clusters, min_review_count=physio_min, ) data["personas"] = personas # ── 主题(差评+好评并发) ── neg_themes, pos_themes = discover_themes_both( cluster_data, stats.neg_review_count, stats.pos_review_count, ) neg_themes = enrich_theme_keywords(neg_themes, cluster_data.get("global_negative", [])) pos_themes = enrich_theme_keywords(pos_themes, cluster_data.get("global_positive", [])) data["neg_themes"] = neg_themes data["pos_themes"] = pos_themes # ── 主题频次统计(结构化 category+aspect 优先) ── neg_freq = calc_theme_freq(neg_themes, reviews, is_neg=True, extractions=extractions) pos_freq = calc_theme_freq(pos_themes, reviews, is_neg=False, extractions=extractions) per_asin_neg = calc_per_asin_theme_freq( neg_themes, reviews, is_neg=True, extractions=extractions, ) per_asin_pos = calc_per_asin_theme_freq( pos_themes, reviews, is_neg=False, extractions=extractions, ) neg_themes = recalc_neg_priorities( neg_themes, neg_freq, stats.neg_review_count, len(stats.asins), per_asin_neg, ) data["neg_themes"] = neg_themes data["neg_freq"] = neg_freq data["pos_freq"] = pos_freq data["per_asin_neg"] = per_asin_neg data["per_asin_pos"] = per_asin_pos # ── Persona 统计与引用 ── personas = compute_persona_pcts(personas, reviews, all_clusters) personas = sort_personas_by_evidence(personas) data["personas"] = personas data["persona_quotes"] = pick_persona_quotes( personas, reviews, extractions, max_quotes=persona_quote_max, ) # ── KANO + JTBD + 情感关键词(三者并发) ── neg_kw_groups = build_sentiment_keyword_groups(neg_themes, reviews, limit=6, is_neg=True) pos_kw_groups = build_sentiment_keyword_groups(pos_themes, reviews, limit=6, is_neg=False) kano_raw, jtbd_raw, keywords_raw = analyze_kano_jtbd_keywords_parallel( neg_themes, pos_themes, personas, neg_kw_groups, pos_kw_groups, product_name=product, ) kano = normalize_kano_reverse_items(fix_kano_items(kano_raw)) data["kano"] = kano jtbd = fix_jtbd_fields(jtbd_raw, personas) data["jtbd"] = jtbd data["keywords_llm"] = keywords_raw _, _, keywords_neg, keywords_pos = prepare_keyword_display( neg_themes, pos_themes, reviews, personas, keywords_raw, limit=6, ) data["keywords_neg"] = keywords_neg data["keywords_pos"] = keywords_pos data["keywords"] = keywords_neg + keywords_pos # ── 矩阵(依赖 KANO) ── matrix = enrich_matrix_scene_evidence( analyze_matrix( personas, kano, cluster_data, stats.total_reviews, market_avg=stats.weighted_avg_rating, ), personas, reviews, ) matrix = filter_matrix_rows( matrix, market_avg=stats.weighted_avg_rating, ) data["matrix"] = matrix # ── 根因分析(各 Persona 并发) ── rootcauses = analyze_all_rootcauses( personas, neg_themes, all_clusters=all_clusters, max_workers=llm_workers, product_name=product, industry=industry, min_hit_count=min_hit_count, ) rootcauses = enrich_rootcause_quotes( rootcauses, personas, loader, reviews, persona_quotes=data["persona_quotes"], extractions=extractions, max_quotes=rootcause_quote_max, ) rootcauses = normalize_rootcauses(rootcauses, neg_themes) rootcauses = sort_rootcauses_by_evidence(rootcauses, personas) data["rootcauses"] = rootcauses # ── 市场竞争 + 决策摘要 + ASIN 标签 ── asin_labels = build_asin_labels(stats) data["asin_labels"] = asin_labels data["asin_short_codes"] = build_asin_short_codes(stats) market_title, market_desc = enhanced_market_judgment(stats) data["market_title"] = market_title data["market_desc"] = market_desc data["executive_summary"] = build_executive_summary( stats, neg_freq, pos_freq, neg_themes, asin_labels=asin_labels, pos_themes=pos_themes, ) data["asin_theme_insights"] = build_asin_theme_insights( stats, neg_themes, per_asin_neg, asin_labels, ) return data def render_html(template_path: Path, data: Dict[str, Any], cfg: dict) -> str: """用数据填充模板并返回完整 HTML。""" if not template_path.is_file(): raise FileNotFoundError(f"模板文件不存在: {template_path}") html = template_path.read_text(encoding="utf-8") # ── 简单占位符 ── product = cfg.get("product_name", "Product") analysis_date = cfg.get("analysis_date", "2026-06-12") data_source = cfg.get("data_source", "卖家精灵 realtime CSV") version = cfg.get("report_version", "v1") html = html.replace("{{PRODUCT_NAME}}", product) html = html.replace("{{ANALYSIS_DATE}}", analysis_date) html = html.replace("{{DATA_SOURCE}}", data_source) html = html.replace("{{VERSION}}", version) # ── ECharts ── echarts_inline = cfg.get("echarts_inline", True) html = html.replace("{{ECHARTS_SCRIPT}}", get_echarts_script(inline=echarts_inline)) # ── 导航 ASIN 列表 ── stats: MarketStats = data["stats"] asins = [a.asin for a in stats.asins] html = html.replace("{{ASIN_COUNT}}", str(len(asins))) # ── KPI ── html = html.replace("{{TOTAL_REVIEWS}}", f"{stats.total_reviews:,}") html = html.replace("{{WEIGHTED_AVG}}", f"{stats.weighted_avg_rating}") html = html.replace("{{POS_RATE}}", f"{int(stats.pos_rate * 100)}%") html = html.replace("{{NEG_RATE}}", f"{int(stats.neg_rate * 100)}%") neutral_count = stats.total_reviews - stats.neg_review_count - stats.pos_review_count neutral_pct = round(neutral_count / max(stats.total_reviews, 1) * 100) html = html.replace("{{NEUTRAL_RATE}}", f"{neutral_pct}%") html = html.replace("{{NEUTRAL_COUNT}}", str(neutral_count)) # KPI 颜色 avg_color = "danger" if stats.weighted_avg_rating < 3.5 else ("warn" if stats.weighted_avg_rating <= 4.0 else "success") html = html.replace("{{AVG_COLOR}}", avg_color) # ── 市场竞争 callout ── callout_type = "danger" if stats.weighted_avg_rating < 3.5 else ("warn" if stats.weighted_avg_rating <= 4.0 else "success") html = html.replace("{{MARKET_CALLOUT_TYPE}}", callout_type) html = html.replace("{{MARKET_TITLE}}", data["market_title"]) html = html.replace("{{MARKET_DESC}}", data["market_desc"]) # ── 决策摘要 ── es = data.get("executive_summary") or {} bullets = es.get("bullets", []) opps = es.get("opportunities", []) conclusion = es.get("conclusion", "") insights = es.get("insights", []) summary_html = "" if conclusion: summary_html += ( f'
' f'
产品定义结论
' f'

{conclusion}

' ) summary_html += '" if opps: summary_html += '
产品机会清单(按差评频次排序)
    ' for o in opps: summary_html += f"
  1. {o}
  2. " summary_html += "
" html = html.replace("{{EXEC_SUMMARY_HTML}}", summary_html) if insights: ins_html = '
关键不对称洞察
" else: ins_html = "" html = html.replace("{{INSIGHTS_CALLOUT}}", ins_html) asin_theme_ins = data.get("asin_theme_insights") or [] if asin_theme_ins: ath = '
ASIN 主题对比结论
" else: ath = "" html = html.replace("{{ASIN_THEME_INSIGHTS_HTML}}", ath) asin_labels = data.get("asin_labels") or build_asin_labels(stats) asin_short = data.get("asin_short_codes") or build_asin_short_codes(stats) layout_cfg = get_layout_config(cfg) chart_layout = get_chart_layout(stats, cfg) # ── ASIN 表格(大品类摘要 + 附录)── summary_table, appendix_table, table_note = build_asin_tables_html( stats, asin_labels, asin_link_html, top_n=layout_cfg["asin_table_top_n"], large_threshold=layout_cfg["large_asin_threshold"], ) html = html.replace("{{ASIN_TABLE_NOTE}}", table_note) html = html.replace("{{ASIN_TABLE_SUMMARY}}", summary_table) html = html.replace("{{ASIN_TABLE_APPENDIX}}", appendix_table) # ── 评分分布图布局 ── if chart_layout["large_market"]: star_note = ( f"共 {chart_layout['asin_count']} 个竞品:横向堆叠图按评论量排序," f"可在下方滚动查看全部;轴标签为短码(A/B/…)。" ) star_scroll_max = min(720, chart_layout["star_chart_height"]) else: star_note = "按竞品展示 5★–1★ 评论堆叠分布。" star_scroll_max = chart_layout["star_chart_height"] html = html.replace("{{STAR_CHART_NOTE}}", star_note) html = html.replace("{{STAR_CHART_HEIGHT}}", str(chart_layout["star_chart_height"])) html = html.replace("{{STAR_SCROLL_MAX}}", str(star_scroll_max)) if chart_layout.get("show_star_summary"): html = html.replace( "{{STAR_SUMMARY_HTML}}", '

评分分布摘要(Top 15 + 其余聚合)

' f'
', ) else: html = html.replace("{{STAR_SUMMARY_HTML}}", "") neg_top6 = sorted(data["neg_freq"].items(), key=lambda x: x[1], reverse=True)[:6] pos_top6 = sorted(data["pos_freq"].items(), key=lambda x: x[1], reverse=True)[:6] neg_theme_names = [n for n, _ in neg_top6] pos_theme_names = [n for n, _ in pos_top6] hm_th = layout_cfg["heatmap_asin_threshold"] if chart_layout["use_heatmap"]: neg_theme_note = f"热力图:行=主题、列=竞品短码;颜色越深命中越多。下方为各主题 Top ASIN 明细。竞品 >{hm_th},请用滑块横向浏览。" pos_theme_note = neg_theme_note.replace("差评", "好评") elif chart_layout["large_market"]: neg_theme_note = "分组柱:X 轴=竞品短码,图例=6 个差评主题;竞品较多时请拖动下方滑块。" pos_theme_note = "分组柱:X 轴=竞品短码,图例=6 个好评主题;竞品较多时请拖动下方滑块。" else: neg_theme_note = "分组柱:X 轴=竞品,图例=Top 6 差评主题。" pos_theme_note = "分组柱:X 轴=竞品,图例=Top 6 好评主题。" html = html.replace("{{NEG_THEME_CHART_NOTE}}", neg_theme_note) html = html.replace("{{POS_THEME_CHART_NOTE}}", pos_theme_note) html = html.replace("{{NEG_THEME_CHART_HEIGHT}}", str(chart_layout["neg_theme_height"])) html = html.replace("{{POS_THEME_CHART_HEIGHT}}", str(chart_layout["pos_theme_height"])) html = html.replace( "{{NEG_THEME_MINI_HTML}}", build_theme_mini_charts_html("neg", neg_theme_names, chart_layout["asin_count"], large_threshold=hm_th), ) html = html.replace( "{{POS_THEME_MINI_HTML}}", build_theme_mini_charts_html("pos", pos_theme_names, chart_layout["asin_count"], large_threshold=hm_th), ) # 移除旧占位符兼容 html = html.replace("{{ASIN_TABLE_ROWS}}", "") # ── Persona 卡片 ── persona_map = {p.get("name"): p for p in data.get("personas", [])} persona_cards = [] for i, p in enumerate(data["personas"]): pq = next((q for q in data["persona_quotes"] if q["persona"] == p["name"]), None) quote_items = (pq.get("quotes") or []) if pq else [] if not quote_items and pq and pq.get("quote"): quote_items = [pq] quote_html = build_quote_blocks_html( quote_items, asin_labels, empty_msg="暂无命中池内的代表性评论", ) normalize_persona_dimension(p) min_hit = int( data.get("persona_min_hit_count") or cfg.get("persona_min_hit_count", 5) ) meta = persona_display_meta(p, stats.total_reviews, min_hit_count=min_hit) persona_cards.append(f"""
{p.get("name", "?")}
{meta}
核心痛点:{p.get("core_pain", "")}
核心需求:{p.get("core_need", "")}
购买动机:{p.get("purchase_motivation", "")}
{quote_html}
""") html = html.replace("{{PERSONA_CARDS}}", "\n".join(persona_cards)) # ── 差评主题表格 ── neg_total = stats.neg_review_count neg_rows_html = build_neg_theme_table_rows( data["neg_freq"], data["neg_themes"], neg_total, len(stats.asins), data["per_asin_neg"], asin_labels, ) html = html.replace("{{NEG_THEME_TABLE_ROWS}}", neg_rows_html) neg_summary_note = build_neg_theme_summary_note( stats, data["neg_freq"], data["neg_themes"], data.get("per_asin_neg"), ) html = html.replace("{{NEG_THEME_SUMMARY_NOTE}}", neg_summary_note) # ── 好评主题表格 ── pos_total = stats.pos_review_count pos_rows_html = build_pos_theme_table_rows( data["pos_freq"], data.get("pos_themes") or [], pos_total, data["neg_themes"], data["neg_freq"], neg_total, data.get("kano") or [], ) html = html.replace("{{POS_THEME_TABLE_ROWS}}", pos_rows_html) html = html.replace("{{POS_REVIEW_COUNT}}", str(pos_total)) pos_pct_sum = sum( round(c / max(pos_total, 1) * 100) for c in data["pos_freq"].values() if c > 0 ) html = html.replace("{{POS_PCT_SUM}}", str(pos_pct_sum)) # ── KANO 四象限卡片 ── kano_display = normalize_kano_reverse_items(data.get("kano") or []) html = html.replace("{{KANO_GRID_HTML}}", build_kano_grid_html(kano_display)) html = html.replace("{{KANO_TABLE_ROWS}}", "") # ── JTBD 表格 ── jtbd_rows = [] jtbd_items = fix_jtbd_fields(data.get("jtbd") or [], data.get("personas") or []) for item in jtbd_items: jtbd_rows.append( f'{item.get("persona", "?")}' f'{jtbd_cell_html(item.get("core_job", ""))}' f'{jtbd_cell_html(item.get("functional_motivation", ""))}' f'{jtbd_cell_html(item.get("emotional_motivation", ""))}' f'{jtbd_cell_html(item.get("social_motivation", ""))}' f'{jtbd_cell_html(item.get("trigger", ""))}' ) html = html.replace("{{JTBD_TABLE_ROWS}}", "\n".join(jtbd_rows)) # ── 矩阵(Persona 分组卡片)── matrix_html = build_matrix_table_rows_html(data["matrix"], persona_map) html = html.replace("{{MATRIX_TABLE_ROWS}}", matrix_html) html = html.replace("{{MATRIX_HTML}}", "") # ── 根因卡片 ── rc_cards = [] rc_idx = 0 for i, rc in enumerate(data["rootcauses"]): pname = rc.get("persona_name", f"Persona {i}") if rc.get("skipped"): rc_cards.append(f"""""") continue causes = rc.get("root_causes", []) affected = ", ".join(rc.get("affected_themes", [])) rc_html_parts = [] for cause in causes: quotes_html = "" for q in cause.get("quotes", []): asin = q.get("asin", "?") if asin == "?" or not re.match(r"^B[A-Z0-9]{9}$", asin): continue valid_asins = {a.asin for a in stats.asins} if asin not in valid_asins: continue amazon_url = build_amazon_url(asin) cn = q.get("cn_summary") or quote_cn_summary(q.get("text", "")) alabel = asin_labels.get(asin, asin) quotes_html += ( f'
"{q.get("text", "")}" ' f'— {alabel}' f'
摘要:{cn}
\n' ) rc_html_parts.append(f"""
{cause.get("title", "根因")}

{cause.get("mechanism", "")}

{quotes_html}
→ 产品开发方向

{cause.get("dev_direction", "")}

""") collapsed = " collapsed" if rc_idx >= 2 else "" rc_idx += 1 rc_cards.append(f"""
👤 {pname} — 核心痛点根因 {affected}
差评主题归因(该群体命中)

{affected}

{"".join(rc_html_parts)}
""") html = html.replace("{{ROOTCAUSE_CARDS}}", "\n".join(rc_cards)) # ── 情感关键词双表 ── neg_kw = data.get("keywords_neg") pos_kw = data.get("keywords_pos") if neg_kw is None or pos_kw is None: neg_kw, pos_kw = [], [] neg_kw_rows, pos_kw_rows = build_keyword_tables_html(neg_kw or [], pos_kw or []) html = html.replace("{{KEYWORD_NEG_TABLE_ROWS}}", neg_kw_rows) html = html.replace("{{KEYWORD_POS_TABLE_ROWS}}", pos_kw_rows) # ── 图表 JS ── charts_js = build_all_charts( stats, data["neg_themes"], data["pos_themes"], data["neg_freq"], data["pos_freq"], data["per_asin_neg"], data["per_asin_pos"], asin_labels=asin_labels, asin_short_codes=asin_short, layout=chart_layout, cfg=cfg, ) html = html.replace("{{CHART_JS}}", charts_js) html = html.replace( "{{FOOTER_ASIN_LINKS}}", build_footer_asin_links(stats, asin_labels), ) return finalize_template(html) def main(): parser = argparse.ArgumentParser(description="构建 VOC 分析报告 HTML") parser.add_argument("--product", help="产品名(覆盖 config.yaml)") parser.add_argument("--industry", help="行业名") parser.add_argument("--output", default=None, help="输出 HTML 路径") parser.add_argument("--no-llm", action="store_true", help="跳过 LLM 调用(仅渲染模板,用于测试)") parser.add_argument("--save-data", action="store_true", help="将 LLM 分析结果保存为 JSON") parser.add_argument("--render-from-json", help="从已保存的 JSON 渲染 HTML(跳过 LLM)") args = parser.parse_args() cfg = load_config() if args.product: cfg["product_name"] = args.product if args.industry: cfg["industry"] = args.industry product = cfg.get("product_name", "").strip() industry = cfg.get("industry", "").strip() # 如果产品名或行业为空,用 LLM 从原始评论中自动识别 need_detect = (not product or product == "亚马逊商品" or not industry or industry == "亚马逊电商") if need_detect: input_dir_raw = cfg.get("input_dir", "") input_dir_path_raw = (SCRIPT_DIR / input_dir_raw).resolve() if input_dir_raw else None if input_dir_path_raw and input_dir_path_raw.is_dir(): samples, dir_name = DataLoader.load_raw_review_samples(input_dir_path_raw, max_samples=50) if samples: from llm_analyzer import detect_product_and_industry detected_product, detected_industry = detect_product_and_industry(samples, dir_name) if not product or product == "亚马逊商品": product = detected_product cfg["product_name"] = product logger.info("LLM 自动识别产品名: %s", product) if not industry or industry == "亚马逊电商": industry = detected_industry cfg["industry"] = industry logger.info("LLM 自动识别行业: %s", industry) else: product = product or "亚马逊商品" industry = industry or "亚马逊电商" else: product = product or "亚马逊商品" industry = industry or "亚马逊电商" output_dir = SCRIPT_DIR / cfg.get("output_dir", "./output") output_dir.mkdir(parents=True, exist_ok=True) slug = product_file_slug(product, industry) output_html = Path(args.output) if args.output else (output_dir / f"{slug}.html") logger.info("=== VOC 报告构建 ===") logger.info("产品: %s | 行业: %s", product, industry) logger.info("项目根目录: %s", PROJECT_ROOT) # 加载数据 loader = DataLoader(PROJECT_ROOT, product, industry) if args.render_from_json: json_path = Path(args.render_from_json) if not json_path.is_file(): slug = product_file_slug(product, industry) json_path = output_dir / f"{slug}-analysis-data.json" logger.info("从 JSON 渲染: %s", json_path) with json_path.open(encoding="utf-8") as f: raw = json.load(f) stats = loader.load_basic_stats() reviews = loader.load_reviews() extractions = loader.load_comment_extractions() persona_quote_max = int(cfg.get("persona_quote_max", 3)) rootcause_quote_max = int(cfg.get("rootcause_quote_max", 4)) all_clusters = loader.load_cluster_data() personas = assign_persona_clusters(raw.get("personas", []), all_clusters) cluster_data_rr = raw.get("cluster_data") or loader.build_cluster_prompt_data( persona_sample_reviews=int(cfg.get("persona_sample_reviews", 5)), ) physio_min = int(cfg.get("persona_physio_min_reviews", 5)) prepare_persona_catalog_for_llm( cluster_data_rr.get("persona_cluster_catalog") or [], all_clusters, reviews, min_physio_reviews=physio_min, ) personas = validate_persona_physiological_labels( personas, cluster_data_rr, reviews, all_clusters, min_review_count=physio_min, ) personas = compute_persona_pcts(personas, reviews, all_clusters) personas = sort_personas_by_evidence(personas) persona_quotes = pick_persona_quotes( personas, reviews, extractions, max_quotes=persona_quote_max, ) neg_freq = calc_theme_freq( raw.get("neg_themes", []), reviews, is_neg=True, extractions=extractions, ) pos_freq = calc_theme_freq( raw.get("pos_themes", []), reviews, is_neg=False, extractions=extractions, ) per_asin_neg = calc_per_asin_theme_freq( raw.get("neg_themes", []), reviews, is_neg=True, extractions=extractions, ) per_asin_pos = calc_per_asin_theme_freq( raw.get("pos_themes", []), reviews, is_neg=False, extractions=extractions, ) neg_themes = recalc_neg_priorities( raw.get("neg_themes", []), neg_freq, stats.neg_review_count, len(stats.asins), per_asin_neg, ) asin_labels = build_asin_labels(stats) data = { **raw, "stats": stats, "personas": personas, "persona_quotes": persona_quotes, "neg_themes": neg_themes, "neg_freq": neg_freq, "pos_freq": pos_freq, "per_asin_neg": per_asin_neg, "per_asin_pos": per_asin_pos, "asin_labels": asin_labels, "asin_short_codes": build_asin_short_codes(stats), "executive_summary": build_executive_summary( stats, neg_freq, pos_freq, neg_themes, asin_labels=asin_labels, pos_themes=raw.get("pos_themes", []), ), "asin_theme_insights": build_asin_theme_insights( stats, neg_themes, per_asin_neg, asin_labels, ), } if not data.get("keywords_neg") and data.get("neg_themes"): _, _, kw_neg, kw_pos = prepare_keyword_display( data["neg_themes"], data.get("pos_themes") or [], reviews, personas, data.get("keywords_llm") or {}, limit=6, ) data["keywords_neg"] = kw_neg data["keywords_pos"] = kw_pos data["keywords"] = kw_neg + kw_pos market_title, market_desc = enhanced_market_judgment(stats) data["market_title"] = market_title data["market_desc"] = market_desc data["rootcauses"] = enrich_rootcause_quotes( data.get("rootcauses") or [], personas, loader, reviews, persona_quotes=persona_quotes, extractions=extractions, max_quotes=rootcause_quote_max, ) elif args.no_llm: logger.warning("--no-llm 模式:跳过 LLM 调用,仅渲染模板") # 使用空数据渲染(测试模板) data = { "stats": loader.load_basic_stats(), "personas": [], "neg_themes": [], "pos_themes": [], "neg_freq": {}, "pos_freq": {}, "per_asin_neg": {}, "per_asin_pos": {}, "kano": [], "jtbd": [], "matrix": [], "rootcauses": [], "keywords": [], "keywords_neg": [], "keywords_pos": [], "keywords_llm": {}, "persona_quotes": [], "asin_labels": build_asin_labels(loader.load_basic_stats()), "asin_short_codes": build_asin_short_codes(loader.load_basic_stats()), "asin_theme_insights": [], "market_title": "数据待生成", "market_desc": "请运行完整流程", "executive_summary": { "bullets": ["请运行完整流程生成分析"], "opportunities": [], "conclusion": "", "insights": [], }, } else: data = build_report_data(loader, cfg) # 保存中间数据(调试用) if args.save_data: data_json = output_dir / f"{slug}-analysis-data.json" with data_json.open("w", encoding="utf-8") as f: json.dump(data, f, ensure_ascii=False, indent=2, default=str) logger.info("分析数据已保存: %s", data_json) # 渲染 HTML logger.info("渲染 HTML...") html_content = render_html(TEMPLATE_FILE, data, cfg) output_html.parent.mkdir(parents=True, exist_ok=True) output_html.write_text(html_content, encoding="utf-8") logger.info("报告已生成: %s (%s KB)", output_html, len(html_content) // 1024) # 如果 ECharts 内联模式且缓存存在,报告大小 if cfg.get("echarts_inline", True): from echarts_builder import CACHE_FILE if CACHE_FILE.is_file(): logger.info("ECharts 已内联(缓存: %s KB)", CACHE_FILE.stat().st_size // 1024) print(f"\n✅ 报告已生成: {output_html}") if __name__ == "__main__": main()