#!/usr/bin/env python3 """ process_data.py — 对 Holy Crab 任务结果进行数据分析 用法: python3 process_data.py summarize <任务ID> 榜单摘要 python3 process_data.py sentiment <任务ID> 评论情感统计 python3 process_data.py compare <任务ID> 产品横向对比 python3 process_data.py detail <任务ID> 完整数据(含召回内容) Examples: python3 process_data.py summarize HC-20260727-ABC12 python3 process_data.py sentiment HC-20260727-ABC12 """ import json import os import sys from pathlib import Path SCRIPT_DIR = Path(__file__).resolve().parent DEFAULT_BACKEND = "http://localhost:8000" def load_cookie() -> str | None: env = os.environ.get("HC_SESSION") if env: return env cookie_file = Path.home() / ".holy_crab_cookie" if cookie_file.exists(): return cookie_file.read_text().strip() return None def fetch_task(task_id: str) -> dict: backend = os.environ.get("HC_BACKEND_URL", DEFAULT_BACKEND).rstrip("/") url = f"{backend}/api/v1/tasks/{task_id}" cookie = load_cookie() headers = {} if cookie: headers["Cookie"] = f"hc_session={cookie}" import urllib.request req = urllib.request.Request(url, headers=headers) try: with urllib.request.urlopen(req, timeout=15) as resp: data = json.loads(resp.read().decode("utf-8")) except urllib.error.HTTPError as exc: body = exc.read().decode("utf-8", errors="replace") try: err = json.loads(body) print(f"❌ API错误 {exc.code}: {err}", file=sys.stderr) except Exception: print(f"❌ HTTP {exc.code}: {body[:300]}", file=sys.stderr) sys.exit(1) result = data.get("data", {}).get("result", {}) extracted = result.get("extracted") if not extracted: print(f"❌ 任务 {task_id} 暂无 extracted 数据(任务未完成或格式不兼容)", file=sys.stderr) sys.exit(1) return extracted # --------------------------------------------------------------------------- # 命令处理器 # --------------------------------------------------------------------------- def cmd_summarize(extracted: dict) -> None: """榜单摘要""" ranking = extracted.get("榜单", []) products = extracted.get("产品详情", []) src = extracted.get("参考来源笔记总量", {}) meta = extracted.get("提取元数据", {}) stats = meta.get("召回统计", {}) warns = extracted.get("提取警告", []) print(f"\n{'='*60}") print(f" 📊 榜单摘要") print(f"{'='*60}") print(f" 参考来源笔记总量 : {src.get('原始文本', 'N/A')} (提取: {src.get('提取数量', 'N/A')})") print(f" 包含召回明细 : {'是' if meta.get('包含召回明细') else '否'}") print(f" 标签总数 : {stats.get('标签总数', 0)}") print(f" 实际召回总数 : {stats.get('实际召回内容总数', 0)}") print(f"\n 【榜单排名】") print(f" {'排名':<4} {'产品名称':<20} {'推荐比例':<8} {'产品详情'}") print(f" {'─'*60}") for item in ranking: rank = item.get("当前排名", "") name = item.get("产品名称", "")[:20] ratio = item.get("推荐比例", "") # 找对应的产品详情 prod = next((p for p in products if p.get("产品名称") == name), None) if prod: top3 = [t["内容标签"] for t in prod.get("标签数量Top3", [])] detail = f"Top标签: {', '.join(top3)}" else: detail = "" print(f" {str(rank):<4} {name:<20} {ratio:<8} {detail}") if products: print(f"\n 【各产品经验标签统计】") for prod in products: tags = prod.get("内容标签", []) top3 = prod.get("标签数量Top3", []) print(f"\n ▶ {prod.get('产品名称')} (排行{prod.get('当前排行')})") print(f" 参考经验: {prod.get('参考经验人数')} | 推荐比例: {prod.get('推荐比例')}") print(f" 标签 Top3: ", end="") print(", ".join(f"{t['内容标签']}({t['经验数量']})" for t in top3) if top3 else "无") recall_total = sum(t.get("实际召回数量", 0) for t in tags) print(f" 实际召回总数: {recall_total}") if warns: print(f"\n ⚠️ 提取警告 ({len(warns)} 条):") for w in warns[:5]: print(f" - {w}") if len(warns) > 5: print(f" ... 共 {len(warns)} 条") print(f"\n{'='*60}\n") def cmd_sentiment(extracted: dict) -> None: """评论情感统计(从 comment 类型召回内容中提取)""" all_comments = [] for prod in extracted.get("产品详情", []): for tag in prod.get("内容标签", []): for note in tag.get("召回内容", []): if note.get("召回类型") == "comment": all_comments.append({ "评论内容": note.get("评论内容"), "所属笔记点赞": note.get("所属笔记点赞数量"), "发布时间": note.get("发布时间"), "产品": prod.get("产品名称"), "标签": tag.get("内容标签"), }) if not all_comments: print(f"\n ⚠️ 当前任务无评论召回数据(comment 类型召回内容为空)") print(f" 轻度/常规任务默认不包含评论召回,深度任务(deep模式)才有。\n") return print(f"\n{'='*60}") print(f" 💬 评论召回统计 (共 {len(all_comments)} 条)") print(f"{'='*60}") # 按产品分组 by_product: dict[str, list] = {} for c in all_comments: p = c["产品"] by_product.setdefault(p, []).append(c) for product, comments in by_product.items(): print(f"\n 【{product}】({len(comments)} 条评论)") for c in comments[:10]: content = c["评论内容"] or "" print(f" • {content[:60]}{'...' if len(content)>60 else ''}") print(f" 所属笔记 👍{c['所属笔记点赞']} | {c['发布时间']}") if len(comments) > 10: print(f" ... 还有 {len(comments)-10} 条") print(f"\n 📌 AI 情感分析建议:") print(f" 请 AI 读取上述评论内容,判断每条评论的情感(正面/负面/中性),") print(f" 并统计各情感类别的数量和占比。") print(f"{'='*60}\n") def cmd_compare(extracted: dict) -> None: """产品横向对比""" products = extracted.get("产品详情", []) if not products: print(f"❌ 无产品详情数据", file=sys.stderr) return print(f"\n{'='*60}") print(f" 🔍 产品横向对比") print(f"{'='*60}") headers = ["指标"] + [p.get("产品名称", f"产品{i+1}")[:12] for i, p in enumerate(products)] col_w = 14 print(f" {'指标':<14} " + " ".join(f"{h:<{col_w}}" for h in headers[1:])) print(f" {'─'*60}") def col(val: str) -> str: return f"{str(val):<{col_w}}" print(f" {'当前排行':<14} " + " ".join(col(p.get("当前排行", "")) for p in products)) print(f" {'参考经验人数':<14} " + " ".join(col(p.get("参考经验人数", "")) for p in products)) print(f" {'推荐比例':<14} " + " ".join(col(p.get("推荐比例", "")) for p in products)) tag_counts = [len(p.get("内容标签", [])) for p in products] print(f" {'标签数量':<14} " + " ".join(col(c) for c in tag_counts)) recall_totals = [] for p in products: total = sum(t.get("实际召回数量", 0) for t in p.get("内容标签", [])) recall_totals.append(total) print(f" {'实际召回总数':<14} " + " ".join(col(r) for r in recall_totals)) top_tags = [] for p in products: top3 = p.get("标签数量Top3", []) tags_str = ", ".join(t["内容标签"] for t in top3[:2]) if top3 else "—" top_tags.append(tags_str) print(f" {'Top1-2标签':<14} " + " ".join(f"{t[:col_w*2-2]:<{col_w*2}}" for t in top_tags)) print(f"\n 【各产品 Top3 标签详情】") for i, prod in enumerate(products): print(f"\n {i+1}. {prod.get('产品名称')}") for t in prod.get("标签数量Top3", []): recall = next( (x.get("实际召回数量", 0) for x in prod.get("内容标签", []) if x.get("内容标签") == t["内容标签"]), 0 ) print(f" {t['排名']}. {t['内容标签']} {t['经验数量']}人 召回:{recall}") print(f"\n{'='*60}\n") def cmd_detail(extracted: dict) -> None: """完整数据(含所有召回内容)""" print(json.dumps(extracted, ensure_ascii=False, indent=2)) # --------------------------------------------------------------------------- # 主入口 # --------------------------------------------------------------------------- COMMANDS = { "summarize": ("榜单摘要", cmd_summarize), "sentiment": ("评论情感统计", cmd_sentiment), "compare": ("产品横向对比", cmd_compare), "detail": ("完整数据", cmd_detail), } def main() -> None: if len(sys.argv) < 3 or sys.argv[1] not in COMMANDS: print(f"用法: {sys.argv[0]} <{'|'.join(COMMANDS)}> <任务ID>") print(f"示例: {sys.argv[0]} summarize HC-20260727-ABC12") sys.exit(1) cmd_key = sys.argv[1] task_id = sys.argv[2] _, handler = COMMANDS[cmd_key] extracted = fetch_task(task_id) handler(extracted) if __name__ == "__main__": main()