#!/usr/bin/env python3 """ 批量合规审查脚本 - 数据库版 遍历指定日期所有发布版本,执行合规检查,生成汇总报告 """ import json, re, datetime from pathlib import Path import sys PROJECT_ROOT = Path(__file__).parent.parent sys.path.insert(0, str(PROJECT_ROOT)) from scripts.compliance_checker import check_article # 尝试导入数据库 try: from db_helper import export_topics_to_json HAVE_DB = True except ImportError: HAVE_DB = False # 配置 RELEASE_DIR = PROJECT_ROOT / "automation" / "data" / "releases" TODAY = datetime.date.today().isoformat() # 默认今天,可修改 def load_topics_from_db(): if not HAVE_DB: raise RuntimeError("Database not available") topics = export_topics_to_json() return {t['id']: t for t in topics} def load_topics_from_json(): json_path = PROJECT_ROOT / "automation" / "data" / "sustainability_topics.json" with open(json_path, 'r', encoding='utf-8') as f: topics = json.load(f) return {t['id']: t for t in topics} def extract_topic_id(filename: Path) -> str: stem = filename.stem parts = stem.split('_') if len(parts) >= 2: return parts[1] return None def main(target_date: str = None): if target_date is None: target_date = TODAY print(f"批量合规审查: {target_date}") # 加载选题数据(优先DB,失败则备援JSON) try: topics_by_id = load_topics_from_db() print("[数据源] 数据库") except Exception as e: print(f"[数据源] 数据库失败: {e}, 改用 JSON") topics_by_id = load_topics_from_json() release_path = RELEASE_DIR / target_date if not release_path.exists(): print(f"错误:发布日期目录不存在 {release_path}") return html_files = list(release_path.rglob("*.html")) print(f"找到 {len(html_files)} 个HTML文件,开始合规审查...\n") results = [] for html_file in html_files: platform = html_file.parent.name topic_id = extract_topic_id(html_file) topic_data = topics_by_id.get(topic_id) if topic_id else None with open(html_file, 'r', encoding='utf-8') as f: html_content = f.read() result = check_article(html_content, platform, topic_data) result['file'] = str(html_file.relative_to(PROJECT_ROOT)) result['platform'] = platform result['topic_id'] = topic_id result['topic_title'] = topic_data.get('title') if topic_data else "未知" results.append(result) # 输出摘要 passed = sum(1 for r in results if r['passed']) failed = len(results) - passed print(f"\n✅ 通过: {passed}, ⚠️ 需人工: {failed}") for r in results: status = "✅" if r['passed'] else "⚠️" print(f" {status} {r['topic_id']} {r['topic_title'][:40]}...") if __name__ == "__main__": import argparse parser = argparse.ArgumentParser() parser.add_argument('--date', help='审查的日期目录,默认今天') args = parser.parse_args() main(args.date)