Files
yu-zhi-ran/scripts/batch_compliance_check.py
T

109 lines
3.6 KiB
Python

#!/usr/bin/env python3
"""
批量合规审查脚本
遍历指定日期所有发布版本,执行合规检查,生成汇总报告
"""
import json
import re
from pathlib import Path
from datetime import datetime
import sys
PROJECT_ROOT = Path(__file__).parent.parent
sys.path.insert(0, str(PROJECT_ROOT))
from scripts.compliance_checker import check_article
# 配置
RELEASE_DIR = PROJECT_ROOT / "automation" / "data" / "releases"
TOPICS_FILE = PROJECT_ROOT / "automation" / "data" / "sustainability_topics.json"
TODAY = "2026-04-16" # 可参数化
def load_topics():
with open(TOPICS_FILE, 'r', encoding='utf-8') as f:
return json.load(f)
def extract_topic_id(filename: str) -> str:
"""从文件名提取 topic ID,如 zhihu_A01_zhihu.html -> A01"""
parts = filename.stem.split('_')
if len(parts) >= 2:
return parts[1]
return None
def main():
topics = load_topics()
topics_by_id = {t['id']: t for t in topics}
release_path = RELEASE_DIR / TODAY
if not release_path.exists():
print(f"错误:发布日期目录不存在 {release_path}")
return
html_files = list(release_path.rglob("*.html"))
print(f"找到 {len(html_files)} 个HTML文件,开始合规审查...\n")
results = []
for html_file in html_files:
platform = html_file.parent.name
topic_id = extract_topic_id(html_file)
topic_data = topics_by_id.get(topic_id) if topic_id else None
# 读取HTML
with open(html_file, 'r', encoding='utf-8') as f:
html_content = f.read()
# 执行合规检查
result = check_article(html_content, platform, topic_data)
result['file'] = str(html_file.relative_to(PROJECT_ROOT))
result['platform'] = platform
result['topic_id'] = topic_id
result['topic_title'] = topic_data.get('title') if topic_data else "未知"
results.append(result)
status = "✅ PASS" if result['passed'] else "❌ FAIL"
print(f"{status} {topic_id} {platform:12} {result['topic_title'][:30]:30} 问题数: {len(result['issues'])} 得分: {result['score']}")
# 汇总报告
passed = sum(1 for r in results if r['passed'])
failed = len(results) - passed
avg_score = sum(r['score'] for r in results) / len(results) if results else 0
print(f"\n========== 合规审查汇总 ==========")
print(f"总计: {len(results)}")
print(f"通过: {passed}")
print(f"失败: {failed}")
print(f"平均分: {avg_score:.1f}")
# 保存详细报告
report = {
"date": TODAY,
"summary": {
"total": len(results),
"passed": passed,
"failed": failed,
"average_score": avg_score
},
"details": results
}
report_file = PROJECT_ROOT / "automation" / "data" / "drafts" / TODAY / "compliance_summary.json"
report_file.parent.mkdir(parents=True, exist_ok=True)
with open(report_file, 'w', encoding='utf-8') as f:
json.dump(report, f, ensure_ascii=False, indent=2)
print(f"\n📁 详细报告已保存: {report_file}")
# 列出失败项
if failed > 0:
print("\n⚠️ 需要修复的文章:")
for r in results:
if not r['passed']:
print(f" {r['file']}")
for issue in r['issues'][:3]: # 只显示前3个问题
print(f" - {issue['type']}/{issue.get('category','')}: {issue.get('suggestion','')}")
if len(r['issues']) > 3:
print(f" ... 等共{len(r['issues'])}个问题")
else:
print("\n🎉 所有文章均通过合规审查!")
if __name__ == "__main__":
main()