109 lines
3.6 KiB
Python
109 lines
3.6 KiB
Python
#!/usr/bin/env python3
|
|
"""
|
|
批量合规审查脚本
|
|
遍历指定日期所有发布版本,执行合规检查,生成汇总报告
|
|
"""
|
|
|
|
import json
|
|
import re
|
|
from pathlib import Path
|
|
from datetime import datetime
|
|
import sys
|
|
|
|
PROJECT_ROOT = Path(__file__).parent.parent
|
|
sys.path.insert(0, str(PROJECT_ROOT))
|
|
from scripts.compliance_checker import check_article
|
|
|
|
# 配置
|
|
RELEASE_DIR = PROJECT_ROOT / "automation" / "data" / "releases"
|
|
TOPICS_FILE = PROJECT_ROOT / "automation" / "data" / "sustainability_topics.json"
|
|
TODAY = "2026-04-16" # 可参数化
|
|
|
|
def load_topics():
|
|
with open(TOPICS_FILE, 'r', encoding='utf-8') as f:
|
|
return json.load(f)
|
|
|
|
def extract_topic_id(filename: str) -> str:
|
|
"""从文件名提取 topic ID,如 zhihu_A01_zhihu.html -> A01"""
|
|
parts = filename.stem.split('_')
|
|
if len(parts) >= 2:
|
|
return parts[1]
|
|
return None
|
|
|
|
def main():
|
|
topics = load_topics()
|
|
topics_by_id = {t['id']: t for t in topics}
|
|
|
|
release_path = RELEASE_DIR / TODAY
|
|
if not release_path.exists():
|
|
print(f"错误:发布日期目录不存在 {release_path}")
|
|
return
|
|
|
|
html_files = list(release_path.rglob("*.html"))
|
|
print(f"找到 {len(html_files)} 个HTML文件,开始合规审查...\n")
|
|
|
|
results = []
|
|
for html_file in html_files:
|
|
platform = html_file.parent.name
|
|
topic_id = extract_topic_id(html_file)
|
|
topic_data = topics_by_id.get(topic_id) if topic_id else None
|
|
|
|
# 读取HTML
|
|
with open(html_file, 'r', encoding='utf-8') as f:
|
|
html_content = f.read()
|
|
|
|
# 执行合规检查
|
|
result = check_article(html_content, platform, topic_data)
|
|
result['file'] = str(html_file.relative_to(PROJECT_ROOT))
|
|
result['platform'] = platform
|
|
result['topic_id'] = topic_id
|
|
result['topic_title'] = topic_data.get('title') if topic_data else "未知"
|
|
results.append(result)
|
|
|
|
status = "✅ PASS" if result['passed'] else "❌ FAIL"
|
|
print(f"{status} {topic_id} {platform:12} {result['topic_title'][:30]:30} 问题数: {len(result['issues'])} 得分: {result['score']}")
|
|
|
|
# 汇总报告
|
|
passed = sum(1 for r in results if r['passed'])
|
|
failed = len(results) - passed
|
|
avg_score = sum(r['score'] for r in results) / len(results) if results else 0
|
|
|
|
print(f"\n========== 合规审查汇总 ==========")
|
|
print(f"总计: {len(results)} 篇")
|
|
print(f"通过: {passed} 篇")
|
|
print(f"失败: {failed} 篇")
|
|
print(f"平均分: {avg_score:.1f}")
|
|
|
|
# 保存详细报告
|
|
report = {
|
|
"date": TODAY,
|
|
"summary": {
|
|
"total": len(results),
|
|
"passed": passed,
|
|
"failed": failed,
|
|
"average_score": avg_score
|
|
},
|
|
"details": results
|
|
}
|
|
report_file = PROJECT_ROOT / "automation" / "data" / "drafts" / TODAY / "compliance_summary.json"
|
|
report_file.parent.mkdir(parents=True, exist_ok=True)
|
|
with open(report_file, 'w', encoding='utf-8') as f:
|
|
json.dump(report, f, ensure_ascii=False, indent=2)
|
|
print(f"\n📁 详细报告已保存: {report_file}")
|
|
|
|
# 列出失败项
|
|
if failed > 0:
|
|
print("\n⚠️ 需要修复的文章:")
|
|
for r in results:
|
|
if not r['passed']:
|
|
print(f" {r['file']}")
|
|
for issue in r['issues'][:3]: # 只显示前3个问题
|
|
print(f" - {issue['type']}/{issue.get('category','')}: {issue.get('suggestion','')}")
|
|
if len(r['issues']) > 3:
|
|
print(f" ... 等共{len(r['issues'])}个问题")
|
|
else:
|
|
print("\n🎉 所有文章均通过合规审查!")
|
|
|
|
if __name__ == "__main__":
|
|
main()
|