218 lines
8.2 KiB
Python
218 lines
8.2 KiB
Python
#!/usr/bin/env python3
|
||
"""
|
||
内容清洗工具集:所有 AI 思考内容/噪音段落的清洗逻辑集中管理
|
||
各脚本(writer/outline/compliance_optimizer)统一引用此模块
|
||
"""
|
||
import re
|
||
from typing import List, Tuple
|
||
|
||
THINKING_PATTERNS: List[str] = [
|
||
r'^(好的|好的,|好[之,]|我来|让我|我将|我这就).*?(?=\n|$)',
|
||
r'^(以下|下面是|这是|为您|根据).*?(?=\n|$)',
|
||
r'^基于.*?(?=\n|$)',
|
||
r'^【.*?】',
|
||
r'^这里.*?(?=\n|$)',
|
||
r'\n+希望[这以].*?$',
|
||
r'\n+如果.*?$',
|
||
r'\n+若有.*?$',
|
||
r'\n+如有.*?$',
|
||
r'\n+\*\*免责.*?$',
|
||
r'^(这是按照要求|我已按|根据您的要求|^首先|^其次|^最后|^补充|^完成后).*?(?=\n|$)',
|
||
r'^(以下是|下面为|这是完整|已按要求|已完成|处理完成).*?(?=\n|$)',
|
||
]
|
||
|
||
AI_PREFACE_PATTERNS: List[str] = [
|
||
r'^(好的[,,]?|好的 |我来|让我|我将|我这就|以下|下面|这是|为您|基于)',
|
||
r'^(这是按照要求|我已按|根据您的要求|^首先|^其次|^最后|^补充|^完成后|以下是|下面为|这是完整|已按要求|已完成)',
|
||
]
|
||
|
||
AI_VERBAL_PATTERNS: List[str] = [
|
||
r'^(首先|其次|最后)(,|,)?',
|
||
r'^总的来说',
|
||
r'^值得注意的是',
|
||
r'^换句话说',
|
||
r'^总而言之',
|
||
r'^简而言之',
|
||
r'^一言以蔽之',
|
||
r'^可以说',
|
||
r'^不难发现',
|
||
r'^由此可见',
|
||
r'^综上所述',
|
||
r'^通过以上',
|
||
]
|
||
|
||
_cached_clean_rules = None
|
||
|
||
def _load_clean_rules():
|
||
global _cached_clean_rules
|
||
if _cached_clean_rules is not None:
|
||
return _cached_clean_rules
|
||
|
||
try:
|
||
from app.core.prompt_loader import _get_session
|
||
from app.models import ContentCleanRule
|
||
session = _get_session()
|
||
try:
|
||
rows = session.query(ContentCleanRule).filter(ContentCleanRule.is_active == True).order_by(ContentCleanRule.sort_order).all()
|
||
if rows:
|
||
result = {"thinking": [], "preface": [], "verbosity": [], "html_thinking": []}
|
||
for r in rows:
|
||
rule_type = r.rule_type or "thinking"
|
||
if rule_type in result:
|
||
result[rule_type].append(r.pattern)
|
||
_cached_clean_rules = result
|
||
return _cached_clean_rules
|
||
finally:
|
||
session.close()
|
||
except Exception:
|
||
pass
|
||
|
||
_cached_clean_rules = {
|
||
"thinking": THINKING_PATTERNS,
|
||
"preface": AI_PREFACE_PATTERNS,
|
||
"verbosity": AI_VERBAL_PATTERNS,
|
||
"html_thinking": [],
|
||
}
|
||
return _cached_clean_rules
|
||
|
||
|
||
def _get_thinking_patterns() -> List[str]:
|
||
rules = _load_clean_rules()
|
||
return rules.get("thinking", THINKING_PATTERNS)
|
||
|
||
|
||
def _get_preface_patterns() -> List[str]:
|
||
rules = _load_clean_rules()
|
||
return rules.get("preface", AI_PREFACE_PATTERNS)
|
||
|
||
|
||
def _get_verbal_patterns() -> List[str]:
|
||
rules = _load_clean_rules()
|
||
return rules.get("verbosity", AI_VERBAL_PATTERNS)
|
||
|
||
|
||
def strip_thinking(text: str) -> str:
|
||
"""清洗 AI 思考前缀/后缀(正则替换,支持纯文本和 HTML 内联)"""
|
||
for pat in _get_thinking_patterns():
|
||
text = re.sub(pat, '', text, flags=re.MULTILINE)
|
||
return text.strip()
|
||
|
||
def strip_thinking_html(html: str) -> str:
|
||
"""清洗 HTML 中的 AI 思考段落(处理 <p>/<div> 包裹的情况)"""
|
||
rules = _load_clean_rules()
|
||
patterns = rules.get("html_thinking", [])
|
||
if not patterns:
|
||
patterns = [
|
||
r'<p[^>]*>(好的|好的,|好[的,]|我来|让我|我将|我这就|以下|下面|这是|为您|基于|这是按照要求|我已按|根据您的要求|^首先|^其次|^最后|^补充|^完成后|以下是|下面为|这是完整|已按要求|已完成).*?</p>',
|
||
r'<div[^>]*>(好的|好的,|好[的,]|我来|让我|我将|我这就|以下|下面|这是|为您|基于|这是按照要求|我已按|根据您的要求|^首先|^其次|^最后|^补充|^完成后|以下是|下面为|这是完整|已按要求|已完成).*?</div>',
|
||
r'<p[^>]*>首先.*?</p>',
|
||
r'<p[^>]*>其次.*?</p>',
|
||
r'<p[^>]*>最后.*?</p>',
|
||
r'<p[^>]*>(总的来说|值得注意的是|换句话说|总而言之|简而言之|一言以蔽之|可以说|不难发现|由此可见|综上所述).*?</p>',
|
||
r'<div[^>]*>(总的来说|值得注意的是|换句话说|总而言之|简而言之|一言以蔽之|可以说|不难发现|由此可见|综上所述).*?</div>',
|
||
r'<p[^>]*>(开头钩子|核心观点|受众痛点|独特视角|差异化切入|内容形式).*?</p>',
|
||
]
|
||
for pat in patterns:
|
||
html = re.sub(pat, '', html, flags=re.IGNORECASE)
|
||
return html
|
||
|
||
def strip_ai_preface(text: str) -> str:
|
||
"""清洗以 AI 自述开头的整段说明文字(含代码围栏块)"""
|
||
lines = text.split('\n')
|
||
result = []
|
||
skip_mode = False
|
||
code_start = re.compile(r'^```')
|
||
for line in lines:
|
||
stripped = line.strip()
|
||
if skip_mode:
|
||
if code_start.match(stripped):
|
||
skip_mode = False
|
||
continue
|
||
should_skip = False
|
||
for pat in _get_preface_patterns():
|
||
if re.match(pat, stripped):
|
||
should_skip = True
|
||
break
|
||
if should_skip:
|
||
if code_start.match(stripped) or '```' in stripped:
|
||
skip_mode = True
|
||
continue
|
||
result.append(line)
|
||
return '\n'.join(result).strip()
|
||
|
||
def strip_ai_verbosity(text: str) -> str:
|
||
"""清洗正文中常见的 AI 套话段落"""
|
||
lines = text.split('\n')
|
||
result = []
|
||
for line in lines:
|
||
stripped = line.strip()
|
||
skip = False
|
||
for pat in _get_verbal_patterns():
|
||
if re.match(pat, stripped):
|
||
skip = True
|
||
break
|
||
if not skip:
|
||
result.append(line)
|
||
return '\n'.join(result).strip()
|
||
|
||
def clean_markdown_content(text: str) -> str:
|
||
"""清洗 markdown 正文:去思考内容 + 去 AI 套话 + 去格式噪音"""
|
||
text = strip_thinking(text)
|
||
text = strip_ai_preface(text)
|
||
text = strip_ai_verbosity(text)
|
||
lines = text.split('\n')
|
||
cleaned = []
|
||
in_code = False
|
||
for line in lines:
|
||
if line.strip().startswith('```'):
|
||
in_code = not in_code
|
||
continue
|
||
if in_code:
|
||
continue
|
||
line = re.sub(r'^#{1,6}\s+', '', line)
|
||
line = re.sub(r'^[\-\*\+]\s+', '', line)
|
||
line = re.sub(r'^\d+[\.\)]\s+', '', line)
|
||
line = re.sub(r'\*{1,3}([^*]+)\*{1,3}', r'\1', line)
|
||
cleaned.append(line)
|
||
return '\n'.join(cleaned).strip()
|
||
|
||
def clean_html_content(html: str) -> str:
|
||
"""清洗 HTML 输出:去 markdown 代码围栏 + AI 思考 + 图片占位 + 结构标签"""
|
||
html = re.sub(r'^```+\w*\s*\n?', '', html)
|
||
html = html.strip()
|
||
html = re.sub(r'\n?```+\s*$', '', html)
|
||
html = strip_thinking_html(html)
|
||
# 移除 LLM 插入的建议配图占位(<p><!-- 建议配图:... --></p>)
|
||
html = re.sub(r'<p>\s*<!--\s*建议配图.*?-->\s*</p>', '', html, flags=re.DOTALL)
|
||
# 移除可能变成可见文本的建议配图文字
|
||
html = re.sub(r'<!--\s*建议配图.*?-->', '', html, flags=re.DOTALL)
|
||
html = re.sub(r'建议配图:.*?(?=<|$)', '', html)
|
||
return html
|
||
|
||
def clean_full_pipeline(text: str, output_format: str = 'markdown') -> str:
|
||
"""
|
||
完整清洗流程:
|
||
- markdown 输入:先去思考前缀 → 再去格式噪音 → 再转 HTML
|
||
- html 输入:直接去代码围栏 + 思考注释
|
||
"""
|
||
if output_format == 'html':
|
||
return clean_html_content(text)
|
||
return clean_markdown_content(text)
|
||
|
||
def get_statistics(text: str) -> dict:
|
||
"""返回清洗前后的行数/字数统计(用于日志)"""
|
||
original_lines = len(text.split('\n'))
|
||
original_chars = len(text)
|
||
cleaned = strip_thinking(text)
|
||
cleaned = strip_ai_preface(cleaned)
|
||
cleaned = strip_ai_verbosity(cleaned)
|
||
cleaned_lines = len(cleaned.split('\n'))
|
||
cleaned_chars = len(cleaned)
|
||
return {
|
||
'original_lines': original_lines,
|
||
'cleaned_lines': cleaned_lines,
|
||
'original_chars': original_chars,
|
||
'cleaned_chars': cleaned_chars,
|
||
'dropped_lines': original_lines - cleaned_lines,
|
||
'dropped_chars': original_chars - cleaned_chars,
|
||
} |