Files

218 lines
8.2 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
"""
内容清洗工具集:所有 AI 思考内容/噪音段落的清洗逻辑集中管理
各脚本(writer/outline/compliance_optimizer)统一引用此模块
"""
import re
from typing import List, Tuple
THINKING_PATTERNS: List[str] = [
r'^(好的|好的,|好[之,]|我来|让我|我将|我这就).*?(?=\n|$)',
r'^(以下|下面是|这是|为您|根据).*?(?=\n|$)',
r'^基于.*?(?=\n|$)',
r'^【.*?】',
r'^这里.*?(?=\n|$)',
r'\n+希望[这以].*?$',
r'\n+如果.*?$',
r'\n+若有.*?$',
r'\n+如有.*?$',
r'\n+\*\*免责.*?$',
r'^(这是按照要求|我已按|根据您的要求|^首先|^其次|^最后|^补充|^完成后).*?(?=\n|$)',
r'^(以下是|下面为|这是完整|已按要求|已完成|处理完成).*?(?=\n|$)',
]
AI_PREFACE_PATTERNS: List[str] = [
r'^(好的[,,]?|好的 |我来|让我|我将|我这就|以下|下面|这是|为您|基于)',
r'^(这是按照要求|我已按|根据您的要求|^首先|^其次|^最后|^补充|^完成后|以下是|下面为|这是完整|已按要求|已完成)',
]
AI_VERBAL_PATTERNS: List[str] = [
r'^(首先|其次|最后)(|,)?',
r'^总的来说',
r'^值得注意的是',
r'^换句话说',
r'^总而言之',
r'^简而言之',
r'^一言以蔽之',
r'^可以说',
r'^不难发现',
r'^由此可见',
r'^综上所述',
r'^通过以上',
]
_cached_clean_rules = None
def _load_clean_rules():
global _cached_clean_rules
if _cached_clean_rules is not None:
return _cached_clean_rules
try:
from app.core.prompt_loader import _get_session
from app.models import ContentCleanRule
session = _get_session()
try:
rows = session.query(ContentCleanRule).filter(ContentCleanRule.is_active == True).order_by(ContentCleanRule.sort_order).all()
if rows:
result = {"thinking": [], "preface": [], "verbosity": [], "html_thinking": []}
for r in rows:
rule_type = r.rule_type or "thinking"
if rule_type in result:
result[rule_type].append(r.pattern)
_cached_clean_rules = result
return _cached_clean_rules
finally:
session.close()
except Exception:
pass
_cached_clean_rules = {
"thinking": THINKING_PATTERNS,
"preface": AI_PREFACE_PATTERNS,
"verbosity": AI_VERBAL_PATTERNS,
"html_thinking": [],
}
return _cached_clean_rules
def _get_thinking_patterns() -> List[str]:
rules = _load_clean_rules()
return rules.get("thinking", THINKING_PATTERNS)
def _get_preface_patterns() -> List[str]:
rules = _load_clean_rules()
return rules.get("preface", AI_PREFACE_PATTERNS)
def _get_verbal_patterns() -> List[str]:
rules = _load_clean_rules()
return rules.get("verbosity", AI_VERBAL_PATTERNS)
def strip_thinking(text: str) -> str:
"""清洗 AI 思考前缀/后缀(正则替换,支持纯文本和 HTML 内联)"""
for pat in _get_thinking_patterns():
text = re.sub(pat, '', text, flags=re.MULTILINE)
return text.strip()
def strip_thinking_html(html: str) -> str:
"""清洗 HTML 中的 AI 思考段落(处理 <p>/<div> 包裹的情况)"""
rules = _load_clean_rules()
patterns = rules.get("html_thinking", [])
if not patterns:
patterns = [
r'<p[^>]*>(好的|好的,|好[的,]|我来|让我|我将|我这就|以下|下面|这是|为您|基于|这是按照要求|我已按|根据您的要求|^首先|^其次|^最后|^补充|^完成后|以下是|下面为|这是完整|已按要求|已完成).*?</p>',
r'<div[^>]*>(好的|好的,|好[的,]|我来|让我|我将|我这就|以下|下面|这是|为您|基于|这是按照要求|我已按|根据您的要求|^首先|^其次|^最后|^补充|^完成后|以下是|下面为|这是完整|已按要求|已完成).*?</div>',
r'<p[^>]*>首先.*?</p>',
r'<p[^>]*>其次.*?</p>',
r'<p[^>]*>最后.*?</p>',
r'<p[^>]*>(总的来说|值得注意的是|换句话说|总而言之|简而言之|一言以蔽之|可以说|不难发现|由此可见|综上所述).*?</p>',
r'<div[^>]*>(总的来说|值得注意的是|换句话说|总而言之|简而言之|一言以蔽之|可以说|不难发现|由此可见|综上所述).*?</div>',
r'<p[^>]*>(开头钩子|核心观点|受众痛点|独特视角|差异化切入|内容形式).*?</p>',
]
for pat in patterns:
html = re.sub(pat, '', html, flags=re.IGNORECASE)
return html
def strip_ai_preface(text: str) -> str:
"""清洗以 AI 自述开头的整段说明文字(含代码围栏块)"""
lines = text.split('\n')
result = []
skip_mode = False
code_start = re.compile(r'^```')
for line in lines:
stripped = line.strip()
if skip_mode:
if code_start.match(stripped):
skip_mode = False
continue
should_skip = False
for pat in _get_preface_patterns():
if re.match(pat, stripped):
should_skip = True
break
if should_skip:
if code_start.match(stripped) or '```' in stripped:
skip_mode = True
continue
result.append(line)
return '\n'.join(result).strip()
def strip_ai_verbosity(text: str) -> str:
"""清洗正文中常见的 AI 套话段落"""
lines = text.split('\n')
result = []
for line in lines:
stripped = line.strip()
skip = False
for pat in _get_verbal_patterns():
if re.match(pat, stripped):
skip = True
break
if not skip:
result.append(line)
return '\n'.join(result).strip()
def clean_markdown_content(text: str) -> str:
"""清洗 markdown 正文:去思考内容 + 去 AI 套话 + 去格式噪音"""
text = strip_thinking(text)
text = strip_ai_preface(text)
text = strip_ai_verbosity(text)
lines = text.split('\n')
cleaned = []
in_code = False
for line in lines:
if line.strip().startswith('```'):
in_code = not in_code
continue
if in_code:
continue
line = re.sub(r'^#{1,6}\s+', '', line)
line = re.sub(r'^[\-\*\+]\s+', '', line)
line = re.sub(r'^\d+[\.\)]\s+', '', line)
line = re.sub(r'\*{1,3}([^*]+)\*{1,3}', r'\1', line)
cleaned.append(line)
return '\n'.join(cleaned).strip()
def clean_html_content(html: str) -> str:
"""清洗 HTML 输出:去 markdown 代码围栏 + AI 思考 + 图片占位 + 结构标签"""
html = re.sub(r'^```+\w*\s*\n?', '', html)
html = html.strip()
html = re.sub(r'\n?```+\s*$', '', html)
html = strip_thinking_html(html)
# 移除 LLM 插入的建议配图占位(<p><!-- 建议配图:... --></p>
html = re.sub(r'<p>\s*<!--\s*建议配图.*?-->\s*</p>', '', html, flags=re.DOTALL)
# 移除可能变成可见文本的建议配图文字
html = re.sub(r'<!--\s*建议配图.*?-->', '', html, flags=re.DOTALL)
html = re.sub(r'建议配图:.*?(?=<|$)', '', html)
return html
def clean_full_pipeline(text: str, output_format: str = 'markdown') -> str:
"""
完整清洗流程:
- markdown 输入:先去思考前缀 → 再去格式噪音 → 再转 HTML
- html 输入:直接去代码围栏 + 思考注释
"""
if output_format == 'html':
return clean_html_content(text)
return clean_markdown_content(text)
def get_statistics(text: str) -> dict:
"""返回清洗前后的行数/字数统计(用于日志)"""
original_lines = len(text.split('\n'))
original_chars = len(text)
cleaned = strip_thinking(text)
cleaned = strip_ai_preface(cleaned)
cleaned = strip_ai_verbosity(cleaned)
cleaned_lines = len(cleaned.split('\n'))
cleaned_chars = len(cleaned)
return {
'original_lines': original_lines,
'cleaned_lines': cleaned_lines,
'original_chars': original_chars,
'cleaned_chars': cleaned_chars,
'dropped_lines': original_lines - cleaned_lines,
'dropped_chars': original_chars - cleaned_chars,
}