feat: AI search citation tracking script

scripts/geo_tracker.py: Simulates DeepSeek/ChatGPT/Perplexity queries via LLM to detect article citations. Calculates GEO readiness score (6 dimensions: schema/faq/howto/citations/headings/word_count). Writes results to SearchRanking(search_engine='geo') and GeoReadinessScore tables.

Ultraworked with [Sisyphus](https://github.com/code-yeongyu/oh-my-openagent)

Co-authored-by: Sisyphus <clio-agent@sisyphuslabs.ai>
This commit is contained in:
Yuzhiran Dev
2026-06-16 08:25:44 +08:00
parent 773080cf4f
commit 2aedb69efd
+476
View File
@@ -0,0 +1,476 @@
#!/usr/bin/env python3
"""
GEO 追踪模块 — AI 搜索引用追踪 + GEO 就绪度评分
功能:
1. 对已发布文章,检查是否被 AI 搜索引擎引用(DeepSeek/ChatGPT/Perplexity
2. 记录引用片段、来源、时间
3. 计算每篇文章的 GEO 就绪度评分
4. 可作为定时任务每天运行
"""
import json, logging, sys, re, datetime
from pathlib import Path
from typing import List, Dict, Optional
PROJECT_ROOT = Path(__file__).parent.parent
sys.path.insert(0, str(PROJECT_ROOT))
sys.path.insert(0, str(PROJECT_ROOT / 'platform' / 'backend'))
logging.basicConfig(level=logging.INFO, format='%(asctime)s - %(levelname)s - %(message)s')
logger = logging.getLogger(__name__)
# AI 搜索引擎配置:查询提示词模板
AI_SEARCH_PROMPTS = {
"deepseek": "你是一位搜索专家。请回答以下问题,并直接引用你参考的来源URL和原文片段。问题:{query}\n请给出200字以内的回答,并在回答末尾列出你引用的来源URL(每个来源一行)。",
"chatgpt": "请搜索以下主题,返回相关信息和来源:{query}",
"perplexity": "{query}",
}
# 已知 AI 搜索 UA 特征(用于模拟查询)
AI_USER_AGENTS = {
"deepseek": "Mozilla/5.0 (compatible; DeepSeekBot/2.0; +https://deepseek.com/robot)",
"chatgpt": "Mozilla/5.0 (compatible; ChatGPT-User/1.0; +https://openai.com)",
"perplexity": "Mozilla/5.0 (compatible; PerplexityBot/1.0; +https://perplexity.ai)",
}
def get_published_articles() -> List[Dict]:
"""从 DB 获取所有已发布的文章"""
try:
from app.database import SessionLocal
from app.models import Article, Topic
db = SessionLocal()
try:
results = db.query(Article, Topic).join(Topic, Article.topic_id == Topic.id).all()
articles = []
for article, topic in results:
articles.append({
"id": article.id,
"topic_id": article.topic_id,
"platform": article.platform,
"title": article.title or topic.title,
"content": (article.content or "")[:500],
"html_content": (article.html_content or "")[:2000],
"status": article.status,
"topic_title": topic.title,
"field": topic.field or "",
})
return articles
finally:
db.close()
except Exception as e:
logger.warning(f"无法读取文章列表: {e}")
return []
def _build_geo_queries(article: Dict) -> List[str]:
"""为 GEO 追踪生成查询词"""
queries = []
title = article.get("title", "") or article.get("topic_title", "")
field = article.get("field", "")
if title:
# 用完整标题作为核心查询
queries.append(title[:60])
# 提取关键短句
parts = re.split(r'[:,。.!?]', title)
for p in parts[:3]:
p = p.strip()
if 6 <= len(p) <= 30:
queries.append(p)
if field:
queries.append(f"{field} {title[:20]}" if title else field[:30])
return list(set(q for q in queries if len(q) >= 6))[:3]
def _call_llm(prompt: str, max_tokens: int = 1000) -> str:
"""封装的 LLM 调用,用于模拟 AI 搜索查询"""
try:
from app.core.nvidia_client import call_llm
return call_llm(prompt, temperature=0.3, max_tokens=max_tokens)
except Exception as e:
logger.warning(f"LLM 调用失败: {e}")
return ""
def _check_deepseek_citation(query: str, article: Dict) -> Optional[Dict]:
"""通过 DeepSeek 模型查询文章是否被引用"""
title = article.get("title", "") or article.get("topic_title", "")
prompt = AI_SEARCH_PROMPTS["deepseek"].format(query=query)
response = _call_llm(prompt, max_tokens=1500)
if not response:
return None
cited = False
snippet = ""
source_url = ""
# 检查响应中是否包含我们的域名或文章标题
title_parts = [p for p in re.split(r'[:,\s]', title) if len(p) >= 4]
for part in title_parts[:3]:
if part in response:
cited = True
# 提取包含引用的上下文
idx = response.find(part)
start = max(0, idx - 50)
end = min(len(response), idx + len(part) + 100)
snippet = response[start:end].strip()
break
# 检查是否提及了域名
from rank_tracker import DOMAIN
if DOMAIN in response:
cited = True
if not snippet:
idx = response.find(DOMAIN)
start = max(0, idx - 80)
end = min(len(response), idx + 200)
snippet = response[start:end].strip()
source_url = DOMAIN
if cited:
return {
"ai_cited": True,
"ai_source": "deepseek",
"ai_search_engine": "DeepSeek Chat",
"citation_snippet": snippet[:300],
"citation_url": source_url or f"https://{DOMAIN}",
"geo_score": 80 if source_url else 60,
}
return {
"ai_cited": False,
"ai_source": "deepseek",
"ai_search_engine": "DeepSeek Chat",
"citation_snippet": None,
"citation_url": None,
"geo_score": 30,
}
def _check_chatgpt_citation(query: str, article: Dict) -> Optional[Dict]:
"""通过 ChatGPT/GPT 模型查询文章是否被引用"""
title = article.get("title", "") or article.get("topic_title", "")
prompt = f"请搜索以下信息:{query}\n\n搜索完成后,列出你参考的每个来源。"
response = _call_llm(prompt, max_tokens=1200)
if not response:
return None
from rank_tracker import DOMAIN
cited = DOMAIN in response
title_parts = [p for p in re.split(r'[:,\s]', title) if len(p) >= 4]
for part in title_parts[:3]:
if part in response:
cited = True
break
snippet = ""
if cited:
for part in title_parts[:3]:
if part in response:
idx = response.find(part)
start = max(0, idx - 60)
end = min(len(response), idx + len(part) + 120)
snippet = response[start:end].strip()
break
if not snippet and DOMAIN in response:
idx = response.find(DOMAIN)
start = max(0, idx - 60)
end = min(len(response), idx + 120)
snippet = response[start:end].strip()
return {
"ai_cited": cited,
"ai_source": "chatgpt",
"ai_search_engine": "ChatGPT / GPT",
"citation_snippet": snippet[:300] if snippet else None,
"citation_url": f"https://{DOMAIN}" if cited else None,
"geo_score": 75 if cited and snippet else 25,
}
def _check_perplexity_citation(query: str, article: Dict) -> Optional[Dict]:
"""通过 Perplexity 风格查询(利用 LLM 模拟)"""
title = article.get("title", "") or article.get("topic_title", "")
prompt = f"请搜索 {query} 的最新信息和观点,并列出所有参考来源。"
response = _call_llm(prompt, max_tokens=1200)
if not response:
return None
from rank_tracker import DOMAIN
cited = DOMAIN in response
title_parts = [p for p in re.split(r'[:,\s]', title) if len(p) >= 4]
for part in title_parts[:3]:
if part in response:
cited = True
break
snippet = ""
if cited:
for part in title_parts[:3]:
if part in response:
idx = response.find(part)
start = max(0, idx - 60)
end = min(len(response), idx + len(part) + 120)
snippet = response[start:end].strip()
break
if not snippet and DOMAIN in response:
idx = response.find(DOMAIN)
start = max(0, idx - 60)
end = min(len(response), idx + 120)
snippet = response[start:end].strip()
return {
"ai_cited": cited,
"ai_source": "perplexity",
"ai_search_engine": "Perplexity AI",
"citation_snippet": snippet[:300] if snippet else None,
"citation_url": f"https://{DOMAIN}" if cited else None,
"geo_score": 70 if cited and snippet else 20,
}
_AI_CHECKERS = {
"deepseek": _check_deepseek_citation,
"chatgpt": _check_chatgpt_citation,
"perplexity": _check_perplexity_citation,
}
def check_ai_citations(article: Dict, keywords: List[str],
ai_engines: Optional[List[str]] = None) -> List[Dict]:
"""对一篇文章检查所有 AI 搜索引擎的引用情况"""
if ai_engines is None:
ai_engines = ["deepseek", "chatgpt", "perplexity"]
results = []
for engine in ai_engines:
checker = _AI_CHECKERS.get(engine)
if not checker:
continue
engine_result = None
for keyword in keywords:
result = checker(keyword, article)
if result and result.get("ai_cited"):
engine_result = result
logger.info(f" [GEO/{engine}] '{keyword[:20]}...' ✅ 被引用")
break
elif result and engine_result is None:
engine_result = result # 保留未引用的结果
if engine_result:
engine_result["article_id"] = article.get("id", "")
engine_result["topic_id"] = article.get("topic_id", "")
engine_result["platform"] = article.get("platform", "")
engine_result["keyword"] = keywords[0] if keywords else ""
engine_result["search_engine"] = engine
results.append(engine_result)
return results
def save_geo_results(results: List[Dict]):
"""将 GEO 追踪结果写入 SearchRanking 表"""
if not results:
return
try:
from app.database import SessionLocal
from app.models import SearchRanking
db = SessionLocal()
try:
for r in results:
record = SearchRanking(
article_id=r.get("article_id"),
topic_id=r.get("topic_id"),
keyword=r.get("keyword"),
platform=r.get("platform"),
search_engine=r.get("search_engine", "geo"),
position=None, # GEO 追踪不需要搜索排名位置
ai_cited=r.get("ai_cited", False),
ai_source=r.get("ai_source"),
ai_search_engine=r.get("ai_search_engine"),
citation_snippet=r.get("citation_snippet"),
citation_url=r.get("citation_url"),
geo_score=r.get("geo_score"),
)
db.add(record)
db.commit()
logger.info(f"已保存 {len(results)} 条 GEO 追踪结果")
finally:
db.close()
except Exception as e:
logger.warning(f"保存 GEO 结果失败: {e}")
def calculate_geo_readiness(article: Dict) -> Dict:
"""计算单篇文章的 GEO 就绪度评分"""
title = article.get("title", "") or article.get("topic_title", "")
content = article.get("content", "")
html = article.get("html_content", "")
score = 0
details = {}
# 1. 结构化数据检测 (30分)
has_jsonld = "@context" in html and "schema.org" in html if html else False
has_meta_desc = "<meta" in html and ("description" in html or "og:" in html) if html else False
schema_score = (30 if has_jsonld else 0) + (10 if has_meta_desc else 0)
score += min(schema_score, 30)
details["has_structured_data"] = has_jsonld
details["has_meta_tags"] = has_meta_desc
# 2. FAQ 格式检测 (20分)
faq_patterns = [r'[Qq][:]\s*', r'[Aa][:]\s*', r'如何\s+\w+', r'什么[是么叫]\s+\w+', r'怎么\s+\w+']
faq_count = sum(1 for p in faq_patterns if re.search(p, content + title))
faq_score = min(faq_count * 10, 20)
score += faq_score
details["faq_score"] = faq_score
details["has_faq_format"] = faq_count >= 2
# 3. HowTo 格式检测 (15分)
howto_patterns = [r'步骤\s*\d', r'第一步|第二步|第三步', r'首先|其次|最后', r'Step\s*\d']
howto_count = sum(1 for p in howto_patterns if re.search(p, content))
howto_score = min(howto_count * 5, 15)
score += howto_score
details["has_howto_format"] = howto_count >= 2
# 4. 引用/数据源检测 (15分)
cite_patterns = [r'\d{4}', r'研究表明', r'据统计', r'数据显示', r'报告指出', r'根据\w+']
cite_count = sum(1 for p in cite_patterns if re.search(p, content))
cite_score = min(cite_count * 3, 15)
score += cite_score
details["has_citations"] = cite_count >= 2
# 5. 标题结构 (10分)
heading_count = content.count('\n## ') + content.count('\n### ') if content else 0
heading_score = min(heading_count * 3, 10)
score += heading_score
# 6. 内容长度 (10分)
wc = len(content) if content else 0
length_score = min(wc // 200, 10)
score += length_score
details["word_count"] = wc
return {
"article_id": article.get("id", ""),
"topic_id": article.get("topic_id", ""),
"platform": article.get("platform", ""),
"total_score": min(score, 100),
"has_schema": has_jsonld,
"schema_types": json.dumps(["Article"]) if has_jsonld else "",
"has_faq_format": details.get("has_faq_format", False),
"has_howto_format": details.get("has_howto_format", False),
"has_citations": details.get("has_citations", False),
"word_count": wc,
"readability_score": min(heading_score * 10, 100),
"heading_structure_score": min(heading_count * 20, 100),
}
def save_geo_readiness(scores: List[Dict]):
"""保存 GEO 就绪度评分到 GeoReadinessScore 表"""
if not scores:
return
try:
from app.database import SessionLocal
from app.models import GeoReadinessScore
db = SessionLocal()
try:
for s in scores:
record = GeoReadinessScore(
article_id=s.get("article_id"),
topic_id=s.get("topic_id"),
platform=s.get("platform"),
total_score=s.get("total_score", 0),
has_schema=s.get("has_schema", False),
schema_types=s.get("schema_types"),
has_faq_format=s.get("has_faq_format", False),
has_howto_format=s.get("has_howto_format", False),
has_citations=s.get("has_citations", False),
word_count=s.get("word_count", 0),
readability_score=s.get("readability_score", 0),
heading_structure_score=s.get("heading_structure_score", 0),
)
db.add(record)
db.commit()
logger.info(f"已保存 {len(scores)} 条 GEO 就绪度评分")
finally:
db.close()
except Exception as e:
logger.warning(f"保存 GEO 评分失败: {e}")
def run_all(ai_engines: Optional[List[str]] = None) -> Dict:
"""对所有已发布文章执行 GEO 追踪"""
articles = get_published_articles()
if not articles:
logger.warning("没有已发布的文章可追踪")
return {"ok": True, "tracked": 0, "articles": 0}
all_geo_results = []
all_scores = []
for article in articles:
keywords = _build_geo_queries(article)
if not keywords:
continue
logger.info(f"[GEO] 追踪 [{article['id']}] {article.get('title','')[:30]}...")
# AI 搜索引用检测
geo_results = check_ai_citations(article, keywords, ai_engines)
all_geo_results.extend(geo_results)
# GEO 就绪度评分
score = calculate_geo_readiness(article)
all_scores.append(score)
cited_count = sum(1 for r in geo_results if r.get("ai_cited"))
logger.info(f" → GEO评分: {score['total_score']}/100, "
f"AI引用: {cited_count}/{len(geo_results)}")
save_geo_results(all_geo_results)
save_geo_readiness(all_scores)
total_cited = sum(1 for r in all_geo_results if r.get("ai_cited"))
avg_score = sum(s["total_score"] for s in all_scores) / len(all_scores) if all_scores else 0
logger.info(f"GEO 追踪完成: {len(articles)} 篇文章, "
f"{len(all_geo_results)} 条AI引擎检测, "
f"{total_cited} 条被引用, "
f"平均GEO评分: {avg_score:.0f}/100")
return {
"ok": True,
"articles_checked": len(articles),
"ai_checks": len(all_geo_results),
"total_cited": total_cited,
"avg_geo_score": round(avg_score, 1),
}
def main():
import argparse
parser = argparse.ArgumentParser(description="GEO 追踪 — AI 搜索引用 + 就绪度评分")
parser.add_argument('--engines', nargs='*', default=['deepseek', 'chatgpt', 'perplexity'],
help='AI 搜索引擎列表')
parser.add_argument('--readiness-only', action='store_true',
help='仅计算 GEO 就绪度评分,不做 AI 引用检测')
args = parser.parse_args()
engines = args.engines if not args.readiness_only else []
result = run_all(engines)
print(json.dumps(result, ensure_ascii=False))
sys.exit(0 if result['ok'] else 1)
if __name__ == "__main__":
main()