feat: AI search citation tracking script
scripts/geo_tracker.py: Simulates DeepSeek/ChatGPT/Perplexity queries via LLM to detect article citations. Calculates GEO readiness score (6 dimensions: schema/faq/howto/citations/headings/word_count). Writes results to SearchRanking(search_engine='geo') and GeoReadinessScore tables. Ultraworked with [Sisyphus](https://github.com/code-yeongyu/oh-my-openagent) Co-authored-by: Sisyphus <clio-agent@sisyphuslabs.ai>
This commit is contained in:
@@ -0,0 +1,476 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
GEO 追踪模块 — AI 搜索引用追踪 + GEO 就绪度评分
|
||||
|
||||
功能:
|
||||
1. 对已发布文章,检查是否被 AI 搜索引擎引用(DeepSeek/ChatGPT/Perplexity)
|
||||
2. 记录引用片段、来源、时间
|
||||
3. 计算每篇文章的 GEO 就绪度评分
|
||||
4. 可作为定时任务每天运行
|
||||
"""
|
||||
import json, logging, sys, re, datetime
|
||||
from pathlib import Path
|
||||
from typing import List, Dict, Optional
|
||||
|
||||
PROJECT_ROOT = Path(__file__).parent.parent
|
||||
sys.path.insert(0, str(PROJECT_ROOT))
|
||||
sys.path.insert(0, str(PROJECT_ROOT / 'platform' / 'backend'))
|
||||
|
||||
logging.basicConfig(level=logging.INFO, format='%(asctime)s - %(levelname)s - %(message)s')
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# AI 搜索引擎配置:查询提示词模板
|
||||
AI_SEARCH_PROMPTS = {
|
||||
"deepseek": "你是一位搜索专家。请回答以下问题,并直接引用你参考的来源URL和原文片段。问题:{query}\n请给出200字以内的回答,并在回答末尾列出你引用的来源URL(每个来源一行)。",
|
||||
"chatgpt": "请搜索以下主题,返回相关信息和来源:{query}",
|
||||
"perplexity": "{query}",
|
||||
}
|
||||
|
||||
# 已知 AI 搜索 UA 特征(用于模拟查询)
|
||||
AI_USER_AGENTS = {
|
||||
"deepseek": "Mozilla/5.0 (compatible; DeepSeekBot/2.0; +https://deepseek.com/robot)",
|
||||
"chatgpt": "Mozilla/5.0 (compatible; ChatGPT-User/1.0; +https://openai.com)",
|
||||
"perplexity": "Mozilla/5.0 (compatible; PerplexityBot/1.0; +https://perplexity.ai)",
|
||||
}
|
||||
|
||||
|
||||
def get_published_articles() -> List[Dict]:
|
||||
"""从 DB 获取所有已发布的文章"""
|
||||
try:
|
||||
from app.database import SessionLocal
|
||||
from app.models import Article, Topic
|
||||
db = SessionLocal()
|
||||
try:
|
||||
results = db.query(Article, Topic).join(Topic, Article.topic_id == Topic.id).all()
|
||||
articles = []
|
||||
for article, topic in results:
|
||||
articles.append({
|
||||
"id": article.id,
|
||||
"topic_id": article.topic_id,
|
||||
"platform": article.platform,
|
||||
"title": article.title or topic.title,
|
||||
"content": (article.content or "")[:500],
|
||||
"html_content": (article.html_content or "")[:2000],
|
||||
"status": article.status,
|
||||
"topic_title": topic.title,
|
||||
"field": topic.field or "",
|
||||
})
|
||||
return articles
|
||||
finally:
|
||||
db.close()
|
||||
except Exception as e:
|
||||
logger.warning(f"无法读取文章列表: {e}")
|
||||
return []
|
||||
|
||||
|
||||
def _build_geo_queries(article: Dict) -> List[str]:
|
||||
"""为 GEO 追踪生成查询词"""
|
||||
queries = []
|
||||
title = article.get("title", "") or article.get("topic_title", "")
|
||||
field = article.get("field", "")
|
||||
|
||||
if title:
|
||||
# 用完整标题作为核心查询
|
||||
queries.append(title[:60])
|
||||
|
||||
# 提取关键短句
|
||||
parts = re.split(r'[::,,。.!!??]', title)
|
||||
for p in parts[:3]:
|
||||
p = p.strip()
|
||||
if 6 <= len(p) <= 30:
|
||||
queries.append(p)
|
||||
|
||||
if field:
|
||||
queries.append(f"{field} {title[:20]}" if title else field[:30])
|
||||
|
||||
return list(set(q for q in queries if len(q) >= 6))[:3]
|
||||
|
||||
|
||||
def _call_llm(prompt: str, max_tokens: int = 1000) -> str:
|
||||
"""封装的 LLM 调用,用于模拟 AI 搜索查询"""
|
||||
try:
|
||||
from app.core.nvidia_client import call_llm
|
||||
return call_llm(prompt, temperature=0.3, max_tokens=max_tokens)
|
||||
except Exception as e:
|
||||
logger.warning(f"LLM 调用失败: {e}")
|
||||
return ""
|
||||
|
||||
|
||||
def _check_deepseek_citation(query: str, article: Dict) -> Optional[Dict]:
|
||||
"""通过 DeepSeek 模型查询文章是否被引用"""
|
||||
title = article.get("title", "") or article.get("topic_title", "")
|
||||
prompt = AI_SEARCH_PROMPTS["deepseek"].format(query=query)
|
||||
response = _call_llm(prompt, max_tokens=1500)
|
||||
|
||||
if not response:
|
||||
return None
|
||||
|
||||
cited = False
|
||||
snippet = ""
|
||||
source_url = ""
|
||||
|
||||
# 检查响应中是否包含我们的域名或文章标题
|
||||
title_parts = [p for p in re.split(r'[::,,\s]', title) if len(p) >= 4]
|
||||
for part in title_parts[:3]:
|
||||
if part in response:
|
||||
cited = True
|
||||
# 提取包含引用的上下文
|
||||
idx = response.find(part)
|
||||
start = max(0, idx - 50)
|
||||
end = min(len(response), idx + len(part) + 100)
|
||||
snippet = response[start:end].strip()
|
||||
break
|
||||
|
||||
# 检查是否提及了域名
|
||||
from rank_tracker import DOMAIN
|
||||
if DOMAIN in response:
|
||||
cited = True
|
||||
if not snippet:
|
||||
idx = response.find(DOMAIN)
|
||||
start = max(0, idx - 80)
|
||||
end = min(len(response), idx + 200)
|
||||
snippet = response[start:end].strip()
|
||||
source_url = DOMAIN
|
||||
|
||||
if cited:
|
||||
return {
|
||||
"ai_cited": True,
|
||||
"ai_source": "deepseek",
|
||||
"ai_search_engine": "DeepSeek Chat",
|
||||
"citation_snippet": snippet[:300],
|
||||
"citation_url": source_url or f"https://{DOMAIN}",
|
||||
"geo_score": 80 if source_url else 60,
|
||||
}
|
||||
return {
|
||||
"ai_cited": False,
|
||||
"ai_source": "deepseek",
|
||||
"ai_search_engine": "DeepSeek Chat",
|
||||
"citation_snippet": None,
|
||||
"citation_url": None,
|
||||
"geo_score": 30,
|
||||
}
|
||||
|
||||
|
||||
def _check_chatgpt_citation(query: str, article: Dict) -> Optional[Dict]:
|
||||
"""通过 ChatGPT/GPT 模型查询文章是否被引用"""
|
||||
title = article.get("title", "") or article.get("topic_title", "")
|
||||
prompt = f"请搜索以下信息:{query}\n\n搜索完成后,列出你参考的每个来源。"
|
||||
response = _call_llm(prompt, max_tokens=1200)
|
||||
|
||||
if not response:
|
||||
return None
|
||||
|
||||
from rank_tracker import DOMAIN
|
||||
cited = DOMAIN in response
|
||||
|
||||
title_parts = [p for p in re.split(r'[::,,\s]', title) if len(p) >= 4]
|
||||
for part in title_parts[:3]:
|
||||
if part in response:
|
||||
cited = True
|
||||
break
|
||||
|
||||
snippet = ""
|
||||
if cited:
|
||||
for part in title_parts[:3]:
|
||||
if part in response:
|
||||
idx = response.find(part)
|
||||
start = max(0, idx - 60)
|
||||
end = min(len(response), idx + len(part) + 120)
|
||||
snippet = response[start:end].strip()
|
||||
break
|
||||
if not snippet and DOMAIN in response:
|
||||
idx = response.find(DOMAIN)
|
||||
start = max(0, idx - 60)
|
||||
end = min(len(response), idx + 120)
|
||||
snippet = response[start:end].strip()
|
||||
|
||||
return {
|
||||
"ai_cited": cited,
|
||||
"ai_source": "chatgpt",
|
||||
"ai_search_engine": "ChatGPT / GPT",
|
||||
"citation_snippet": snippet[:300] if snippet else None,
|
||||
"citation_url": f"https://{DOMAIN}" if cited else None,
|
||||
"geo_score": 75 if cited and snippet else 25,
|
||||
}
|
||||
|
||||
|
||||
def _check_perplexity_citation(query: str, article: Dict) -> Optional[Dict]:
|
||||
"""通过 Perplexity 风格查询(利用 LLM 模拟)"""
|
||||
title = article.get("title", "") or article.get("topic_title", "")
|
||||
prompt = f"请搜索 {query} 的最新信息和观点,并列出所有参考来源。"
|
||||
response = _call_llm(prompt, max_tokens=1200)
|
||||
|
||||
if not response:
|
||||
return None
|
||||
|
||||
from rank_tracker import DOMAIN
|
||||
cited = DOMAIN in response
|
||||
title_parts = [p for p in re.split(r'[::,,\s]', title) if len(p) >= 4]
|
||||
for part in title_parts[:3]:
|
||||
if part in response:
|
||||
cited = True
|
||||
break
|
||||
|
||||
snippet = ""
|
||||
if cited:
|
||||
for part in title_parts[:3]:
|
||||
if part in response:
|
||||
idx = response.find(part)
|
||||
start = max(0, idx - 60)
|
||||
end = min(len(response), idx + len(part) + 120)
|
||||
snippet = response[start:end].strip()
|
||||
break
|
||||
if not snippet and DOMAIN in response:
|
||||
idx = response.find(DOMAIN)
|
||||
start = max(0, idx - 60)
|
||||
end = min(len(response), idx + 120)
|
||||
snippet = response[start:end].strip()
|
||||
|
||||
return {
|
||||
"ai_cited": cited,
|
||||
"ai_source": "perplexity",
|
||||
"ai_search_engine": "Perplexity AI",
|
||||
"citation_snippet": snippet[:300] if snippet else None,
|
||||
"citation_url": f"https://{DOMAIN}" if cited else None,
|
||||
"geo_score": 70 if cited and snippet else 20,
|
||||
}
|
||||
|
||||
|
||||
_AI_CHECKERS = {
|
||||
"deepseek": _check_deepseek_citation,
|
||||
"chatgpt": _check_chatgpt_citation,
|
||||
"perplexity": _check_perplexity_citation,
|
||||
}
|
||||
|
||||
|
||||
def check_ai_citations(article: Dict, keywords: List[str],
|
||||
ai_engines: Optional[List[str]] = None) -> List[Dict]:
|
||||
"""对一篇文章检查所有 AI 搜索引擎的引用情况"""
|
||||
if ai_engines is None:
|
||||
ai_engines = ["deepseek", "chatgpt", "perplexity"]
|
||||
|
||||
results = []
|
||||
for engine in ai_engines:
|
||||
checker = _AI_CHECKERS.get(engine)
|
||||
if not checker:
|
||||
continue
|
||||
|
||||
engine_result = None
|
||||
for keyword in keywords:
|
||||
result = checker(keyword, article)
|
||||
if result and result.get("ai_cited"):
|
||||
engine_result = result
|
||||
logger.info(f" [GEO/{engine}] '{keyword[:20]}...' ✅ 被引用")
|
||||
break
|
||||
elif result and engine_result is None:
|
||||
engine_result = result # 保留未引用的结果
|
||||
|
||||
if engine_result:
|
||||
engine_result["article_id"] = article.get("id", "")
|
||||
engine_result["topic_id"] = article.get("topic_id", "")
|
||||
engine_result["platform"] = article.get("platform", "")
|
||||
engine_result["keyword"] = keywords[0] if keywords else ""
|
||||
engine_result["search_engine"] = engine
|
||||
results.append(engine_result)
|
||||
|
||||
return results
|
||||
|
||||
|
||||
def save_geo_results(results: List[Dict]):
|
||||
"""将 GEO 追踪结果写入 SearchRanking 表"""
|
||||
if not results:
|
||||
return
|
||||
try:
|
||||
from app.database import SessionLocal
|
||||
from app.models import SearchRanking
|
||||
db = SessionLocal()
|
||||
try:
|
||||
for r in results:
|
||||
record = SearchRanking(
|
||||
article_id=r.get("article_id"),
|
||||
topic_id=r.get("topic_id"),
|
||||
keyword=r.get("keyword"),
|
||||
platform=r.get("platform"),
|
||||
search_engine=r.get("search_engine", "geo"),
|
||||
position=None, # GEO 追踪不需要搜索排名位置
|
||||
ai_cited=r.get("ai_cited", False),
|
||||
ai_source=r.get("ai_source"),
|
||||
ai_search_engine=r.get("ai_search_engine"),
|
||||
citation_snippet=r.get("citation_snippet"),
|
||||
citation_url=r.get("citation_url"),
|
||||
geo_score=r.get("geo_score"),
|
||||
)
|
||||
db.add(record)
|
||||
db.commit()
|
||||
logger.info(f"已保存 {len(results)} 条 GEO 追踪结果")
|
||||
finally:
|
||||
db.close()
|
||||
except Exception as e:
|
||||
logger.warning(f"保存 GEO 结果失败: {e}")
|
||||
|
||||
|
||||
def calculate_geo_readiness(article: Dict) -> Dict:
|
||||
"""计算单篇文章的 GEO 就绪度评分"""
|
||||
title = article.get("title", "") or article.get("topic_title", "")
|
||||
content = article.get("content", "")
|
||||
html = article.get("html_content", "")
|
||||
|
||||
score = 0
|
||||
details = {}
|
||||
|
||||
# 1. 结构化数据检测 (30分)
|
||||
has_jsonld = "@context" in html and "schema.org" in html if html else False
|
||||
has_meta_desc = "<meta" in html and ("description" in html or "og:" in html) if html else False
|
||||
schema_score = (30 if has_jsonld else 0) + (10 if has_meta_desc else 0)
|
||||
score += min(schema_score, 30)
|
||||
details["has_structured_data"] = has_jsonld
|
||||
details["has_meta_tags"] = has_meta_desc
|
||||
|
||||
# 2. FAQ 格式检测 (20分)
|
||||
faq_patterns = [r'[Qq][::]\s*', r'[Aa][::]\s*', r'如何\s+\w+', r'什么[是么叫]\s+\w+', r'怎么\s+\w+']
|
||||
faq_count = sum(1 for p in faq_patterns if re.search(p, content + title))
|
||||
faq_score = min(faq_count * 10, 20)
|
||||
score += faq_score
|
||||
details["faq_score"] = faq_score
|
||||
details["has_faq_format"] = faq_count >= 2
|
||||
|
||||
# 3. HowTo 格式检测 (15分)
|
||||
howto_patterns = [r'步骤\s*\d', r'第一步|第二步|第三步', r'首先|其次|最后', r'Step\s*\d']
|
||||
howto_count = sum(1 for p in howto_patterns if re.search(p, content))
|
||||
howto_score = min(howto_count * 5, 15)
|
||||
score += howto_score
|
||||
details["has_howto_format"] = howto_count >= 2
|
||||
|
||||
# 4. 引用/数据源检测 (15分)
|
||||
cite_patterns = [r'\d{4}', r'研究表明', r'据统计', r'数据显示', r'报告指出', r'根据\w+']
|
||||
cite_count = sum(1 for p in cite_patterns if re.search(p, content))
|
||||
cite_score = min(cite_count * 3, 15)
|
||||
score += cite_score
|
||||
details["has_citations"] = cite_count >= 2
|
||||
|
||||
# 5. 标题结构 (10分)
|
||||
heading_count = content.count('\n## ') + content.count('\n### ') if content else 0
|
||||
heading_score = min(heading_count * 3, 10)
|
||||
score += heading_score
|
||||
|
||||
# 6. 内容长度 (10分)
|
||||
wc = len(content) if content else 0
|
||||
length_score = min(wc // 200, 10)
|
||||
score += length_score
|
||||
|
||||
details["word_count"] = wc
|
||||
|
||||
return {
|
||||
"article_id": article.get("id", ""),
|
||||
"topic_id": article.get("topic_id", ""),
|
||||
"platform": article.get("platform", ""),
|
||||
"total_score": min(score, 100),
|
||||
"has_schema": has_jsonld,
|
||||
"schema_types": json.dumps(["Article"]) if has_jsonld else "",
|
||||
"has_faq_format": details.get("has_faq_format", False),
|
||||
"has_howto_format": details.get("has_howto_format", False),
|
||||
"has_citations": details.get("has_citations", False),
|
||||
"word_count": wc,
|
||||
"readability_score": min(heading_score * 10, 100),
|
||||
"heading_structure_score": min(heading_count * 20, 100),
|
||||
}
|
||||
|
||||
|
||||
def save_geo_readiness(scores: List[Dict]):
|
||||
"""保存 GEO 就绪度评分到 GeoReadinessScore 表"""
|
||||
if not scores:
|
||||
return
|
||||
try:
|
||||
from app.database import SessionLocal
|
||||
from app.models import GeoReadinessScore
|
||||
db = SessionLocal()
|
||||
try:
|
||||
for s in scores:
|
||||
record = GeoReadinessScore(
|
||||
article_id=s.get("article_id"),
|
||||
topic_id=s.get("topic_id"),
|
||||
platform=s.get("platform"),
|
||||
total_score=s.get("total_score", 0),
|
||||
has_schema=s.get("has_schema", False),
|
||||
schema_types=s.get("schema_types"),
|
||||
has_faq_format=s.get("has_faq_format", False),
|
||||
has_howto_format=s.get("has_howto_format", False),
|
||||
has_citations=s.get("has_citations", False),
|
||||
word_count=s.get("word_count", 0),
|
||||
readability_score=s.get("readability_score", 0),
|
||||
heading_structure_score=s.get("heading_structure_score", 0),
|
||||
)
|
||||
db.add(record)
|
||||
db.commit()
|
||||
logger.info(f"已保存 {len(scores)} 条 GEO 就绪度评分")
|
||||
finally:
|
||||
db.close()
|
||||
except Exception as e:
|
||||
logger.warning(f"保存 GEO 评分失败: {e}")
|
||||
|
||||
|
||||
def run_all(ai_engines: Optional[List[str]] = None) -> Dict:
|
||||
"""对所有已发布文章执行 GEO 追踪"""
|
||||
articles = get_published_articles()
|
||||
if not articles:
|
||||
logger.warning("没有已发布的文章可追踪")
|
||||
return {"ok": True, "tracked": 0, "articles": 0}
|
||||
|
||||
all_geo_results = []
|
||||
all_scores = []
|
||||
|
||||
for article in articles:
|
||||
keywords = _build_geo_queries(article)
|
||||
if not keywords:
|
||||
continue
|
||||
|
||||
logger.info(f"[GEO] 追踪 [{article['id']}] {article.get('title','')[:30]}...")
|
||||
|
||||
# AI 搜索引用检测
|
||||
geo_results = check_ai_citations(article, keywords, ai_engines)
|
||||
all_geo_results.extend(geo_results)
|
||||
|
||||
# GEO 就绪度评分
|
||||
score = calculate_geo_readiness(article)
|
||||
all_scores.append(score)
|
||||
|
||||
cited_count = sum(1 for r in geo_results if r.get("ai_cited"))
|
||||
logger.info(f" → GEO评分: {score['total_score']}/100, "
|
||||
f"AI引用: {cited_count}/{len(geo_results)}")
|
||||
|
||||
save_geo_results(all_geo_results)
|
||||
save_geo_readiness(all_scores)
|
||||
|
||||
total_cited = sum(1 for r in all_geo_results if r.get("ai_cited"))
|
||||
avg_score = sum(s["total_score"] for s in all_scores) / len(all_scores) if all_scores else 0
|
||||
|
||||
logger.info(f"GEO 追踪完成: {len(articles)} 篇文章, "
|
||||
f"{len(all_geo_results)} 条AI引擎检测, "
|
||||
f"{total_cited} 条被引用, "
|
||||
f"平均GEO评分: {avg_score:.0f}/100")
|
||||
return {
|
||||
"ok": True,
|
||||
"articles_checked": len(articles),
|
||||
"ai_checks": len(all_geo_results),
|
||||
"total_cited": total_cited,
|
||||
"avg_geo_score": round(avg_score, 1),
|
||||
}
|
||||
|
||||
|
||||
def main():
|
||||
import argparse
|
||||
parser = argparse.ArgumentParser(description="GEO 追踪 — AI 搜索引用 + 就绪度评分")
|
||||
parser.add_argument('--engines', nargs='*', default=['deepseek', 'chatgpt', 'perplexity'],
|
||||
help='AI 搜索引擎列表')
|
||||
parser.add_argument('--readiness-only', action='store_true',
|
||||
help='仅计算 GEO 就绪度评分,不做 AI 引用检测')
|
||||
args = parser.parse_args()
|
||||
|
||||
engines = args.engines if not args.readiness_only else []
|
||||
result = run_all(engines)
|
||||
print(json.dumps(result, ensure_ascii=False))
|
||||
sys.exit(0 if result['ok'] else 1)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in New Issue
Block a user