#!/usr/bin/env python3 """ GEO 追踪模块 — AI 搜索引用追踪 + GEO 就绪度评分 功能: 1. 对已发布文章,检查是否被 AI 搜索引擎引用(DeepSeek/ChatGPT/Perplexity) 2. 记录引用片段、来源、时间 3. 计算每篇文章的 GEO 就绪度评分 4. 可作为定时任务每天运行 """ import json, logging, sys, re, datetime from pathlib import Path from typing import List, Dict, Optional PROJECT_ROOT = Path(__file__).parent.parent sys.path.insert(0, str(PROJECT_ROOT)) sys.path.insert(0, str(PROJECT_ROOT / 'platform' / 'backend')) logging.basicConfig(level=logging.INFO, format='%(asctime)s - %(levelname)s - %(message)s') logger = logging.getLogger(__name__) # AI 搜索引擎配置:查询提示词模板 AI_SEARCH_PROMPTS = { "deepseek": "你是一位搜索专家。请回答以下问题,并直接引用你参考的来源URL和原文片段。问题:{query}\n请给出200字以内的回答,并在回答末尾列出你引用的来源URL(每个来源一行)。", "chatgpt": "请搜索以下主题,返回相关信息和来源:{query}", "perplexity": "{query}", } # 已知 AI 搜索 UA 特征(用于模拟查询) AI_USER_AGENTS = { "deepseek": "Mozilla/5.0 (compatible; DeepSeekBot/2.0; +https://deepseek.com/robot)", "chatgpt": "Mozilla/5.0 (compatible; ChatGPT-User/1.0; +https://openai.com)", "perplexity": "Mozilla/5.0 (compatible; PerplexityBot/1.0; +https://perplexity.ai)", } def get_published_articles() -> List[Dict]: """从 DB 获取所有已发布的文章""" try: from app.database import SessionLocal from app.models import Article, Topic db = SessionLocal() try: results = db.query(Article, Topic).join(Topic, Article.topic_id == Topic.id).all() articles = [] for article, topic in results: articles.append({ "id": article.id, "topic_id": article.topic_id, "platform": article.platform, "title": article.title or topic.title, "content": (article.content or "")[:500], "html_content": (article.html_content or "")[:2000], "status": article.status, "topic_title": topic.title, "field": topic.field or "", }) return articles finally: db.close() except Exception as e: logger.warning(f"无法读取文章列表: {e}") return [] def _build_geo_queries(article: Dict) -> List[str]: """为 GEO 追踪生成查询词""" queries = [] title = article.get("title", "") or article.get("topic_title", "") field = article.get("field", "") if title: # 用完整标题作为核心查询 queries.append(title[:60]) # 提取关键短句 parts = re.split(r'[::,,。.!!??]', title) for p in parts[:3]: p = p.strip() if 6 <= len(p) <= 30: queries.append(p) if field: queries.append(f"{field} {title[:20]}" if title else field[:30]) return list(set(q for q in queries if len(q) >= 6))[:3] def _call_llm(prompt: str, max_tokens: int = 1000) -> str: """封装的 LLM 调用,用于模拟 AI 搜索查询""" try: from app.core.nvidia_client import call_llm return call_llm(prompt, temperature=0.3, max_tokens=max_tokens) except Exception as e: logger.warning(f"LLM 调用失败: {e}") return "" def _check_deepseek_citation(query: str, article: Dict) -> Optional[Dict]: """通过 DeepSeek 模型查询文章是否被引用""" title = article.get("title", "") or article.get("topic_title", "") prompt = AI_SEARCH_PROMPTS["deepseek"].format(query=query) response = _call_llm(prompt, max_tokens=1500) if not response: return None cited = False snippet = "" source_url = "" # 检查响应中是否包含我们的域名或文章标题 title_parts = [p for p in re.split(r'[::,,\s]', title) if len(p) >= 4] for part in title_parts[:3]: if part in response: cited = True # 提取包含引用的上下文 idx = response.find(part) start = max(0, idx - 50) end = min(len(response), idx + len(part) + 100) snippet = response[start:end].strip() break # 检查是否提及了域名 from rank_tracker import DOMAIN if DOMAIN in response: cited = True if not snippet: idx = response.find(DOMAIN) start = max(0, idx - 80) end = min(len(response), idx + 200) snippet = response[start:end].strip() source_url = DOMAIN if cited: return { "ai_cited": True, "ai_source": "deepseek", "ai_search_engine": "DeepSeek Chat", "citation_snippet": snippet[:300], "citation_url": source_url or f"https://{DOMAIN}", "geo_score": 80 if source_url else 60, } return { "ai_cited": False, "ai_source": "deepseek", "ai_search_engine": "DeepSeek Chat", "citation_snippet": None, "citation_url": None, "geo_score": 30, } def _check_chatgpt_citation(query: str, article: Dict) -> Optional[Dict]: """通过 ChatGPT/GPT 模型查询文章是否被引用""" title = article.get("title", "") or article.get("topic_title", "") prompt = f"请搜索以下信息:{query}\n\n搜索完成后,列出你参考的每个来源。" response = _call_llm(prompt, max_tokens=1200) if not response: return None from rank_tracker import DOMAIN cited = DOMAIN in response title_parts = [p for p in re.split(r'[::,,\s]', title) if len(p) >= 4] for part in title_parts[:3]: if part in response: cited = True break snippet = "" if cited: for part in title_parts[:3]: if part in response: idx = response.find(part) start = max(0, idx - 60) end = min(len(response), idx + len(part) + 120) snippet = response[start:end].strip() break if not snippet and DOMAIN in response: idx = response.find(DOMAIN) start = max(0, idx - 60) end = min(len(response), idx + 120) snippet = response[start:end].strip() return { "ai_cited": cited, "ai_source": "chatgpt", "ai_search_engine": "ChatGPT / GPT", "citation_snippet": snippet[:300] if snippet else None, "citation_url": f"https://{DOMAIN}" if cited else None, "geo_score": 75 if cited and snippet else 25, } def _check_perplexity_citation(query: str, article: Dict) -> Optional[Dict]: """通过 Perplexity 风格查询(利用 LLM 模拟)""" title = article.get("title", "") or article.get("topic_title", "") prompt = f"请搜索 {query} 的最新信息和观点,并列出所有参考来源。" response = _call_llm(prompt, max_tokens=1200) if not response: return None from rank_tracker import DOMAIN cited = DOMAIN in response title_parts = [p for p in re.split(r'[::,,\s]', title) if len(p) >= 4] for part in title_parts[:3]: if part in response: cited = True break snippet = "" if cited: for part in title_parts[:3]: if part in response: idx = response.find(part) start = max(0, idx - 60) end = min(len(response), idx + len(part) + 120) snippet = response[start:end].strip() break if not snippet and DOMAIN in response: idx = response.find(DOMAIN) start = max(0, idx - 60) end = min(len(response), idx + 120) snippet = response[start:end].strip() return { "ai_cited": cited, "ai_source": "perplexity", "ai_search_engine": "Perplexity AI", "citation_snippet": snippet[:300] if snippet else None, "citation_url": f"https://{DOMAIN}" if cited else None, "geo_score": 70 if cited and snippet else 20, } _AI_CHECKERS = { "deepseek": _check_deepseek_citation, "chatgpt": _check_chatgpt_citation, "perplexity": _check_perplexity_citation, } def check_ai_citations(article: Dict, keywords: List[str], ai_engines: Optional[List[str]] = None) -> List[Dict]: """对一篇文章检查所有 AI 搜索引擎的引用情况""" if ai_engines is None: ai_engines = ["deepseek", "chatgpt", "perplexity"] results = [] for engine in ai_engines: checker = _AI_CHECKERS.get(engine) if not checker: continue engine_result = None for keyword in keywords: result = checker(keyword, article) if result and result.get("ai_cited"): engine_result = result logger.info(f" [GEO/{engine}] '{keyword[:20]}...' ✅ 被引用") break elif result and engine_result is None: engine_result = result # 保留未引用的结果 if engine_result: engine_result["article_id"] = article.get("id", "") engine_result["topic_id"] = article.get("topic_id", "") engine_result["platform"] = article.get("platform", "") engine_result["keyword"] = keywords[0] if keywords else "" engine_result["search_engine"] = engine results.append(engine_result) return results def save_geo_results(results: List[Dict]): """将 GEO 追踪结果写入 SearchRanking 表""" if not results: return try: from app.database import SessionLocal from app.models import SearchRanking db = SessionLocal() try: for r in results: record = SearchRanking( article_id=r.get("article_id"), topic_id=r.get("topic_id"), keyword=r.get("keyword"), platform=r.get("platform"), search_engine=r.get("search_engine", "geo"), position=None, # GEO 追踪不需要搜索排名位置 ai_cited=r.get("ai_cited", False), ai_source=r.get("ai_source"), ai_search_engine=r.get("ai_search_engine"), citation_snippet=r.get("citation_snippet"), citation_url=r.get("citation_url"), geo_score=r.get("geo_score"), ) db.add(record) db.commit() logger.info(f"已保存 {len(results)} 条 GEO 追踪结果") finally: db.close() except Exception as e: logger.warning(f"保存 GEO 结果失败: {e}") def calculate_geo_readiness(article: Dict) -> Dict: """计算单篇文章的 GEO 就绪度评分""" title = article.get("title", "") or article.get("topic_title", "") content = article.get("content", "") html = article.get("html_content", "") score = 0 details = {} # 1. 结构化数据检测 (30分) has_jsonld = "@context" in html and "schema.org" in html if html else False has_meta_desc = "= 2 # 3. HowTo 格式检测 (15分) howto_patterns = [r'步骤\s*\d', r'第一步|第二步|第三步', r'首先|其次|最后', r'Step\s*\d'] howto_count = sum(1 for p in howto_patterns if re.search(p, content)) howto_score = min(howto_count * 5, 15) score += howto_score details["has_howto_format"] = howto_count >= 2 # 4. 引用/数据源检测 (15分) cite_patterns = [r'\d{4}', r'研究表明', r'据统计', r'数据显示', r'报告指出', r'根据\w+'] cite_count = sum(1 for p in cite_patterns if re.search(p, content)) cite_score = min(cite_count * 3, 15) score += cite_score details["has_citations"] = cite_count >= 2 # 5. 标题结构 (10分) heading_count = content.count('\n## ') + content.count('\n### ') if content else 0 heading_score = min(heading_count * 3, 10) score += heading_score # 6. 内容长度 (10分) wc = len(content) if content else 0 length_score = min(wc // 200, 10) score += length_score details["word_count"] = wc return { "article_id": article.get("id", ""), "topic_id": article.get("topic_id", ""), "platform": article.get("platform", ""), "total_score": min(score, 100), "has_schema": has_jsonld, "schema_types": json.dumps(["Article"]) if has_jsonld else "", "has_faq_format": details.get("has_faq_format", False), "has_howto_format": details.get("has_howto_format", False), "has_citations": details.get("has_citations", False), "word_count": wc, "readability_score": min(heading_score * 10, 100), "heading_structure_score": min(heading_count * 20, 100), } def save_geo_readiness(scores: List[Dict]): """保存 GEO 就绪度评分到 GeoReadinessScore 表""" if not scores: return try: from app.database import SessionLocal from app.models import GeoReadinessScore db = SessionLocal() try: for s in scores: record = GeoReadinessScore( article_id=s.get("article_id"), topic_id=s.get("topic_id"), platform=s.get("platform"), total_score=s.get("total_score", 0), has_schema=s.get("has_schema", False), schema_types=s.get("schema_types"), has_faq_format=s.get("has_faq_format", False), has_howto_format=s.get("has_howto_format", False), has_citations=s.get("has_citations", False), word_count=s.get("word_count", 0), readability_score=s.get("readability_score", 0), heading_structure_score=s.get("heading_structure_score", 0), ) db.add(record) db.commit() logger.info(f"已保存 {len(scores)} 条 GEO 就绪度评分") finally: db.close() except Exception as e: logger.warning(f"保存 GEO 评分失败: {e}") def run_all(ai_engines: Optional[List[str]] = None) -> Dict: """对所有已发布文章执行 GEO 追踪""" articles = get_published_articles() if not articles: logger.warning("没有已发布的文章可追踪") return {"ok": True, "tracked": 0, "articles": 0} all_geo_results = [] all_scores = [] for article in articles: keywords = _build_geo_queries(article) if not keywords: continue logger.info(f"[GEO] 追踪 [{article['id']}] {article.get('title','')[:30]}...") # AI 搜索引用检测 geo_results = check_ai_citations(article, keywords, ai_engines) all_geo_results.extend(geo_results) # GEO 就绪度评分 score = calculate_geo_readiness(article) all_scores.append(score) cited_count = sum(1 for r in geo_results if r.get("ai_cited")) logger.info(f" → GEO评分: {score['total_score']}/100, " f"AI引用: {cited_count}/{len(geo_results)}") save_geo_results(all_geo_results) save_geo_readiness(all_scores) total_cited = sum(1 for r in all_geo_results if r.get("ai_cited")) avg_score = sum(s["total_score"] for s in all_scores) / len(all_scores) if all_scores else 0 logger.info(f"GEO 追踪完成: {len(articles)} 篇文章, " f"{len(all_geo_results)} 条AI引擎检测, " f"{total_cited} 条被引用, " f"平均GEO评分: {avg_score:.0f}/100") return { "ok": True, "articles_checked": len(articles), "ai_checks": len(all_geo_results), "total_cited": total_cited, "avg_geo_score": round(avg_score, 1), } def main(): import argparse parser = argparse.ArgumentParser(description="GEO 追踪 — AI 搜索引用 + 就绪度评分") parser.add_argument('--engines', nargs='*', default=['deepseek', 'chatgpt', 'perplexity'], help='AI 搜索引擎列表') parser.add_argument('--readiness-only', action='store_true', help='仅计算 GEO 就绪度评分,不做 AI 引用检测') args = parser.parse_args() engines = args.engines if not args.readiness_only else [] result = run_all(engines) print(json.dumps(result, ensure_ascii=False)) sys.exit(0 if result['ok'] else 1) if __name__ == "__main__": main()