数据新鲜度校验: 所有模块加上日期/时效检查
1. search_cache.json → 添加 _metadata.updated_at 时间戳 web_search.search_from_cache() 跳过超过36h的旧缓存 防止某查询失败时残留旧数据 2. metrics_feedback.json → collector 检查mtime,超过24h不采用 3. trends.json → 已有 date==TODAY 校验(load_trends) 4. collector_ai_advice → DB每日覆盖,时序安全 creator→topic状态位避免重复生成 optimizer→文章状态位避免重复审查
This commit is contained in:
@@ -398,6 +398,11 @@ class SustainabilityCollector:
|
|||||||
metrics_file = DATA_DIR / "metrics_feedback.json"
|
metrics_file = DATA_DIR / "metrics_feedback.json"
|
||||||
if metrics_file.exists():
|
if metrics_file.exists():
|
||||||
try:
|
try:
|
||||||
|
mtime = datetime.datetime.fromtimestamp(metrics_file.stat().st_mtime)
|
||||||
|
age = (datetime.datetime.now() - mtime).total_seconds()
|
||||||
|
if age > 86400: # 超过24h的数据不采用
|
||||||
|
logger.debug("metrics_feedback 过时(%.0fh),跳过", age / 3600)
|
||||||
|
else:
|
||||||
feedback = json.loads(metrics_file.read_text(encoding='utf-8'))
|
feedback = json.loads(metrics_file.read_text(encoding='utf-8'))
|
||||||
top_domains = feedback.get("top_domains", [])
|
top_domains = feedback.get("top_domains", [])
|
||||||
if top_domains:
|
if top_domains:
|
||||||
|
|||||||
@@ -108,10 +108,13 @@ def refresh_cache():
|
|||||||
"AI工具 人工智能 效率提升 2026",
|
"AI工具 人工智能 效率提升 2026",
|
||||||
]
|
]
|
||||||
|
|
||||||
cache = {}
|
cache = {"_metadata": {"updated_at": datetime.datetime.now().isoformat()}}
|
||||||
if SEARCH_CACHE_FILE.exists():
|
if SEARCH_CACHE_FILE.exists():
|
||||||
try:
|
try:
|
||||||
cache = json.loads(SEARCH_CACHE_FILE.read_text(encoding="utf-8"))
|
old = json.loads(SEARCH_CACHE_FILE.read_text(encoding="utf-8"))
|
||||||
|
for k, v in old.items():
|
||||||
|
if not k.startswith("_"):
|
||||||
|
cache.setdefault(k, v)
|
||||||
except Exception:
|
except Exception:
|
||||||
pass
|
pass
|
||||||
|
|
||||||
|
|||||||
+12
-2
@@ -7,7 +7,7 @@
|
|||||||
2. Bing Web Search API(设 BING_API_KEY)
|
2. Bing Web Search API(设 BING_API_KEY)
|
||||||
3. Bing 网页抓取(服务器环境常反爬拦截)
|
3. Bing 网页抓取(服务器环境常反爬拦截)
|
||||||
"""
|
"""
|
||||||
import json, logging, os, re
|
import json, logging, os, re, datetime
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import List, Dict, Optional
|
from typing import List, Dict, Optional
|
||||||
from urllib.parse import quote_plus
|
from urllib.parse import quote_plus
|
||||||
@@ -92,11 +92,21 @@ def search_scrape(query: str, max_results: int = 5) -> List[Dict]:
|
|||||||
|
|
||||||
|
|
||||||
def search_from_cache(query: str, max_results: int = 5) -> List[Dict]:
|
def search_from_cache(query: str, max_results: int = 5) -> List[Dict]:
|
||||||
"""从 opencode webfetch 预填充的缓存中读取"""
|
"""从 opencode webfetch 预填充的缓存中读取(跳过超过36小时的缓存)"""
|
||||||
if not SEARCH_CACHE_FILE.exists():
|
if not SEARCH_CACHE_FILE.exists():
|
||||||
return []
|
return []
|
||||||
try:
|
try:
|
||||||
cache = json.loads(SEARCH_CACHE_FILE.read_text(encoding="utf-8"))
|
cache = json.loads(SEARCH_CACHE_FILE.read_text(encoding="utf-8"))
|
||||||
|
meta = cache.get("_metadata", {})
|
||||||
|
updated = meta.get("updated_at", "")
|
||||||
|
if updated:
|
||||||
|
try:
|
||||||
|
age = (datetime.datetime.now() - datetime.datetime.fromisoformat(updated)).total_seconds()
|
||||||
|
if age > 129600: # 36h
|
||||||
|
logger.warning("搜索缓存过时(%dh),跳过", int(age // 3600))
|
||||||
|
return []
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
results = cache.get(query, [])
|
results = cache.get(query, [])
|
||||||
return results[:max_results]
|
return results[:max_results]
|
||||||
except Exception:
|
except Exception:
|
||||||
|
|||||||
Reference in New Issue
Block a user