Files
yu-zhi-ran/scripts/web_search.py
T
Yuzhiran Dev 7d50878c7d 搜索缓存: 可追踪的search_cache.json, webfetch预填充26条
新机制:
- web_search.py按优先级: 缓存→Bing API→Bing抓取
- search_cache.json由opencode webfetch手动填充(不依赖搜索引擎)
- 搜索失效不影响采集器(LLM直接生成选题)
- .gitignore移除search_cache.json(应被版本追踪)
2026-05-21 08:29:04 +08:00

136 lines
4.6 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
"""
网络搜索模块
三种模式(优先级从高到低):
1. 本地缓存(opencode webfetch 预填充)
2. Bing Web Search API(设 BING_API_KEY
3. Bing 网页抓取(服务器环境常反爬拦截)
"""
import json, logging, os, re
from pathlib import Path
from typing import List, Dict, Optional
from urllib.parse import quote_plus
import requests
logger = logging.getLogger(__name__)
UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/126.0.0.0 Safari/537.36"
BING_API_KEY = os.getenv("BING_API_KEY", "")
SEARCH_CACHE_FILE = Path(__file__).parent.parent / "automation" / "data" / "search_cache.json"
def search_api(query: str, max_results: int = 5) -> List[Dict]:
"""Bing Web Search API(需要 BING_API_KEY 环境变量)"""
if not BING_API_KEY:
return []
try:
resp = requests.get(
"https://api.bing.microsoft.com/v7.0/search",
params={"q": query, "count": max_results, "mkt": "zh-CN"},
headers={"Ocp-Apim-Subscription-Key": BING_API_KEY},
timeout=10
)
if resp.status_code != 200:
logger.warning(f"Bing API 返回 {resp.status_code}")
return []
data = resp.json()
results = []
for item in data.get("webPages", {}).get("value", [])[:max_results]:
results.append({
"title": item.get("name", "")[:120],
"url": item.get("url", ""),
"content": item.get("snippet", "")[:300],
"source": "bing_api"
})
logger.info(f"Bing API '{query[:20]}': {len(results)}")
return results
except Exception as e:
logger.warning(f"Bing API 失败: {e}")
return []
def search_scrape(query: str, max_results: int = 5) -> List[Dict]:
"""从 cn.bing.com 抓取搜索结果(服务器环境常遭受反爬,返回空为正常)"""
try:
resp = requests.get(
"https://cn.bing.com/search",
params={"q": query, "setlang": "zh-cn", "cc": "cn", "count": "15"},
headers={"User-Agent": UA, "Accept-Language": "zh-CN,zh;q=0.9"},
timeout=15
)
if resp.status_code != 200:
return []
html = resp.text
results = []
seen = set()
for m in re.finditer(
r'<li[^>]*class="[^"]*b_algo[^"]*"[^>]*>.*?<a[^>]*href="([^"]*)"[^>]*>(.*?)</a>',
html, re.DOTALL
):
href, title_raw = m.group(1), m.group(2)
title = re.sub(r'<[^>]+>', '', title_raw).strip()
if not title or len(title) < 8 or href in seen:
continue
if re.search(r'(bing\.com|microsoft\.com|beian\.miit|beian\.mps)', href, re.I):
continue
if re.search(r'(zdic|hanyu|hancibao|chengyu|dict\.|iciba|bishun)', href, re.I):
continue
if len(title) <= 5:
continue
seen.add(href)
results.append({"title": title[:120], "url": href, "content": "", "source": "bing"})
if len(results) >= max_results:
break
return results
except Exception as e:
logger.warning(f"Bing 抓取失败: {e}")
return []
def search_from_cache(query: str, max_results: int = 5) -> List[Dict]:
"""从 opencode webfetch 预填充的缓存中读取"""
if not SEARCH_CACHE_FILE.exists():
return []
try:
cache = json.loads(SEARCH_CACHE_FILE.read_text(encoding="utf-8"))
results = cache.get(query, [])
return results[:max_results]
except Exception:
return []
def save_to_cache(query: str, results: List[Dict]):
"""保存搜索结果到缓存(供 opencode webfetch 填充时使用)"""
cache = {}
if SEARCH_CACHE_FILE.exists():
try:
cache = json.loads(SEARCH_CACHE_FILE.read_text(encoding="utf-8"))
except Exception:
pass
cache[query] = results
SEARCH_CACHE_FILE.parent.mkdir(parents=True, exist_ok=True)
SEARCH_CACHE_FILE.write_text(json.dumps(cache, ensure_ascii=False, indent=2), encoding="utf-8")
def search(query: str, max_results: int = 5) -> List[Dict]:
"""统一搜索接口:缓存 → API → 网页抓取"""
results = search_from_cache(query, max_results)
if results:
return results
if BING_API_KEY:
results = search_api(query, max_results)
if results:
return results
results = search_scrape(query, max_results)
if results:
return results
logger.info(f"搜索 '{query[:20]}' 无结果")
return []