#!/usr/bin/env python3
"""
网络搜索模块
三种模式(优先级从高到低):
1. 本地缓存(opencode webfetch 预填充)
2. Bing Web Search API(设 BING_API_KEY)
3. Bing 网页抓取(服务器环境常反爬拦截)
"""
import json, logging, os, re, datetime
from pathlib import Path
from typing import List, Dict, Optional
from urllib.parse import quote_plus
import requests
logger = logging.getLogger(__name__)
UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/126.0.0.0 Safari/537.36"
BING_API_KEY = os.getenv("BING_API_KEY", "")
SEARCH_CACHE_FILE = Path(__file__).parent.parent / "automation" / "data" / "search_cache.json"
def search_api(query: str, max_results: int = 5) -> List[Dict]:
"""Bing Web Search API(需要 BING_API_KEY 环境变量)"""
if not BING_API_KEY:
return []
try:
resp = requests.get(
"https://api.bing.microsoft.com/v7.0/search",
params={"q": query, "count": max_results, "mkt": "zh-CN"},
headers={"Ocp-Apim-Subscription-Key": BING_API_KEY},
timeout=10
)
if resp.status_code != 200:
logger.warning(f"Bing API 返回 {resp.status_code}")
return []
data = resp.json()
results = []
for item in data.get("webPages", {}).get("value", [])[:max_results]:
results.append({
"title": item.get("name", "")[:120],
"url": item.get("url", ""),
"content": item.get("snippet", "")[:300],
"source": "bing_api"
})
logger.info(f"Bing API '{query[:20]}': {len(results)} 条")
return results
except Exception as e:
logger.warning(f"Bing API 失败: {e}")
return []
def search_scrape(query: str, max_results: int = 5) -> List[Dict]:
"""从 cn.bing.com 抓取搜索结果(服务器环境常遭受反爬,返回空为正常)"""
try:
resp = requests.get(
"https://cn.bing.com/search",
params={"q": query, "setlang": "zh-cn", "cc": "cn", "count": "15"},
headers={"User-Agent": UA, "Accept-Language": "zh-CN,zh;q=0.9"},
timeout=15
)
if resp.status_code != 200:
return []
html = resp.text
results = []
seen = set()
for m in re.finditer(
r'
]*class="[^"]*b_algo[^"]*"[^>]*>.*?]*href="([^"]*)"[^>]*>(.*?)',
html, re.DOTALL
):
href, title_raw = m.group(1), m.group(2)
title = re.sub(r'<[^>]+>', '', title_raw).strip()
if not title or len(title) < 8 or href in seen:
continue
if re.search(r'(bing\.com|microsoft\.com|beian\.miit|beian\.mps)', href, re.I):
continue
if re.search(r'(zdic|hanyu|hancibao|chengyu|dict\.|iciba|bishun)', href, re.I):
continue
if len(title) <= 5:
continue
seen.add(href)
results.append({"title": title[:120], "url": href, "content": "", "source": "bing"})
if len(results) >= max_results:
break
return results
except Exception as e:
logger.warning(f"Bing 抓取失败: {e}")
return []
def search_from_cache(query: str, max_results: int = 5) -> List[Dict]:
"""从 opencode webfetch 预填充的缓存中读取(跳过超过36小时的缓存)"""
if not SEARCH_CACHE_FILE.exists():
return []
try:
cache = json.loads(SEARCH_CACHE_FILE.read_text(encoding="utf-8"))
meta = cache.get("_metadata", {})
updated = meta.get("updated_at", "")
if updated:
try:
age = (datetime.datetime.now() - datetime.datetime.fromisoformat(updated)).total_seconds()
if age > 129600: # 36h
logger.warning("搜索缓存过时(%dh),跳过", int(age // 3600))
return []
except Exception:
pass
results = cache.get(query, [])
return results[:max_results]
except Exception:
return []
def save_to_cache(query: str, results: List[Dict]):
"""保存搜索结果到缓存(供 opencode webfetch 填充时使用)"""
cache = {}
if SEARCH_CACHE_FILE.exists():
try:
cache = json.loads(SEARCH_CACHE_FILE.read_text(encoding="utf-8"))
except Exception:
pass
cache[query] = results
SEARCH_CACHE_FILE.parent.mkdir(parents=True, exist_ok=True)
SEARCH_CACHE_FILE.write_text(json.dumps(cache, ensure_ascii=False, indent=2), encoding="utf-8")
def search(query: str, max_results: int = 5) -> List[Dict]:
"""统一搜索接口:DB提供商 → 缓存 → API → 网页抓取"""
try:
from search_utils import search as db_search
results = db_search(query, max_results)
if results:
return results
except Exception:
pass
results = search_from_cache(query, max_results)
if results:
return results
if BING_API_KEY:
results = search_api(query, max_results)
if results:
return results
results = search_scrape(query, max_results)
if results:
return results
logger.info(f"搜索 '{query[:20]}' 无结果")
return []
def enrich_topic_research(topic: dict, max_results: int = 5) -> str:
"""对选题进行网络搜索,返回格式化的研究发现文本"""
title = topic.get('title', '')
field = topic.get('field', '')
queries = [title]
if field and field not in title:
queries.append(f"{field} {title[:40]}")
seen_urls = set()
results = []
for q in queries:
for r in search(q, max_results):
url = r.get('url', '')
if url and url not in seen_urls:
seen_urls.add(url)
results.append(r)
if not results:
return ""
lines = ["\n## 网络搜索参考", ""]
for r in results[:max_results]:
snippet = r.get('snippet', r.get('content', ''))
lines.append(f"- **{r.get('title', '无标题')}**")
lines.append(f" {snippet[:200]}")
if r.get('url'):
lines.append(f" [{r['url']}]")
lines.append("")
return "\n".join(lines)