b2d043b231
- 用户管理:新增编辑弹窗(修改用户名/角色/密码),增加组织/创建时间/最后登录列 - 角色管理:新增 Role 模型 + CRUD API,admin.html 新增角色管理 tab - 菜单管理:新增 Menu 模型 + CRUD API,导航栏从 API 动态加载菜单项 - 个人中心:右上角下拉菜单(个人信息/修改密码/退出),新增修改密码 API - 种子数据:initial_data.py 自动创建默认角色(admin/editor)和默认菜单(7项) - 修复 research.py 缺少 enrich_topic_research 函数导致导入失败 - 修复 db_helper.py 中 generated_at 条件导致重创作不更新时间戳 - admin.html 操作列加宽防止按钮换行,平台配置增加删除按钮 - articles.html 预览弹窗加 lock-scroll=false 防止页面尺寸跳动
174 lines
6.0 KiB
Python
174 lines
6.0 KiB
Python
#!/usr/bin/env python3
|
||
"""
|
||
网络搜索模块
|
||
|
||
三种模式(优先级从高到低):
|
||
1. 本地缓存(opencode webfetch 预填充)
|
||
2. Bing Web Search API(设 BING_API_KEY)
|
||
3. Bing 网页抓取(服务器环境常反爬拦截)
|
||
"""
|
||
import json, logging, os, re, datetime
|
||
from pathlib import Path
|
||
from typing import List, Dict, Optional
|
||
from urllib.parse import quote_plus
|
||
import requests
|
||
|
||
logger = logging.getLogger(__name__)
|
||
|
||
UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/126.0.0.0 Safari/537.36"
|
||
BING_API_KEY = os.getenv("BING_API_KEY", "")
|
||
SEARCH_CACHE_FILE = Path(__file__).parent.parent / "automation" / "data" / "search_cache.json"
|
||
|
||
|
||
def search_api(query: str, max_results: int = 5) -> List[Dict]:
|
||
"""Bing Web Search API(需要 BING_API_KEY 环境变量)"""
|
||
if not BING_API_KEY:
|
||
return []
|
||
try:
|
||
resp = requests.get(
|
||
"https://api.bing.microsoft.com/v7.0/search",
|
||
params={"q": query, "count": max_results, "mkt": "zh-CN"},
|
||
headers={"Ocp-Apim-Subscription-Key": BING_API_KEY},
|
||
timeout=10
|
||
)
|
||
if resp.status_code != 200:
|
||
logger.warning(f"Bing API 返回 {resp.status_code}")
|
||
return []
|
||
data = resp.json()
|
||
results = []
|
||
for item in data.get("webPages", {}).get("value", [])[:max_results]:
|
||
results.append({
|
||
"title": item.get("name", "")[:120],
|
||
"url": item.get("url", ""),
|
||
"content": item.get("snippet", "")[:300],
|
||
"source": "bing_api"
|
||
})
|
||
logger.info(f"Bing API '{query[:20]}': {len(results)} 条")
|
||
return results
|
||
except Exception as e:
|
||
logger.warning(f"Bing API 失败: {e}")
|
||
return []
|
||
|
||
|
||
def search_scrape(query: str, max_results: int = 5) -> List[Dict]:
|
||
"""从 cn.bing.com 抓取搜索结果(服务器环境常遭受反爬,返回空为正常)"""
|
||
try:
|
||
resp = requests.get(
|
||
"https://cn.bing.com/search",
|
||
params={"q": query, "setlang": "zh-cn", "cc": "cn", "count": "15"},
|
||
headers={"User-Agent": UA, "Accept-Language": "zh-CN,zh;q=0.9"},
|
||
timeout=15
|
||
)
|
||
if resp.status_code != 200:
|
||
return []
|
||
|
||
html = resp.text
|
||
results = []
|
||
seen = set()
|
||
|
||
for m in re.finditer(
|
||
r'<li[^>]*class="[^"]*b_algo[^"]*"[^>]*>.*?<a[^>]*href="([^"]*)"[^>]*>(.*?)</a>',
|
||
html, re.DOTALL
|
||
):
|
||
href, title_raw = m.group(1), m.group(2)
|
||
title = re.sub(r'<[^>]+>', '', title_raw).strip()
|
||
if not title or len(title) < 8 or href in seen:
|
||
continue
|
||
if re.search(r'(bing\.com|microsoft\.com|beian\.miit|beian\.mps)', href, re.I):
|
||
continue
|
||
if re.search(r'(zdic|hanyu|hancibao|chengyu|dict\.|iciba|bishun)', href, re.I):
|
||
continue
|
||
if len(title) <= 5:
|
||
continue
|
||
seen.add(href)
|
||
results.append({"title": title[:120], "url": href, "content": "", "source": "bing"})
|
||
if len(results) >= max_results:
|
||
break
|
||
|
||
return results
|
||
except Exception as e:
|
||
logger.warning(f"Bing 抓取失败: {e}")
|
||
return []
|
||
|
||
|
||
def search_from_cache(query: str, max_results: int = 5) -> List[Dict]:
|
||
"""从 opencode webfetch 预填充的缓存中读取(跳过超过36小时的缓存)"""
|
||
if not SEARCH_CACHE_FILE.exists():
|
||
return []
|
||
try:
|
||
cache = json.loads(SEARCH_CACHE_FILE.read_text(encoding="utf-8"))
|
||
meta = cache.get("_metadata", {})
|
||
updated = meta.get("updated_at", "")
|
||
if updated:
|
||
try:
|
||
age = (datetime.datetime.now() - datetime.datetime.fromisoformat(updated)).total_seconds()
|
||
if age > 129600: # 36h
|
||
logger.warning("搜索缓存过时(%dh),跳过", int(age // 3600))
|
||
return []
|
||
except Exception:
|
||
pass
|
||
results = cache.get(query, [])
|
||
return results[:max_results]
|
||
except Exception:
|
||
return []
|
||
|
||
|
||
def save_to_cache(query: str, results: List[Dict]):
|
||
"""保存搜索结果到缓存(供 opencode webfetch 填充时使用)"""
|
||
cache = {}
|
||
if SEARCH_CACHE_FILE.exists():
|
||
try:
|
||
cache = json.loads(SEARCH_CACHE_FILE.read_text(encoding="utf-8"))
|
||
except Exception:
|
||
pass
|
||
cache[query] = results
|
||
SEARCH_CACHE_FILE.parent.mkdir(parents=True, exist_ok=True)
|
||
SEARCH_CACHE_FILE.write_text(json.dumps(cache, ensure_ascii=False, indent=2), encoding="utf-8")
|
||
|
||
|
||
def search(query: str, max_results: int = 5) -> List[Dict]:
|
||
"""统一搜索接口:缓存 → API → 网页抓取"""
|
||
results = search_from_cache(query, max_results)
|
||
if results:
|
||
return results
|
||
|
||
if BING_API_KEY:
|
||
results = search_api(query, max_results)
|
||
if results:
|
||
return results
|
||
|
||
results = search_scrape(query, max_results)
|
||
if results:
|
||
return results
|
||
|
||
logger.info(f"搜索 '{query[:20]}' 无结果")
|
||
return []
|
||
|
||
|
||
def enrich_topic_research(topic: dict, max_results: int = 5) -> str:
|
||
"""对选题进行网络搜索,返回格式化的研究发现文本"""
|
||
title = topic.get('title', '')
|
||
field = topic.get('field', '')
|
||
queries = [title]
|
||
if field and field not in title:
|
||
queries.append(f"{field} {title[:40]}")
|
||
seen_urls = set()
|
||
results = []
|
||
for q in queries:
|
||
for r in search(q, max_results):
|
||
url = r.get('url', '')
|
||
if url and url not in seen_urls:
|
||
seen_urls.add(url)
|
||
results.append(r)
|
||
if not results:
|
||
return ""
|
||
lines = ["\n## 网络搜索参考", ""]
|
||
for r in results[:max_results]:
|
||
snippet = r.get('snippet', r.get('content', ''))
|
||
lines.append(f"- **{r.get('title', '无标题')}**")
|
||
lines.append(f" {snippet[:200]}")
|
||
if r.get('url'):
|
||
lines.append(f" [{r['url']}]")
|
||
lines.append("")
|
||
return "\n".join(lines)
|