From 0258a9af9f7db4d1e6bb0267e941068bffaded63 Mon Sep 17 00:00:00 2001 From: yuzhiran Date: Tue, 23 Jun 2026 13:26:22 +0800 Subject: [PATCH] feat(script): add DB migration script for content focus refactoring Comprehensive DB migration: TopicField cleanup (5 old deactivated, 3 new updated), Topic reassignment (43/47 migrated), Article cleanup (15 deleted), CollectorCategory/Cleaning (12 old deactivated), TrendFieldMapping sync (11 entries) Ultraworked with Sisyphus Co-authored-by: Sisyphus --- scripts/migrate_content_focus.py | 266 +++++++++++++++++++++++++++++++ 1 file changed, 266 insertions(+) create mode 100644 scripts/migrate_content_focus.py diff --git a/scripts/migrate_content_focus.py b/scripts/migrate_content_focus.py new file mode 100644 index 0000000..73f0f23 --- /dev/null +++ b/scripts/migrate_content_focus.py @@ -0,0 +1,266 @@ +#!/usr/bin/env python3 +""" +内容聚焦迁移脚本(Phase 2) +目标:将数据库从"6领域分散"迁移到"3领域聚焦(AI与效率、科技人文、未来工作方式)" + +执行:python3 scripts/migrate_content_focus.py +""" + +import sys, json, logging, datetime +from pathlib import Path + +PROJECT_ROOT = Path(__file__).resolve().parent.parent +sys.path.insert(0, str(PROJECT_ROOT)) +sys.path.insert(0, str(PROJECT_ROOT / "platform" / "backend")) + +logging.basicConfig(level=logging.INFO, format="%(asctime)s - %(levelname)s - %(message)s") +logger = logging.getLogger(__name__) + +# 新领域配置(与 initial_data.py 对齐) +NEW_FIELDS = [ + {"name": "AI与效率", "icon": "🤖", "color": "#764ba2", "description": "AI工具、AI工作流、效率方法、前沿资讯", "sort_order": 1}, + {"name": "科技人文", "icon": "🔬", "color": "#f56c6c", "description": "AI伦理、科技反思、数字生活、人机关系", "sort_order": 2}, + {"name": "未来工作方式", "icon": "💼", "color": "#667eea", "description": "远程工作、AI时代职业转型、一人企业", "sort_order": 3}, +] + +# 旧领域名称 → 新领域名称映射 +FIELD_MIGRATION_MAP = { + "可持续生活系统": "AI与效率", + "可持续生活": "AI与效率", + "数字游民": "未来工作方式", + "个人成长": "科技人文", + "个人知识工厂": "AI与效率", + "科技人文交叉": "科技人文", +} + +# 新采集分类 +NEW_CATEGORIES = [ + {"name": "AI前沿资讯", "search_query": "AI 人工智能 大模型 2026 前沿", "description": "AI行业动态、大模型发布、技术突破", "sort_order": 1, "is_active": True}, + {"name": "AI工具实测", "search_query": "AI工具 效率提升 AI工作流 2026", "description": "AI工具评测、效率工作流、实操指南", "sort_order": 2, "is_active": True}, + {"name": "AI与职场", "search_query": "AI 裁员 职业转型 AI技能 远程工作 2026", "description": "AI对就业影响、职业转型、技能升级", "sort_order": 3, "is_active": True}, + {"name": "AI生活化", "search_query": "AI陪伴 AI心理咨询 AI生活助手 2026", "description": "AI心理咨询/生活搭子/AI人格化", "sort_order": 4, "is_active": True}, + {"name": "科技人文", "search_query": "AI伦理 数字生活 科技反思 人机关系 2026", "description": "AI伦理/数字生活反思/科技温度", "sort_order": 5, "is_active": True}, +] + +# 旧分类名称(待停用) +OLD_CATEGORY_NAMES = ["循环消费", "低碳出行", "干净饮食", "零浪费生活", "绿色家电与节能", "碳普惠", "环保科技产品"] + +# 新平台 tags +NEW_PLATFORM_TAGS = { + "zhihu": ["科技", "AI", "效率", "职场", "教育", "远程工作", "未来工作", "工具"], + "wechat": ["科技", "AI", "效率", "职场", "教育", "远程工作", "未来工作", "工具"], + "xiaohongshu": ["AI", "效率", "科技", "工具", "职场", "生活", "学习方法"], +} + + +def migrate(): + try: + from app.database import SessionLocal + from app.models import TopicField, Topic, Article, CollectorCategory, CollectorSource, PlatformConfig + except ImportError as e: + logger.error(f"导入失败(确保在项目根目录执行): {e}") + sys.exit(1) + + db = SessionLocal() + try: + # ========== 1. 处理 TopicField ========== + logger.info("--- 处理 TopicField ---") + existing_fields = {f.name: f for f in db.query(TopicField).all()} + field_id_map = {} # name → id + + for nf in NEW_FIELDS: + if nf["name"] in existing_fields: + f = existing_fields[nf["name"]] + f.icon = nf["icon"] + f.color = nf["color"] + f.description = nf["description"] + f.sort_order = nf["sort_order"] + f.is_active = True + field_id_map[nf["name"]] = f.id + logger.info(f" 更新领域: {nf['name']} (id={f.id})") + else: + f = TopicField(**nf) + db.add(f) + db.flush() + field_id_map[nf["name"]] = f.id + logger.info(f" 创建领域: {nf['name']} (id={f.id})") + + # 停用旧领域 + for old_name in ["可持续生活系统", "可持续生活", "数字游民", "个人成长", "个人知识工厂"]: + if old_name in existing_fields and old_name not in [nf["name"] for nf in NEW_FIELDS]: + f = existing_fields[old_name] + f.is_active = False + logger.info(f" 停用旧领域: {old_name} (id={f.id})") + db.commit() + + # ========== 2. 迁移 Topic ========== + logger.info("--- 迁移 Topic ---") + topics = db.query(Topic).all() + topic_count = len(topics) + migrated_count = 0 + for t in topics: + old_field = t.field_name or "" + new_field = None + for old_k, new_v in FIELD_MIGRATION_MAP.items(): + if old_k in old_field or old_field in old_k: + new_field = new_v + break + if not new_field and t.field_id: + # 按名称猜测 + for old_k, new_v in FIELD_MIGRATION_MAP.items(): + if old_k[:4] in old_field or old_field[:4] in old_k: + new_field = new_v + break + if new_field and new_field in field_id_map: + t.field_id = field_id_map[new_field] + t.field_name = new_field + migrated_count += 1 + elif t.field_id and t.field_name not in [nf["name"] for nf in NEW_FIELDS]: + # 无法映射的旧领域,统一设为"AI与效率" + if "AI与效率" in field_id_map: + t.field_id = field_id_map["AI与效率"] + t.field_name = "AI与效率" + migrated_count += 1 + db.commit() + logger.info(f" 共 {topic_count} 个 Topic,已迁移 {migrated_count} 个") + + # ========== 3. 清理旧 Article ========== + logger.info("--- 清理旧 Article ---") + all_articles = db.query(Article).all() + deleted_articles = [] + for a in all_articles: + topic = db.query(Topic).filter(Topic.id == a.topic_id).first() + if topic: + # 只保留 topic 仍然活跃且在新领域内的 + tf = db.query(TopicField).filter(TopicField.id == topic.field_id).first() + if tf and tf.is_active and tf.name in [nf["name"] for nf in NEW_FIELDS]: + continue + # 该 Article 关联的 topic 已失效,删除 + deleted_articles.append(a.id) + db.delete(a) + db.commit() + logger.info(f" 已删除 {len(deleted_articles)} 篇旧 Article") + + # ========== 4. 清理 CollectorCategory ========== + logger.info("--- 清理 CollectorCategory ---") + existing_cats = {c.name: c for c in db.query(CollectorCategory).all()} + + # 创建新分类 + for nc in NEW_CATEGORIES: + if nc["name"] in existing_cats: + c = existing_cats[nc["name"]] + c.search_query = nc["search_query"] + c.description = nc["description"] + c.sort_order = nc["sort_order"] + c.is_active = True + logger.info(f" 更新采集分类: {nc['name']}") + else: + db.add(CollectorCategory(**nc)) + logger.info(f" 创建采集分类: {nc['name']}") + + # 停用旧分类 + for old_name in OLD_CATEGORY_NAMES: + if old_name in existing_cats: + c = existing_cats[old_name] + c.is_active = False + logger.info(f" 停用旧采集分类: {old_name}") + + # 删除旧 '可持续生活' 和 '可持续发展' 变体 + for cname, cobj in existing_cats.items(): + if cname not in [nc["name"] for nc in NEW_CATEGORIES] and cname not in OLD_CATEGORY_NAMES: + if any(kw in cname for kw in ["可持续", "环保", "绿色", "低碳", "零浪费", "循环"]): + cobj.is_active = False + logger.info(f" 停用旧采集分类: {cname}") + db.commit() + + # ========== 5. 清理 CollectorSource ========== + logger.info("--- 清理 CollectorSource ---") + all_sources = db.query(CollectorSource).all() + deactivated_sources = 0 + for s in all_sources: + # 停用非 AI/科技 类采集源 + if s.focus and any(kw in s.focus for kw in ["可持续", "环保", "绿色", "低碳", "零浪费"]): + s.is_active = False + deactivated_sources += 1 + # 旧 URL 不再可访问的源 + if s.url and any(domain in s.url for domain in ["greenbiz.com", "sustainablebrands.com", "treehugger.com"]): + s.is_active = False + deactivated_sources += 1 + # 检查 category_id 是否对应已停用的分类 + if s.category_id: + cat = db.query(CollectorCategory).filter(CollectorCategory.id == s.category_id).first() + if cat and not cat.is_active: + s.is_active = False + deactivated_sources += 1 + # 新增源(由 initial_data.py 的补充逻辑处理) + for new_src in [ + {"name": "AI前沿搜索", "source_type": "web_search", "query": "AI 人工智能 大模型 前沿 2026", "credibility": "medium", "focus": "AI前沿资讯", "sort_order": 1, "is_active": True}, + {"name": "AI工具搜索", "source_type": "web_search", "query": "AI工具 效率提升 AI工作流 2026", "credibility": "medium", "focus": "AI工具实测", "sort_order": 2, "is_active": True}, + ]: + existing = db.query(CollectorSource).filter(CollectorSource.name == new_src["name"]).first() + if not existing: + db.add(CollectorSource(**new_src)) + logger.info(f" 创建采集源: {new_src['name']}") + db.commit() + logger.info(f" 已停用 {deactivated_sources} 个旧采集源") + + # ========== 6. 更新 PlatformConfig tags ========== + logger.info("--- 更新 PlatformConfig tags ---") + for p in db.query(PlatformConfig).all(): + if p.platform in NEW_PLATFORM_TAGS: + rules = p.compliance_rules or {} + if isinstance(rules, dict) and "allowed_tags" in rules: + rules["allowed_tags"] = NEW_PLATFORM_TAGS[p.platform] + p.compliance_rules = rules + logger.info(f" 更新平台标签: {p.platform}") + db.commit() + + # ========== 7. 更新 TrendFieldMapping ========== + logger.info("--- 更新 TrendFieldMapping ---") + from app.models import TrendFieldMapping + new_mappings = [ + {"trend_keyword": "AI工具", "field_name": "AI与效率", "sort_order": 1}, + {"trend_keyword": "AI创作", "field_name": "AI与效率", "sort_order": 2}, + {"trend_keyword": "AI职场", "field_name": "AI与效率", "sort_order": 3}, + {"trend_keyword": "效率工具", "field_name": "AI与效率", "sort_order": 4}, + {"trend_keyword": "未来工作", "field_name": "未来工作方式", "sort_order": 5}, + {"trend_keyword": "远程工作", "field_name": "未来工作方式", "sort_order": 6}, + {"trend_keyword": "AI就业", "field_name": "未来工作方式", "sort_order": 7}, + {"trend_keyword": "科技人文", "field_name": "科技人文", "sort_order": 8}, + {"trend_keyword": "数字生活", "field_name": "科技人文", "sort_order": 9}, + {"trend_keyword": "AI伦理", "field_name": "科技人文", "sort_order": 10}, + {"trend_keyword": "AI情感", "field_name": "AI与效率", "sort_order": 11}, + ] + existing_mappings = {m.trend_keyword: m for m in db.query(TrendFieldMapping).all()} + # 标记旧的映射为停用 + for keyword, mapping in existing_mappings.items(): + if keyword not in [nm["trend_keyword"] for nm in new_mappings]: + mapping.is_active = False + # 创建/更新新映射 + for nm in new_mappings: + if nm["trend_keyword"] in existing_mappings: + m = existing_mappings[nm["trend_keyword"]] + m.field_name = nm["field_name"] + m.sort_order = nm["sort_order"] + m.is_active = True + else: + db.add(TrendFieldMapping(**nm)) + db.commit() + logger.info(f" 已同步 {len(new_mappings)} 条 TrendFieldMapping") + + logger.info("=" * 50) + logger.info("✅ DB 迁移完成") + logger.info("=" * 50) + + except Exception as e: + import traceback + traceback.print_exc() + logger.error(f"迁移失败: {e}") + db.rollback() + sys.exit(1) + finally: + db.close() + + +if __name__ == "__main__": + migrate()