""" 懂球帝文章资讯采集脚本 - 通过 /api/app/tabs/iphone/{tab_id}.json 接口获取文章列表 - 支持多Tab采集 + 分页翻页 - 按 article_id 去重,已有数据跳过 用法: python scripts/crawl_articles.py # 采集所有Tab第1页 python scripts/crawl_articles.py --pages 5 # 每个Tab采集5页 python scripts/crawl_articles.py --tabs 1,3,5 # 只采集指定Tab """ import asyncio import sys import time from pathlib import Path sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) from curl_cffi import requests as curl_requests from loguru import logger from src.core.config import get_config from src.storage.database import Database API_URL = "https://www.dongqiudi.com/api/app/tabs/iphone/{tab_id}.json" HEADERS = { "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 " "(KHTML, like Gecko) Chrome/124.0.0.0 Safari/537.36", "Referer": "https://www.dongqiudi.com/", "Accept": "application/json, text/plain, */*", "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8", } TAB_CONFIG = { 1: "头条", 3: "英超", 4: "意甲", 5: "西甲", 6: "德甲", } def fetch_tab_page(tab_id: int, page_url: str = "") -> dict: if not page_url: page_url = API_URL.format(tab_id=tab_id) resp = curl_requests.get( page_url, headers=HEADERS, impersonate="chrome", timeout=20, ) resp.raise_for_status() return resp.json() def collect_tab(tab_id: int, max_pages: int = 1) -> list: all_articles = [] page_url = "" for page_idx in range(1, max_pages + 1): label = TAB_CONFIG.get(tab_id, f"Tab{tab_id}") logger.info(f"[{label}] 第{page_idx}页 请求...") try: data = fetch_tab_page(tab_id, page_url) except Exception as e: logger.error(f"[{label}] 第{page_idx}页 请求失败: {e}") break articles = data.get("articles", []) logger.info(f"[{label}] 第{page_idx}页 返回 {len(articles)} 条文章") if not articles: break all_articles.extend(articles) next_url = data.get("next", "") if not next_url: logger.info(f"[{label}] 无下一页,停止") break page_url = next_url time.sleep(2) return all_articles async def run(tab_ids: list, max_pages: int = 1): cfg = get_config() db = Database() try: existing_ids = await db.get_existing_article_ids() cate_map = await db.get_article_cate_map() logger.info(f"数据库已有 {len(existing_ids)} 条文章, {len(cate_map)} 个分类") total_new = 0 total_skip = 0 for tab_id in tab_ids: label = TAB_CONFIG.get(tab_id, f"Tab{tab_id}") cid = cate_map.get(label, 0) if not cid: logger.warning(f"[{label}] 未找到对应分类,使用 cid=0") all_articles = collect_tab(tab_id, max_pages) unique = {} for a in all_articles: aid = int(a.get("id", 0)) if aid and aid not in unique: unique[aid] = a new_items = [a for aid, a in unique.items() if aid not in existing_ids] skip_count = len(unique) - len(new_items) logger.info(f"[{label}] 去重: {len(unique)} 条唯一, 新增: {len(new_items)}, 跳过: {skip_count}") if new_items: count = await db.upsert_articles(new_items, cid=cid) total_new += count existing_ids.update(int(a["id"]) for a in new_items) total_skip += skip_count time.sleep(1) logger.info(f"✅ 采集完成: 新增 {total_new} 条, 跳过 {total_skip} 条") await db.log_crawl( crawl_type="articles", target="dongqiudi_tabs_api", success=True, elapsed=0, record_count=total_new, ) except Exception as e: logger.error(f"采集失败: {e}", exc_info=True) try: await db.log_crawl( crawl_type="articles", target="dongqiudi_tabs_api", success=False, elapsed=0, error=str(e), ) except Exception: pass raise finally: await db.close() def main(): import argparse parser = argparse.ArgumentParser( description="懂球帝文章资讯采集", epilog="示例:\n" " python crawl_articles.py # 所有Tab各1页\n" " python crawl_articles.py --pages 5 # 每Tab采5页\n" " python crawl_articles.py --tabs 1,3,5 # 指定Tab\n", formatter_class=argparse.RawDescriptionHelpFormatter, ) parser.add_argument("--pages", type=int, default=1, help="每个Tab采集页数 (默认1)") parser.add_argument("--tabs", type=str, default="", help="Tab ID列表,逗号分隔 (默认全部)") args = parser.parse_args() if args.tabs: tab_ids = [int(t.strip()) for t in args.tabs.split(",") if t.strip()] else: tab_ids = list(TAB_CONFIG.keys()) tab_names = [TAB_CONFIG.get(t, f"Tab{t}") for t in tab_ids] logger.info("=" * 60) logger.info("懂球帝文章资讯采集") logger.info(f"Tabs: {', '.join(tab_names)}, 每Tab {args.pages} 页") logger.info("=" * 60) asyncio.run(run(tab_ids, args.pages)) if __name__ == "__main__": main()