Files
sbnews/docker/crawler/src/scheduler/task_runner.py
T
2026-06-12 22:27:18 +08:00

1538 lines
69 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""
任务运行器
- 多联赛批量采集
- 单次采集 / 定时采集
"""
import asyncio
import time
from typing import Dict, Any, List, Optional
from loguru import logger
from src.core.config import get_config
from src.core.error_collector import ErrorCollector
from src.engine.hybrid_engine import HybridEngine
from src.parser.data_parser import DongqiudiParser
from src.storage.database import Database
class TaskRunner:
"""采集任务运行器"""
def __init__(self, engine: Optional[HybridEngine] = None, db: Optional[Database] = None):
self._engine = engine or HybridEngine()
self._parser = DongqiudiParser()
self._db = db or Database()
self._cfg = get_config()
self._task_locks: Dict[str, asyncio.Lock] = {}
self._task_running: Dict[str, bool] = {}
async def initialize(self):
await self._engine.initialize()
await self._db.connect()
await self._db.ensure_tables()
async def close(self):
await self._engine.close()
await self._db.close()
def _try_lock_task(self, task_name: str) -> bool:
"""尝试获取任务锁(非阻塞),成功返回 True,已在运行返回 False"""
if self._task_running.get(task_name):
logger.warning(f"[LOCK] 任务 {task_name} 正在执行中,跳过本次触发")
return False
self._task_running[task_name] = True
return True
def _unlock_task(self, task_name: str):
"""释放任务锁"""
self._task_running[task_name] = False
# ── 积分榜采集 ──
async def crawl_standings(self, season_id: int, league_name: str = "") -> Dict[str, Any]:
ErrorCollector.get().set_task_name("standings")
start = time.time()
result = {"league": league_name, "season_id": season_id, "success": False}
try:
logger.info(f"开始采集积分榜: {league_name} (season_id={season_id})")
raw = await self._engine.get_standings(season_id)
standings = self._parser.parse_standings(raw, season_id)
if standings:
count = await self._db.upsert_standings(standings)
result.update({"success": True, "count": count, "candidate_count": len(standings), "saved_count": count, "elapsed": time.time() - start})
else:
result["error"] = "无数据"
await self._db.log_crawl(
"standings", league_name, result["success"],
time.time() - start, len(standings),
)
except Exception as e:
result["error"] = str(e)
logger.error(f"积分榜采集失败 [{league_name}]: {e}")
await self._db.log_crawl(
"standings", league_name, False,
time.time() - start, error=str(e),
)
return result
# ── 赛程采集 ──
async def crawl_schedule(self, season_id: int, league_name: str = "") -> Dict[str, Any]:
ErrorCollector.get().set_task_name("schedule")
start = time.time()
result = {"league": league_name, "season_id": season_id, "success": False}
try:
logger.info(f"开始采集赛程: {league_name} (season_id={season_id})")
raw = await self._engine.get_schedule(season_id)
matches = self._parser.parse_schedule(raw, season_id)
if matches:
count = await self._db.upsert_matches(matches, league_name)
result.update({"success": True, "count": count, "candidate_count": len(matches), "saved_count": count, "elapsed": time.time() - start})
else:
result["error"] = "无数据"
await self._db.log_crawl(
"schedule", league_name, result["success"],
time.time() - start, len(matches),
)
except Exception as e:
result["error"] = str(e)
logger.error(f"赛程采集失败 [{league_name}]: {e}")
await self._db.log_crawl(
"schedule", league_name, False,
time.time() - start, error=str(e),
)
return result
# ── 比赛菜单采集 ──
async def crawl_match_menu(self) -> Dict[str, Any]:
ErrorCollector.get().set_task_name("match_menu")
start = time.time()
try:
logger.info("采集比赛类型菜单")
raw = await self._engine.get_match_menu()
types = self._parser.parse_match_menu(raw)
if types:
count = await self._db.upsert_match_types(types)
return {"success": True, "count": count, "candidate_count": len(types), "saved_count": count, "elapsed": time.time() - start}
return {"success": False, "error": "无数据"}
except Exception as e:
logger.error(f"比赛菜单采集失败: {e}")
return {"success": False, "error": str(e)}
# ── 新闻资讯采集 ──
NEWS_TABS = {
1: "推荐", 56: "热门", 3: "英超", 5: "西甲", 4: "意甲",
6: "德甲", 57: "法甲", 232: "中超", 37: "NBA", 55: "CBA",
}
NEWS_TAB_CIDS = {
3: 10, 5: 12, 4: 11, 6: 13, 57: 14, 232: 15, 37: 17, 55: 18,
}
async def crawl_news(self, tab_ids: list = None) -> Dict[str, Any]:
ErrorCollector.get().set_task_name("news")
start = time.time()
result = {"task": "news", "success": False}
tabs = tab_ids or list(self.NEWS_TABS.keys())
try:
total_count = 0
for tab_id in tabs:
tab_name = self.NEWS_TABS.get(tab_id, str(tab_id))
logger.info(f"采集资讯频道: {tab_name} (tab_id={tab_id})")
raw = await self._engine.get_news(tab_id=tab_id, size=30)
articles = self._parser.parse_news(raw, tab_name=tab_name)
if articles:
cid = self.NEWS_TAB_CIDS.get(tab_id, 0)
count = await self._db.upsert_articles(articles, cid=cid)
total_count += count
logger.info(f" {tab_name}: {count} 条入库 (cid={cid})")
result.update({"success": True, "count": total_count, "candidate_count": total_count, "saved_count": total_count, "elapsed": time.time() - start})
await self._db.log_crawl(
"news", "dongqiudi", True,
time.time() - start, total_count,
)
except Exception as e:
result["error"] = str(e)
logger.error(f"新闻采集失败: {e}")
await self._db.log_crawl(
"news", "dongqiudi", False,
time.time() - start, error=str(e),
)
return result
# ── 联赛资讯采集(iPhone API,支持分页)──
LEAGUE_NEWS_TABS = {
56: {"name": "中超", "cid": 15, "pages": 3},
3: {"name": "英超", "cid": 10, "pages": 3},
5: {"name": "西甲", "cid": 12, "pages": 3},
6: {"name": "德甲", "cid": 13, "pages": 3},
12: {"name": "法甲", "cid": 14, "pages": 3},
4: {"name": "意甲", "cid": 11, "pages": 3},
}
async def crawl_league_news(self, tab_ids: list = None) -> Dict[str, Any]:
"""从懂球帝 iPhone tabs API 采集联赛资讯,写入 article 表并设置正确的 cid"""
ErrorCollector.get().set_task_name("league_news")
start = time.time()
result = {"task": "league_news", "success": False}
tabs = tab_ids or list(self.LEAGUE_NEWS_TABS.keys())
try:
total_count = 0
for tab_id in tabs:
cfg = self.LEAGUE_NEWS_TABS.get(tab_id, {"name": str(tab_id), "cid": 0, "pages": 1})
tab_name = cfg["name"]
cid = cfg["cid"]
max_pages = cfg["pages"]
after = 0
for page in range(1, max_pages + 1):
logger.info(f"[league_news] 采集 {tab_name} tab_id={tab_id} page={page}")
raw = await self._engine.get_league_news(tab_id=tab_id, page=page, after=after)
articles = self._parser.parse_news(raw, tab_name=tab_name)
if not articles:
logger.info(f" {tab_name} page={page}: 无数据,停止翻页")
break
count = await self._db.upsert_articles(articles, cid=cid)
total_count += count
logger.info(f" {tab_name} page={page}: {count} 条入库 (cid={cid})")
after = raw.get("min", 0)
if not after or not raw.get("next"):
break
result.update({"success": True, "count": total_count, "candidate_count": total_count, "saved_count": total_count, "elapsed": time.time() - start})
summary = f"✅ 联赛资讯采集成功: {total_count} 条入库, {time.time()-start:.1f}s"
logger.info(summary)
result["summary"] = summary
await self._db.log_crawl("league_news", "dongqiudi", True, time.time() - start, total_count)
except Exception as e:
result["error"] = str(e)
logger.error(f"联赛资讯采集失败: {e}")
await self._db.log_crawl("league_news", "dongqiudi", False, time.time() - start, error=str(e))
return result
# ── 视频列表采集 ──
VIDEO_TABS = {
233: "中超", 239: "英超", 234: "西甲", 235: "意甲",
236: "德甲", 237: "更多", 238: "闲情",
}
VIDEO_TAB_CIDS = {
233: 15, 239: 10, 234: 12, 235: 11, 236: 13,
}
async def crawl_videos(self, tab_ids: list = None) -> Dict[str, Any]:
ErrorCollector.get().set_task_name("video")
start = time.time()
result = {"task": "video", "success": False}
tabs = tab_ids or list(self.VIDEO_TABS.keys())
try:
total_count = 0
for tab_id in tabs:
tab_name = self.VIDEO_TABS.get(tab_id, str(tab_id))
logger.info(f"采集视频频道: {tab_name} (tab_id={tab_id})")
raw = await self._engine.get_video_list(tab_id=tab_id)
videos = self._parser.parse_video_list(raw, tab_name=tab_name)
if videos:
cid = self.VIDEO_TAB_CIDS.get(tab_id, 0)
count = await self._db.upsert_articles(videos, cid=cid)
total_count += count
logger.info(f" {tab_name}: {count} 条入库 (cid={cid})")
result.update({"success": True, "count": total_count, "candidate_count": total_count, "saved_count": total_count, "elapsed": time.time() - start})
await self._db.log_crawl(
"video", "dongqiudi", True,
time.time() - start, total_count,
)
except Exception as e:
result["error"] = str(e)
logger.error(f"视频采集失败: {e}")
await self._db.log_crawl(
"video", "dongqiudi", False,
time.time() - start, error=str(e),
)
return result
# ── 文章详情补全 ──
async def crawl_article_content(self, batch_size: int = 10) -> Dict[str, Any]:
"""从 m.dongqiudi.com 获取文章 HTML 内容,每次取 batch_size 篇,逐篇随机睡眠 1-3s"""
import random, asyncio
ErrorCollector.get().set_task_name("article_content")
start = time.time()
result = {"task": "article_content", "success": False}
total_ok = 0
total_fail = 0
try:
articles = await self._db.get_articles_without_content(batch_size)
if not articles:
result.update({"success": True, "count": 0, "candidate_count": 0, "saved_count": 0, "elapsed": time.time() - start, "msg": "无待补全文章"})
await self._db.log_crawl("article_content", "dongqiudi", True, time.time() - start, 0)
return result
logger.info(f"[article_content] 本次获取 {len(articles)} 篇待补全文章")
for art in articles:
aid = art["article_id"]
try:
html = await self._engine.get_article_html(aid)
detail = self._parser.parse_article_html(html)
if detail:
await self._db.update_article_content(aid, detail)
total_ok += 1
logger.info(f" 文章 {aid} 内容补全成功: {len(detail.get('content',''))} chars")
else:
await self._db.set_content_retry_max(aid)
total_fail += 1
logger.info(f" 文章 {aid} 页面无可提取内容,跳过")
except Exception as e:
await self._db.increment_content_retry(aid)
total_fail += 1
logger.warning(f" 文章 {aid} 抓取失败: {e} (retry+1)")
sleep_sec = random.uniform(1, 3)
await asyncio.sleep(sleep_sec)
result.update({"success": True, "count": total_ok, "candidate_count": len(articles), "saved_count": total_ok, "fail": total_fail, "elapsed": time.time() - start})
summary = f"✅ 文章内容补全: {total_ok} 成功, {total_fail} 失败, {time.time()-start:.1f}s"
logger.info(summary)
result["summary"] = summary
await self._db.log_crawl("article_content", "dongqiudi", True, time.time() - start, total_ok)
except Exception as e:
result["error"] = str(e)
logger.error(f"文章内容补全失败: {e}")
await self._db.log_crawl("article_content", "dongqiudi", False, time.time() - start, error=str(e))
return result
async def crawl_article_details(self, batch_size: int = 3) -> Dict[str, Any]:
import random, asyncio
ErrorCollector.get().set_task_name("content")
start = time.time()
result = {"task": "content", "success": False}
total_ok = 0
total_processed = 0
consecutive_fail_batches = 0
try:
while True:
articles = await self._db.get_articles_without_content(batch_size)
if not articles:
if total_processed == 0:
result.update({"success": True, "count": 0, "candidate_count": 0, "saved_count": 0, "elapsed": time.time() - start, "msg": "无待补全文章"})
else:
result.update({"success": True, "count": total_ok, "candidate_count": total_processed, "saved_count": total_ok, "total": total_processed, "elapsed": time.time() - start})
break
logger.info(f"补全文章详情: 本批 {len(articles)} 篇 (已完成 {total_ok}/{total_processed})")
batch_ok = 0
for art in articles:
aid = art["article_id"]
total_processed += 1
try:
raw = await self._engine.get_article_detail(aid)
detail = self._parser.parse_article_detail(raw)
if detail:
await self._db.update_article_content(aid, detail)
total_ok += 1
batch_ok += 1
logger.debug(f" 文章 {aid} 已补全: title={detail.get('title','')[:20]}, content={len(detail.get('content',''))} chars")
else:
await self._db.increment_content_retry(aid)
logger.warning(f" 文章 {aid} 未提取到正文 (retry+1)")
except Exception as e:
await self._db.increment_content_retry(aid)
logger.warning(f" 文章 {aid} 抓取失败: {e} (retry+1)")
if batch_ok > 0:
consecutive_fail_batches = 0
else:
consecutive_fail_batches += 1
if consecutive_fail_batches >= 3:
logger.warning(f"连续 {consecutive_fail_batches} 批全部失败,停止本轮")
result.update({"success": True, "count": total_ok, "candidate_count": total_processed, "saved_count": total_ok, "total": total_processed, "elapsed": time.time() - start, "msg": "连续失败过多,已停止"})
break
sleep_sec = random.uniform(1, 5)
logger.info(f" 本批完成({batch_ok}/{len(articles)}成功),随机睡眠 {sleep_sec:.1f}s")
await asyncio.sleep(sleep_sec)
await self._db.log_crawl(
"content", "dongqiudi", True,
time.time() - start, total_ok,
)
except Exception as e:
result["error"] = str(e)
logger.error(f"文章详情补全失败: {e}")
await self._db.log_crawl(
"content", "dongqiudi", False,
time.time() - start, error=str(e),
)
return result
# ── 实时比赛采集 ──
async def crawl_live(self) -> Dict[str, Any]:
ErrorCollector.get().set_task_name("live")
start = time.time()
result = {"task": "live", "success": False}
try:
logger.info("开始采集实时比赛数据")
now_ts = int(time.time())
pending = await self._db.get_live_matches(now_ts)
if not pending:
result.update({"success": True, "count": 0, "elapsed": time.time() - start, "msg": "无进行中比赛"})
return result
raw = await self._engine.get_live_matches()
live_data = raw.get("data", {}).get("matches", []) if isinstance(raw, dict) else []
live_map = {}
for m in live_data:
mid = m.get("match_id")
if mid:
live_map[int(mid)] = m
updated = 0
expired = 0
for match in pending:
mid = int(match["match_id"])
if mid in live_map:
lm = live_map[mid]
status_raw = str(lm.get("status", ""))
if "Played" in status_raw or "finished" in status_raw.lower():
status = 2
elif "Playing" in status_raw or "live" in status_raw.lower():
status = 1
else:
status = match["status"]
await self._db.update_match_live(
mid, status,
int(lm.get("fs_A") or 0),
int(lm.get("fs_B") or 0),
f"{lm.get('hts_A', '')}-{lm.get('hts_B', '')}",
str(lm.get("minute", "")),
)
updated += 1
else:
# API中找不到该比赛,通过文字直播判断是否已结束
if match["status"] == 1:
is_ended = await self._check_match_ended_by_live_text(mid)
if is_ended:
await self._db.update_match_live(
mid, 2,
int(match.get("home_score") or 0),
int(match.get("away_score") or 0),
str(match.get("half_score") or ""),
"FT",
)
expired += 1
else:
# 兜底:开赛超过6小时仍无法确认,强制标记已结束
elapsed_h = (now_ts - int(match["match_time"])) / 3600
if elapsed_h >= 6:
await self._db.update_match_live(
mid, 2,
int(match.get("home_score") or 0),
int(match.get("away_score") or 0),
str(match.get("half_score") or ""),
"FT",
)
expired += 1
logger.warning(f"兜底: match_id={mid} 开赛超6小时,强制标记已结束")
if expired:
logger.info(f"超时自动标记已结束: {expired} 场")
# 采集进行中比赛的文字直播
live_text_count = 0
live_ids = [int(m["match_id"]) for m in pending if m["status"] == 1]
for mid in live_ids:
try:
c = await self._crawl_match_live_text(mid)
live_text_count += c
except Exception as e:
logger.warning(f"[live_text] match_id={mid} 失败: {e}")
if live_text_count:
logger.info(f"文字直播采集: {live_text_count} 条, {len(live_ids)} 场")
result.update({"success": True, "count": updated, "pending": len(pending), "live_text": live_text_count, "elapsed": time.time() - start})
await self._db.log_crawl(
"live", "dongqiudi", True,
time.time() - start, updated,
)
except Exception as e:
result["error"] = str(e)
logger.error(f"实时比赛采集失败: {e}")
await self._db.log_crawl(
"live", "dongqiudi", False,
time.time() - start, error=str(e),
)
return result
# ── 文字直播采集(由 crawl_live 调用) ──
async def _check_match_ended_by_live_text(self, match_id: int) -> bool:
"""通过文字直播最新消息判断比赛是否已结束"""
try:
raw = await self._engine.get_live_text(match_id, num=5)
data = raw.get("data", {})
items = data.get("data", [])
end_keywords = ["比赛结束", "全场结束", "已结束"]
for item in items:
msg = str(item.get("message", ""))
if any(kw in msg for kw in end_keywords):
logger.info(f" [live_text] match_id={match_id} 检测到结束: {msg}")
return True
except Exception as e:
logger.warning(f" [live_text] match_id={match_id} 检查结束状态失败: {e}")
return False
async def _crawl_match_live_text(self, match_id: int) -> int:
"""采集单场比赛的全部文字直播(自动翻页)"""
total = 0
raw = await self._engine.get_live_text(match_id, num=200)
data = raw.get("data", {})
items = data.get("data", [])
if items:
count = await self._db.upsert_match_live_text(match_id, items)
total += count
logger.info(f" [live_text] match_id={match_id}: 首页 {count} 条")
# 向前翻页获取更早的消息
next_url = data.get("next")
page = 1
while next_url and page < 50:
try:
page_raw = await self._engine.get_live_text_page(next_url)
page_data = page_raw.get("data", {})
page_items = page_data.get("data", [])
if not page_items:
break
count = await self._db.upsert_match_live_text(match_id, page_items)
total += count
next_url = page_data.get("next")
page += 1
await asyncio.sleep(0.2)
except Exception as e:
logger.warning(f" [live_text] match_id={match_id} 翻页失败: {e}")
break
logger.info(f" [live_text] match_id={match_id}: 共 {total} 条")
return total
# ── 懂球帝赛程页面通用采集(SSR 页面提取 matchList ──
async def _crawl_page_matches(self, task_name: str, tab_id: int, league_name: str, sport_type: int = 1) -> Dict[str, Any]:
"""通用:从懂球帝移动端赛程页面采集数据(m.dongqiudi.com/match/{tab_id}"""
ErrorCollector.get().set_task_name(task_name)
start = time.time()
result = {"task": task_name, "success": False}
try:
logger.info(f"[{task_name}] 开始采集{league_name}赛程页面 tab_id={tab_id}")
match_list = await self._engine.fetch_match_list_from_page(tab_id)
if not match_list:
result.update({"success": True, "count": 0, "candidate_count": 0, "saved_count": 0, "elapsed": time.time() - start, "msg": "页面无比赛数据"})
await self._db.log_crawl(task_name, league_name, True, time.time() - start, 0)
return result
for m in match_list:
m["_round_name"] = m.get("round_name", "")
m["_gameweek"] = m.get("gameweek", 0)
stats = await self._db.upsert_match_data_batch(match_list, league_name, sport_type=sport_type)
result.update({"success": True, "count": stats["total"], "candidate_count": len(match_list), "saved_count": stats["total"], "inserted": stats["inserted"], "updated": stats["updated"], "unchanged": stats["unchanged"], "total": len(match_list), "elapsed": time.time() - start})
logger.info(f"[{task_name}] 采集完成: 新增 {stats['inserted']}, 更新 {stats['updated']}, 未变 {stats['unchanged']}, 共 {stats['total']}/{len(match_list)} 场, {time.time() - start:.1f}s")
await self._db.log_crawl(task_name, league_name, True, time.time() - start, stats["total"])
except Exception as e:
result["error"] = str(e)
logger.error(f"[{task_name}] 采集失败: {e}")
await self._db.log_crawl(task_name, league_name, False, time.time() - start, error=str(e))
return result
async def crawl_csl_matches(self) -> Dict[str, Any]:
return await self._crawl_page_matches("csl_match", tab_id=5, league_name="中超", sport_type=1)
async def crawl_nba_matches(self) -> Dict[str, Any]:
return await self._crawl_page_matches("nba_match", tab_id=68, league_name="NBA", sport_type=2)
async def crawl_cba_matches(self) -> Dict[str, Any]:
return await self._crawl_page_matches("cba_match", tab_id=67, league_name="CBA", sport_type=2)
async def crawl_epl_matches(self) -> Dict[str, Any]:
return await self._crawl_page_matches("epl_match", tab_id=6, league_name="英超", sport_type=1)
async def crawl_bundesliga_matches(self) -> Dict[str, Any]:
return await self._crawl_page_matches("bundesliga_match", tab_id=8, league_name="德甲", sport_type=1)
async def crawl_laliga_matches(self) -> Dict[str, Any]:
return await self._crawl_page_matches("laliga_match", tab_id=7, league_name="西甲", sport_type=1)
async def crawl_seriea_matches(self) -> Dict[str, Any]:
return await self._crawl_page_matches("seriea_match", tab_id=9, league_name="意甲", sport_type=1)
async def crawl_ligue1_matches(self) -> Dict[str, Any]:
return await self._crawl_page_matches("ligue1_match", tab_id=30, league_name="法甲", sport_type=1)
async def crawl_ucl_matches(self) -> Dict[str, Any]:
return await self._crawl_page_matches("ucl_match", tab_id=31, league_name="欧冠", sport_type=1)
async def crawl_uel_matches(self) -> Dict[str, Any]:
return await self._crawl_page_matches("uel_match", tab_id=34, league_name="欧联", sport_type=1)
async def crawl_tennis_matches(self) -> Dict[str, Any]:
return await self._crawl_page_matches("tennis_match", tab_id=107, league_name="网球", sport_type=3)
async def crawl_esports_matches(self) -> Dict[str, Any]:
return await self._crawl_page_matches("esports_match", tab_id=65, league_name="电竞", sport_type=4)
async def crawl_sports_matches(self) -> Dict[str, Any]:
return await self._crawl_page_matches("sports_match", tab_id=106, league_name="体坛", sport_type=5)
# ── 赛事比赛数据采集(从 la_league 的 sessionid 触发) ──
async def crawl_match_data(self) -> Dict[str, Any]:
ErrorCollector.get().set_task_name("match_data")
start = time.time()
result = {"task": "match_data", "success": False}
try:
leagues = await self._db.get_league_sessions()
if not leagues:
result.update({"success": True, "count": 0, "candidate_count": 0, "saved_count": 0, "elapsed": time.time() - start, "msg": "无配置sessionid的联赛"})
return result
logger.info(f"[match_data] 开始同步,共 {len(leagues)} 个联赛")
total_matches = 0
total_rounds = 0
for league in leagues:
try:
sid = league.get('sessionid')
api_url = league.get('api') or ''
if sid:
# 有 sessionid:通过赛程轮次方式采集
r_count, m_count = await self._crawl_league_match_data(league)
total_rounds += r_count
total_matches += m_count
elif api_url:
# 无 sessionid,有 api:通过 tab API 直接获取
m_count = await self._crawl_league_by_api(league)
total_matches += m_count
except Exception as e:
logger.error(f"[match_data] {league['label']} 同步失败: {e}")
result.update({
"success": True,
"count": total_matches,
"candidate_count": total_matches,
"saved_count": total_matches,
"rounds": total_rounds,
"elapsed": time.time() - start,
})
await self._db.log_crawl(
"match_data", "dongqiudi", True,
time.time() - start, total_matches,
)
except Exception as e:
result["error"] = str(e)
logger.error(f"[match_data] 采集失败: {e}")
await self._db.log_crawl(
"match_data", "dongqiudi", False,
time.time() - start, error=str(e),
)
return result
async def _crawl_league_by_api(self, league: dict) -> int:
"""通过 la_league.api (tab API) 直接获取比赛列表,返回 match_count"""
label = league['label']
api_url = league['api']
logger.info(f"[{label}] 通过 tab API 获取比赛...")
raw = await self._engine.get_league_tab_matches(api_url)
matches = raw.get("list", []) if isinstance(raw, dict) else []
if not matches:
logger.info(f"[{label}] tab API 无比赛数据")
return 0
# 过滤:只保留未结束的比赛(Fixture/Playing),已结束的跳过避免重复写入
pending = [m for m in matches if m.get("status") != "Played"]
if not pending:
logger.info(f"[{label}] 全部已结束,跳过")
return 0
skipped = len(matches) - len(pending)
if skipped:
logger.info(f"[{label}] 过滤 {skipped} 场已结束,剩余 {len(pending)} 场")
for m in pending:
m["_round_name"] = m.get("round_name", "")
m["_gameweek"] = m.get("gameweek", 0)
st = int(league.get('sport_type') or 1)
stats = await self._db.upsert_match_data_batch(pending, label, sport_type=st)
logger.info(f"[{label}] tab API 同步完成: 新增 {stats['inserted']}, 更新 {stats['updated']}, 未变 {stats['unchanged']}, 共 {stats['total']} 场")
return stats["total"]
async def _crawl_league_match_data(self, league: dict) -> tuple:
"""采集单个联赛的所有轮次比赛数据,返回 (round_count, match_count)"""
season_id = int(league['sessionid'])
league_id = int(league['id'])
label = league['label']
logger.info(f"[{label}] season_id={season_id} 获取赛程...")
raw = await self._engine.get_schedule(season_id)
content = raw.get("content", {})
rounds = content.get("rounds", [])
logger.info(f"[{label}] 共 {len(rounds)} 轮次")
# 保存轮次
round_rows = []
for r in rounds:
params = r.get("params", {})
round_rows.append({
"league_id": league_id,
"season_id": season_id,
"round_id": params.get("round_id", 0),
"round_name": r.get("name", ""),
"gameweek": params.get("gameweek", 0),
"url": r.get("url", ""),
})
r_count = await self._db.upsert_match_rounds(round_rows)
# 遍历每个轮次获取比赛(跳过已完结或未开始的轮次)
m_count = 0
skipped = 0
for r in rounds:
round_url = r.get("url", "")
if not round_url:
continue
round_name = r.get("name", "")
# 检查是否可跳过:该轮已有数据且全部 Played 或全部 Fixture
try:
if await self._db.check_round_skip(round_name, label):
skipped += 1
continue
except Exception:
pass
try:
round_raw = await self._engine.get_schedule_round(round_url)
matches = round_raw.get("content", {}).get("matches", [])
if not matches:
continue
params = r.get("params", {})
gameweek = params.get("gameweek", 0)
for m in matches:
m["_round_name"] = round_name
m["_gameweek"] = gameweek
st = int(league.get('sport_type') or 1)
stats = await self._db.upsert_match_data_batch(matches, label, sport_type=st)
m_count += stats["total"]
logger.info(f" [{label}] {round_name}: 新增 {stats['inserted']}, 更新 {stats['updated']}, 未变 {stats['unchanged']}, 共 {stats['total']} 场")
except Exception as e:
logger.warning(f" [{label}] 轮次 {r.get('name','')} 失败: {e}")
await asyncio.sleep(0.2)
if skipped:
logger.info(f"[{label}] 跳过 {skipped} 个已完结/未开始轮次")
logger.info(f"[{label}] 同步完成: {r_count} 轮, {m_count} 场比赛")
return r_count, m_count
# ── 超时比赛收尾(采集详情 + 标记已结束) ──
async def crawl_match_finish(self) -> Dict[str, Any]:
"""
分层兜底处理超时未结束比赛(每次处理1场):
从 m.dongqiudi.com 页面抓取SSR数据获取比赛详情
- 6~12h: 拉详情,有确认结束才标记,否则跳过
- 12~24h: 拉详情后无论如何标记结束
- 24h+: 拉详情后强制标记结束
"""
ErrorCollector.get().set_task_name("match_finish")
start = time.time()
result = {"task": "match_finish", "success": False}
try:
overdue = await self._db.get_overdue_matches(hours=6, limit=1)
if not overdue:
result.update({"success": True, "count": 0, "candidate_count": 0, "saved_count": 0, "elapsed": time.time() - start, "msg": "无超时未结束比赛"})
return result
now_ts = int(time.time())
finished = 0
detail_ok = 0
skipped = 0
for match in overdue:
mid = int(match["match_id"])
match_time = int(match.get("match_time") or 0)
elapsed_h = (now_ts - match_time) / 3600 if match_time else 999
home_score = int(match.get("home_score") or 0)
away_score = int(match.get("away_score") or 0)
half_score = str(match.get("half_score") or "")
got_detail = False
api_confirmed_ended = False
logger.info(
f" [match_finish] 处理 match_id={mid} "
f"{match.get('home_team','')}-{match.get('away_team','')} "
f"league={match.get('league_name','')} "
f"db_status={match.get('status')} db_score={home_score}-{away_score} "
f"elapsed={elapsed_h:.1f}h"
)
# 从懂球帝移动端页面抓取比赛详情
try:
header_data = await self._engine.fetch_match_detail_from_page(mid)
match_obj = header_data.get("match", {})
info_obj = header_data.get("info", {})
if match_obj:
got_detail = True
# 判断页面返回的状态是否已结束
page_status = str(match_obj.get("status", "")).lower()
if "played" in page_status or "finished" in page_status or "ended" in page_status:
api_confirmed_ended = True
logger.info(f" [match_finish] page_status='{match_obj.get('status','')}' confirmed_ended={api_confirmed_ended}")
# 更新比分: team_A.fs / team_B.fs
team_a = match_obj.get("team_A", {})
team_b = match_obj.get("team_B", {})
fs_a = team_a.get("fs") or team_a.get("score")
fs_b = team_b.get("fs") or team_b.get("score")
if fs_a:
home_score = int(fs_a)
if fs_b:
away_score = int(fs_b)
hts_a = team_a.get("hts", "")
hts_b = team_b.get("hts", "")
if hts_a or hts_b:
half_score = f"{hts_a}-{hts_b}"
# 保存事件
events = info_obj.get("events") or []
if events:
ec = await self._db.upsert_match_events(mid, events)
if ec:
detail_ok += 1
logger.info(f" [match_finish] match_id={mid} 保存 {ec} 条事件")
else:
logger.info(f" [match_finish] match_id={mid} 页面match数据为空")
except Exception as e:
logger.warning(f" [match_finish] match_id={mid} 获取详情失败: {type(e).__name__}: {e}")
# 决策:是否标记结束
should_finish = False
reason = ""
if elapsed_h >= 24:
should_finish = True
reason = "超过24h强制结束"
elif elapsed_h >= 12:
should_finish = True
reason = "超过12h强制结束"
else:
# 6~12h: 仅在API确认结束 或 拉不到详情时标记
if api_confirmed_ended:
should_finish = True
reason = "API确认已结束"
elif not got_detail:
should_finish = True
reason = "API无数据/请求失败"
else:
skipped += 1
reason = "API有数据但未结束(可能延期)"
logger.info(f" [match_finish] 决策: 跳过 reason={reason}")
if should_finish:
logger.info(f" [match_finish] 决策: 标记结束 reason={reason} score={home_score}-{away_score} half={half_score}")
await self._db.finish_match(mid, home_score, away_score, half_score)
finished += 1
logger.info(f" [match_finish] ✔ match_id={mid} 已写入数据库 status=2")
result.update({
"success": True,
"count": finished,
"candidate_count": len(overdue),
"saved_count": finished,
"detail_ok": detail_ok,
"skipped": skipped,
"elapsed": time.time() - start,
})
await self._db.log_crawl(
"match_finish", "dongqiudi", True,
time.time() - start, finished,
)
except Exception as e:
result["error"] = str(e)
logger.error(f"[match_finish] 失败: {e}")
await self._db.log_crawl(
"match_finish", "dongqiudi", False,
time.time() - start, error=str(e),
)
return result
# ── 实时比赛详情更新(通用,覆盖所有赛事) ──
_SPORT_CMP_TYPE = {1: "soccer", 2: "basketball", 3: "tennis", 4: "esports", 5: "sports"}
async def crawl_live_detail(self) -> Dict[str, Any]:
"""
从 la_match 表查询已到开赛时间且未结束的比赛,
通过 situation API 获取结构化 JSON 数据,更新比分、状态和事件。
"""
ErrorCollector.get().set_task_name("live_detail")
start = time.time()
result = {"task": "live_detail", "success": False}
try:
now_ts = int(time.time())
pending = await self._db.get_live_matches(now_ts)
if not pending:
result.update({"success": True, "count": 0, "candidate_count": 0, "saved_count": 0, "elapsed": time.time() - start, "msg": "无进行中比赛"})
return result
logger.info(f"[live_detail] 待更新比赛: {len(pending)} 场")
updated = 0
finished = 0
failed = 0
live_text_total = 0
for match in pending:
mid = int(match["match_id"])
try:
raw = await self._engine.get_match_situation(mid)
logger.debug(f"[live_detail] match_id={mid} raw keys={list(raw.keys())}")
match_obj = raw.get("match", {})
if not match_obj:
logger.warning(f"[live_detail] match_id={mid} situation 返回无 match 数据")
failed += 1
continue
info_obj_dbg = raw.get("info", {})
logger.info(f"[live_detail] match_id={mid} info keys={list(info_obj_dbg.keys()) if info_obj_dbg else []}, match keys={list(match_obj.keys())[:20]}")
# 阵容:只采集一次(通过独立 lineup API)
if not await self._db.has_match_lineup(mid):
try:
lu_raw = await self._engine.get_match_lineup(mid)
logger.debug(f"[live_detail] match_id={mid} lineup raw keys={list(lu_raw.keys())}")
lu_info = lu_raw.get("info", lu_raw)
if isinstance(lu_info, dict):
logger.debug(f"[live_detail] match_id={mid} lineup info keys={list(lu_info.keys())}")
lineup_list = lu_info.get("lineup", [])
sub_list = lu_info.get("sub", [])
logger.info(f"[live_detail] match_id={mid} lineup={len(lineup_list)} sub={len(sub_list)}")
if lineup_list:
flat_lineup = self._flatten_situation_lineup(lineup_list, sub_list)
if flat_lineup:
lc = await self._db.upsert_match_lineup(mid, flat_lineup)
if lc:
logger.info(f"[live_detail] match_id={mid} 保存阵容 {lc} 人")
except Exception as lu_e:
logger.warning(f"[live_detail] match_id={mid} 阵容采集失败: {lu_e}")
# 解析状态
page_status = str(match_obj.get("status", "")).lower()
if "played" in page_status or "finished" in page_status or "ended" in page_status:
new_status = 2
elif "playing" in page_status or "live" in page_status:
new_status = 1
else:
new_status = match["status"]
# 解析比分
team_a = match_obj.get("team_A", {})
team_b = match_obj.get("team_B", {})
home_score = int(team_a.get("fs") or team_a.get("score") or match.get("home_score") or 0)
away_score = int(team_b.get("fs") or team_b.get("score") or match.get("away_score") or 0)
# 半场比分
hts_a = str(team_a.get("hts", ""))
hts_b = str(team_b.get("hts", ""))
half_score = f"{hts_a}-{hts_b}" if (hts_a or hts_b) else str(match.get("half_score") or "")
# 比赛分钟 / 阶段
period = str(match_obj.get("period", ""))
minute = str(match_obj.get("minute", ""))
if new_status == 2:
minute = "FT"
elif period == "HT":
minute = "HT"
await self._db.update_match_live(
mid, new_status, home_score, away_score, half_score, minute,
)
updated += 1
if new_status == 2:
finished += 1
logger.info(
f"[live_detail] match_id={mid} "
f"{match.get('home_team','')}-{match.get('away_team','')} "
f"status={new_status} score={home_score}-{away_score} min={minute}"
)
# 保存事件(进行中和结束都采集)
info_obj = raw.get("info", {})
situation_events = info_obj.get("events", {})
if situation_events:
flat_events = self._flatten_situation_events(situation_events)
if flat_events:
ec = await self._db.upsert_match_events(mid, flat_events)
if ec:
logger.info(f"[live_detail] match_id={mid} 保存 {ec} 条事件")
# 提取技术统计并更新
tech_stats = info_obj.get("statistics") or info_obj.get("tech_stats") or info_obj.get("stats") or {}
if not tech_stats:
tech_stats = match_obj.get("statistics") or match_obj.get("tech_stats") or match_obj.get("stats") or {}
if tech_stats:
await self._update_match_tech_stats(mid, tech_stats)
# 进行中的比赛采集文字直播
if new_status == 1:
try:
lt_count = await self._crawl_match_live_text(mid)
if lt_count:
live_text_total += lt_count
except Exception as lt_e:
logger.warning(f"[live_detail] match_id={mid} 文字直播采集失败: {lt_e}")
except Exception as e:
logger.warning(f"[live_detail] match_id={mid} 失败: {type(e).__name__}: {e}")
failed += 1
await asyncio.sleep(0.5)
result.update({
"success": True,
"count": updated,
"candidate_count": len(pending),
"saved_count": updated,
"finished": finished,
"failed": failed,
"pending": len(pending),
"live_text": live_text_total,
"elapsed": time.time() - start,
})
logger.info(
f"[live_detail] 完成: 更新 {updated} 场, 结束 {finished} 场, "
f"失败 {failed} 场, 文字直播 {live_text_total} 条, "
f"共 {len(pending)} 场, {time.time() - start:.1f}s"
)
await self._db.log_crawl(
"live_detail", "dongqiudi", True,
time.time() - start, updated,
)
except Exception as e:
result["error"] = str(e)
logger.error(f"[live_detail] 任务失败: {e}")
await self._db.log_crawl(
"live_detail", "dongqiudi", False,
time.time() - start, error=str(e),
)
return result
# 进球图标关键字 → 事件类型映射
_EVENT_PIC_MAP = {
"rBUC6GcM116": "goal", # 进球图标
"ChMf8FxtCcu": "assist", # 助攻图标
}
async def _update_match_tech_stats(self, match_id: int, tech_stats: dict):
"""从 situation API 的 statistics 中提取技术统计,更新到 la_match。
statistics 格式: {"team_A": {...}, "team_B": {...}, "list": [{"type": "控球率", "team_A": {"value": 57}, "team_B": {"value": 43}}, ...]}
"""
try:
stat_list = tech_stats.get("list", []) if isinstance(tech_stats, dict) else []
if not stat_list:
logger.debug(f"[live_detail] match_id={match_id} statistics 无 list 字段")
return
import json
compact_list = []
for item in stat_list:
compact_list.append({
"type": item.get("type", ""),
"home": item.get("team_A", {}).get("value", ""),
"away": item.get("team_B", {}).get("value", ""),
})
updates = {"tech_stats": json.dumps(compact_list, ensure_ascii=False)}
await self._db.update_match_stats(match_id, updates)
logger.info(f"[live_detail] match_id={match_id} 更新 {len(compact_list)} 项技术统计")
except Exception as e:
logger.warning(f"[live_detail] match_id={match_id} 技术统计更新失败: {e}")
def _flatten_situation_events(self, events_dict: dict) -> list:
"""
将 situation API 的按分钟分组事件展开为 upsert_match_events 期望的扁平列表。
events_dict 格式: {"3": {"minute": "3", "teamAEvents": [...], "teamBEvents": [...]}, ...}
"""
flat = []
for minute_key, evt_group in events_dict.items():
minute = str(evt_group.get("minute", minute_key))
for side, team_events in [("1", evt_group.get("teamAEvents", [])), ("2", evt_group.get("teamBEvents", []))]:
for ev in team_events:
pic_url = str(ev.get("event_pic", ""))
event_type = "goal"
for pic_key, etype in self._EVENT_PIC_MAP.items():
if pic_key in pic_url:
event_type = etype
break
flat.append({
"minute": minute,
"type": event_type,
"team_side": int(side),
"player_name": ev.get("person", ""),
"description": "",
})
return flat
@staticmethod
def _flatten_situation_lineup(lineup: list, sub: list) -> list:
"""
将 situation API 的 lineup + sub 数据展平为 upsert_match_lineup 期望的扁平列表。
lineup 中每个元素含 team_A / team_B 对象(首发),sub 中同理(替补)。
"""
flat = []
for row in lineup:
for side_key, team_side in [("team_A", 1), ("team_B", 2)]:
p = row.get(side_key)
if not p or not isinstance(p, dict) or not p.get("person_id"):
continue
flat.append({
"team_side": team_side,
"is_starter": 1,
"person_id": p.get("person_id", 0),
"person_name": p.get("person", ""),
"person_logo": p.get("person_logo", ""),
"shirt_number": str(p.get("shirtnumber", "")),
"captain": int(p.get("captain", 0)),
"position": p.get("position", ""),
"position_x": p.get("position_x", "") or "",
"position_y": p.get("position_y", "") or "",
"formation_place": int(p.get("formation_place", 0)),
})
for row in sub:
for side_key, team_side in [("team_A", 1), ("team_B", 2)]:
p = row.get(side_key)
if not p or not isinstance(p, dict) or not p.get("person_id"):
continue
flat.append({
"team_side": team_side,
"is_starter": 0,
"person_id": p.get("person_id", 0),
"person_name": p.get("person", ""),
"person_logo": p.get("person_logo", ""),
"shirt_number": str(p.get("shirtnumber", "")),
"captain": int(p.get("captain", 0)),
"position": p.get("position", ""),
"position_x": p.get("position_x", "") or "",
"position_y": p.get("position_y", "") or "",
"formation_place": int(p.get("formation_place", 0)),
})
return flat
# ── 六合彩开奖采集 ──
LOTTERY_LOCK_FILE = "/tmp/lottery_crawler.lock"
LOTTERY_POLL_INTERVAL = 3 # 开奖中轮询间隔(秒)
LOTTERY_POLL_TIMEOUT = 600 # 单期最长轮询时间(秒)
async def crawl_lottery(self) -> Dict[str, Any]:
"""
定时任务入口(每分钟被 cron 触发):
1. 无待开奖记录 → 首次拉取当前期 + 创建下期占位 → 退出
2. 有待开奖记录但未到开奖时间 → 直接退出
3. 已到开奖时间 → 文件锁防并发 → 内部每3秒轮询直到获取完整号码+下期信息 → 退出
"""
import fcntl
from datetime import datetime as _dt
ErrorCollector.get().set_task_name("lottery")
start = time.time()
result = {"task": "lottery", "success": False}
sources = self._cfg.lottery.sources
if not sources:
result.update({"success": True, "count": 0, "msg": "无六合彩数据源配置"})
return result
now = _dt.now()
current_year = now.year
num_mapping = await self._db.get_lottery_number_mapping(current_year)
if not num_mapping:
logger.warning(f"[lottery] {current_year}年号码映射表为空")
src_map = {s.category_id: s for s in sources}
# 获取待开奖记录(status=0
pending_draws = await self._db.get_pending_lottery_draws()
# 无待开奖记录 → 首次拉取
if not pending_draws:
logger.info("[lottery] 无待开奖记录,首次拉取...")
count = await self._lottery_first_fetch(sources, num_mapping, current_year)
result.update({"success": True, "count": count, "elapsed": time.time() - start})
if count > 0:
await self._db.log_crawl("lottery", "lottery", True, time.time() - start, count)
return result
# 检查是否有到了开奖时间的记录
need_poll = []
for draw in pending_draws:
src = src_map.get(draw['category_id'])
if not src:
continue
draw_time = draw.get('draw_time')
if not draw_time:
logger.warning(f"[lottery] {src.name}{draw['period']}期 无开奖时间,跳过")
continue
if isinstance(draw_time, str):
draw_time = _dt.strptime(draw_time, '%Y-%m-%d %H:%M:%S')
if now < draw_time:
remain = (draw_time - now).total_seconds()
logger.info(f"[lottery] {src.name}{draw['period']}期 距开奖还有{remain:.0f}秒,跳过")
continue
need_poll.append((draw, src))
if not need_poll:
result.update({"success": True, "count": 0, "msg": "未到开奖时间"})
return result
# 有需要采集的 → 获取文件锁防并发
lock_fd = None
try:
lock_fd = open(self.LOTTERY_LOCK_FILE, 'w')
fcntl.flock(lock_fd, fcntl.LOCK_EX | fcntl.LOCK_NB)
except (IOError, OSError):
logger.info("[lottery] 另一个采集进程正在运行,退出")
result.update({"success": True, "count": 0, "msg": "已有采集进程运行中"})
if lock_fd:
lock_fd.close()
return result
try:
total_count = 0
for draw, src in need_poll:
count = await self._lottery_poll_one(src, draw['period'], num_mapping, current_year)
total_count += count
result.update({"success": True, "count": total_count, "elapsed": time.time() - start})
if total_count > 0:
await self._db.log_crawl("lottery", "lottery", True, time.time() - start, total_count)
except Exception as e:
logger.error(f"[lottery] 采集异常: {e}")
result.update({"success": False, "error": str(e)})
finally:
fcntl.flock(lock_fd, fcntl.LOCK_UN)
lock_fd.close()
return result
async def _lottery_poll_one(self, src, expected_period: str, num_mapping: dict, year: int) -> int:
"""对单个彩种轮询,每3秒请求一次API,直到获取完整号码+下期信息或超时"""
poll_start = time.time()
attempt = 0
period_num = expected_period[len(str(year)):]
logger.info(f"[lottery] {src.name}{expected_period}期 已到开奖时间,开始轮询...")
while (time.time() - poll_start) < self.LOTTERY_POLL_TIMEOUT:
attempt += 1
try:
raw_text = await self._engine.get_lottery_result(src.url)
if attempt <= 3 or attempt % 20 == 0:
logger.info(f"[lottery] {src.name}{attempt}次请求 响应: {raw_text[:150]}")
parsed = self._parse_lottery_result(raw_text)
if not parsed:
await asyncio.sleep(self.LOTTERY_POLL_INTERVAL)
continue
api_period = parsed['period'].strip()
if api_period.zfill(3) != period_num.zfill(3):
await asyncio.sleep(self.LOTTERY_POLL_INTERVAL)
continue
numbers = parsed['numbers']
if len(numbers) < 7:
await asyncio.sleep(self.LOTTERY_POLL_INTERVAL)
continue
has_next = parsed.get('next_period') and parsed.get('next_month') and parsed.get('next_day')
if not has_next:
await asyncio.sleep(self.LOTTERY_POLL_INTERVAL)
continue
# 完整数据,入库
count = await self._lottery_save_result(src, parsed, num_mapping, year)
elapsed = time.time() - poll_start
logger.info(f"[lottery] ✅ {src.name}{expected_period}期采集完成 "
f"(轮询{attempt}次, 耗时{elapsed:.1f}s)")
return count
except Exception as e:
logger.warning(f"[lottery] {src.name} 请求失败: {e}")
await asyncio.sleep(self.LOTTERY_POLL_INTERVAL)
logger.warning(f"[lottery] {src.name}{expected_period}期轮询超时({self.LOTTERY_POLL_TIMEOUT}s)")
return 0
async def _lottery_first_fetch(self, sources, num_mapping: dict, year: int) -> int:
"""首次拉取(无待开奖记录时),获取最新结果并创建下期占位"""
total_count = 0
for src in sources:
try:
logger.info(f"[lottery] 首次拉取 {src.name}...")
raw_text = await self._engine.get_lottery_result(src.url)
logger.info(f"[lottery] {src.name} 原始响应: {raw_text[:200]}")
parsed = self._parse_lottery_result(raw_text)
if not parsed:
logger.warning(f"[lottery] {src.name} 解析失败")
continue
period = str(year) + parsed['period'].zfill(3)
exists = await self._db.get_lottery_draw_exists(src.category_id, period)
if exists:
logger.info(f"[lottery] {src.name}{period}期已存在,跳过")
await self._lottery_ensure_next(src, parsed, year)
continue
count = await self._lottery_save_result(src, parsed, num_mapping, year)
total_count += count
except Exception as e:
logger.error(f"[lottery] {src.name} 首次拉取失败: {e}")
return total_count
async def _lottery_save_result(self, src, parsed: dict, num_mapping: dict, year: int) -> int:
"""保存解析后的开奖结果到数据库"""
from datetime import datetime as _dt
period = str(year) + parsed['period'].zfill(3)
numbers = parsed['numbers']
special = numbers[-1]
all_nums = numbers
color_code_map = {'红': 'red', '蓝': 'blue', '绿': 'green'}
zodiac_list, element_list, color_codes = [], [], []
for n in all_nums:
m = num_mapping.get(n, {})
zodiac_list.append(m.get('zodiac', ''))
element_list.append(m.get('element', ''))
c = m.get('color', '')
key = c[0] if c else ''
color_codes.append(color_code_map.get(key, ''))
draw_data = {
'category_id': src.category_id,
'period': period,
'draw_date': _dt.now().strftime('%Y-%m-%d'),
'draw_time': None,
'numbers': numbers[:-1],
'special_number': special,
'zodiac': zodiac_list,
'elements': element_list,
'color': color_codes,
'status': 1,
}
await self._db.upsert_lottery_draw(draw_data)
logger.info(f"[lottery] {src.name}{period}期入库: 号码={numbers}, 特码={special}")
# 插入下一期待开奖记录
await self._lottery_ensure_next(src, parsed, year)
return 1
async def _lottery_ensure_next(self, src, parsed: dict, year: int):
"""确保下一期待开奖记录存在"""
if not parsed.get('next_period'):
return
period_prefix = str(year)
next_period = period_prefix + parsed['next_period'].zfill(3)
next_draw_date = None
next_draw_time = None
if parsed.get('next_month') and parsed.get('next_day'):
next_draw_date = f"{year}-{parsed['next_month'].zfill(2)}-{parsed['next_day'].zfill(2)}"
if parsed.get('next_hour') and parsed.get('next_minute'):
if next_draw_date:
next_draw_time = f"{next_draw_date} {parsed['next_hour'].zfill(2)}:{parsed['next_minute'].zfill(2)}:00"
next_data = {
'category_id': src.category_id,
'period': next_period,
'draw_date': next_draw_date,
'draw_time': next_draw_time,
}
await self._db.upsert_lottery_next_draw(next_data)
logger.info(f"[lottery] {src.name} 下期{next_period} 日期={next_draw_date} 时间={next_draw_time}")
@staticmethod
def _parse_lottery_result(raw_text: str) -> Optional[Dict[str, Any]]:
"""
解析六合彩API响应
格式: {"k":"039,14,42,40,11,17,28,02,040,04,14,二,21时32分","t":3000}
k字段逗号分隔:
[0]=当期期数, [1-7]=开奖号码(最后一个是特码),
[8]=下期期数, [9]=下期月, [10]=下期日, [11]=星期, [12]=下期开奖时间
"""
import json as _json, re
try:
data = _json.loads(raw_text)
except _json.JSONDecodeError:
# 尝试从文本中提取JSON
match = re.search(r'\{.*\}', raw_text)
if match:
try:
data = _json.loads(match.group())
except _json.JSONDecodeError:
return None
else:
return None
k = data.get('k', '')
if not k:
return None
parts = k.split(',')
if len(parts) < 9:
return None
period = parts[0].strip()
numbers = []
for p in parts[1:8]:
p = p.strip()
try:
numbers.append(int(p))
except ValueError:
return None
result = {
'period': period,
'numbers': numbers,
}
# 下一期信息
if len(parts) >= 9:
result['next_period'] = parts[8].strip()
if len(parts) >= 10:
result['next_month'] = parts[9].strip()
if len(parts) >= 11:
result['next_day'] = parts[10].strip()
# 解析下期开奖时间 "21时32分" 或 "21:32" 等
if len(parts) >= 13:
time_str = parts[12].strip()
time_match = re.search(r'(\d+)\D+(\d+)', time_str)
if time_match:
result['next_hour'] = time_match.group(1)
result['next_minute'] = time_match.group(2)
return result
# ── 彩种开奖数据采集 (kai8.us API) ──
LOTTERY_DRAW_API = "https://data.kai8.us/api/lottery/peroids/KAI166"
async def crawl_lottery_draw(self, force: bool = False) -> Dict[str, Any]:
"""请求一次 kai8 API,把返回的所有彩种 UPSERT 入库。
外部 cron(如 la_dev_crontab)按需要的频率反复调用本方法,
每次调用都会真实发起一次请求并写库,不做"是否到开奖时间"的预检。
force 参数保留以兼容 main.py 的 cmd_lottery_draw_force 调用方,
当前实现下与默认调用完全等价。
"""
ErrorCollector.get().set_task_name("lottery_draw")
start = time.time()
result = {"task": "lottery_draw", "success": False}
try:
raw = await self._engine.get_lottery_draw_data(self.LOTTERY_DRAW_API)
if not isinstance(raw, dict) or raw.get('status') != 0:
api_status = raw.get('status') if isinstance(raw, dict) else type(raw).__name__
err_msg = f"API返回异常: status={api_status}"
logger.warning(f"[lottery_draw] {err_msg}")
result["error"] = err_msg
await self._db.log_crawl(
"lottery_draw", "kai8", False,
time.time() - start, error=err_msg,
)
return result
items = raw.get('data', []) or []
count = await self._db.upsert_lottery_draw_results(items)
elapsed = time.time() - start
result.update({"success": True, "count": count, "candidate_count": len(items), "saved_count": count, "elapsed": elapsed})
logger.info(f"[lottery_draw] 采集完成: {count} 条, {elapsed:.2f}s")
if count > 0:
await self._db.log_crawl(
"lottery_draw", "kai8", True,
elapsed, count,
)
except Exception as e:
result["error"] = str(e)
logger.error(f"[lottery_draw] 采集失败: {e}")
await self._db.log_crawl(
"lottery_draw", "kai8", False,
time.time() - start, error=str(e),
)
return result
# ── 批量采集所有联赛 ──
async def crawl_all_leagues(
self, task_type: str = "standings",
) -> List[Dict[str, Any]]:
targets = sorted(
[t for t in self._cfg.dongqiudi.targets if t.active],
key=lambda t: t.priority,
)
if not targets:
logger.warning("无激活的采集目标")
return []
logger.info(f"开始批量采集 {len(targets)} 个联赛 ({task_type})")
results = []
for idx, target in enumerate(targets):
if task_type == "standings":
r = await self.crawl_standings(target.season_id, target.league)
elif task_type == "schedule":
r = await self.crawl_schedule(target.season_id, target.league)
else:
logger.warning(f"未知任务类型: {task_type}")
continue
results.append(r)
status = "✅" if r.get("success") else "❌"
logger.info(f" {status} {target.league}: {r.get('count', 0)} 条, {r.get('elapsed', 0):.1f}s")
success_count = sum(1 for r in results if r.get("success"))
logger.info(f"批量采集完成: {success_count}/{len(results)} 成功")
return results
async def __aenter__(self):
await self.initialize()
return self
async def __aexit__(self, *args):
await self.close()