""" 彩票资讯采集模块 - 从 Google News RSS 获取彩票新闻并写入 la_article 表 在 main.py 中通过 lottery_news 命令调用 数据源(RSS): 1. Google News 中文简体 - 彩票开奖(主) 重复判断: 以 source_url 为唯一键,重复则更新标题/摘要 """ import json import re import time import xml.etree.ElementTree as ET from datetime import datetime from email.utils import parsedate_to_datetime import pymysql import requests from article_publish_helper import has_publishable_article_content, resolve_crawled_article_is_show # ─── 配置 ─── ARTICLE_CID = 21 # la_article_cate 中"彩票资讯"的 id RSS_FEEDS = [ { "name": "彩票开奖(中文)", "url": "https://news.google.com/rss/search?q=%E5%BD%A9%E7%A5%A8+%E5%BC%80%E5%A5%96&hl=zh-CN&gl=CN&ceid=CN:zh-Hans", "author_default": "Google News", }, ] NS = { "content": "http://purl.org/rss/1.0/modules/content/", "media": "http://search.yahoo.com/mrss/", "dc": "http://purl.org/dc/elements/1.1/", } def strip_html(text: str) -> str: import html as html_mod text = re.sub(r"<[^>]+>", "", text or "") text = html_mod.unescape(text) text = html_mod.unescape(text) text = text.replace("\xa0", " ") return text.strip() def parse_rfc2822(date_str: str) -> tuple[int, str]: """解析 RSS pubDate (RFC 2822),返回 (unix时间戳, 'Y-m-d H:i:s'格式字符串)""" if not date_str: now = int(time.time()) return now, datetime.fromtimestamp(now).strftime("%Y-%m-%d %H:%M:%S") try: dt = parsedate_to_datetime(date_str) return int(dt.timestamp()), dt.strftime("%Y-%m-%d %H:%M:%S") except Exception: now = int(time.time()) return now, datetime.fromtimestamp(now).strftime("%Y-%m-%d %H:%M:%S") # ─── RSS 采集 ─── def fetch_rss(session: requests.Session, feed: dict) -> list[dict]: """从单个 RSS 源拉取新闻,返回标准化列表""" articles = [] url = feed["url"] name = feed["name"] try: resp = session.get(url, timeout=30) except Exception as e: print(f" [WARN] {name} 请求失败: {e}") return articles if resp.status_code != 200: print(f" [WARN] {name} HTTP {resp.status_code}") return articles try: root = ET.fromstring(resp.content) except ET.ParseError as e: print(f" [WARN] {name} XML 解析失败: {e}") return articles items = root.findall(".//item") for item in items: title = (item.findtext("title", "") or "").strip() link = (item.findtext("link", "") or "").strip() if not title or not link: continue pub_date = item.findtext("pubDate", "") desc_raw = item.findtext("description", "") or "" desc = strip_html(desc_raw)[:255] # Google News 标题格式: "标题 - 来源名",提取来源 author = feed.get("author_default", name) if " - " in title: parts = title.rsplit(" - ", 1) if len(parts) == 2: author = parts[1].strip() title = parts[0].strip() # 尝试提取图片 image = "" media_el = item.find("media:content", NS) if media_el is not None: image = media_el.get("url", "") if not image: media_thumb = item.find("media:thumbnail", NS) if media_thumb is not None: image = media_thumb.get("url", "") if not image: img_match = re.search(r']+src=["\']([^"\']+)["\']', desc_raw) if img_match: image = img_match.group(1) pub_ts, pub_fmt = parse_rfc2822(pub_date) articles.append({ "title": title[:255], "desc": desc, "content": strip_html(desc_raw), "source_url": link, "image": image, "author": author[:64], "published_ts": pub_ts, "published_at": pub_fmt, "category": "lottery", "ext": { "source": "google_news", "feed_name": name, "feed_url": url, }, }) return articles # ─── 数据库同步 ─── def sync_to_db(conn, articles: list[dict]) -> tuple[int, int, int]: """同步到 la_article 表,返回 (总数, 新增数, 更新数)""" if not articles: return 0, 0, 0 urls = [a["source_url"] for a in articles if a.get("source_url")] if not urls: return 0, 0, 0 with conn.cursor() as cur: placeholders = ",".join(["%s"] * len(urls)) cur.execute( f"SELECT id, source_url, content, is_show FROM la_article WHERE source_url IN ({placeholders}) AND cid = %s", urls + [ARTICLE_CID] ) existing = {row["source_url"]: row for row in cur.fetchall()} # 按标题去重:同分类下相同标题的文章不再插入 titles = [a["title"][:255] for a in articles if a.get("title")] existing_titles = set() if titles: with conn.cursor() as cur: t_placeholders = ",".join(["%s"] * len(titles)) cur.execute( f"SELECT title FROM la_article WHERE title IN ({t_placeholders}) AND cid = %s", titles + [ARTICLE_CID] ) existing_titles = {row["title"] for row in cur.fetchall()} new_count = 0 update_count = 0 now = int(time.time()) for article in articles: source_url = article.get("source_url", "") if not source_url or not article.get("title"): continue create_ts = article.get("published_ts") or now ext_json = json.dumps(article.get("ext", {}), ensure_ascii=False) if article.get("ext") else None if source_url in existing: existing_row = existing[source_url] next_is_show = existing_row["is_show"] if has_publishable_article_content(existing_row.get("content", "")) else 0 with conn.cursor() as cur: cur.execute( """UPDATE la_article SET title = %s, `desc` = %s, image = %s, author = %s, category = %s, ext = %s, is_show = %s, update_time = %s WHERE id = %s""", ( article["title"][:255], (article.get("desc") or "")[:255], article.get("image", ""), (article.get("author") or "")[:64], (article.get("category") or "")[:128], ext_json, next_is_show, now, existing_row["id"], ) ) update_count += 1 else: # 标题去重:相同标题跳过 if article["title"][:255] in existing_titles: continue with conn.cursor() as cur: cur.execute( """INSERT INTO la_article (cid, title, `desc`, abstract, image, author, published_at, category, source_url, content, ext, is_show, sort, click_virtual, click_actual, create_time, update_time) VALUES (%s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s)""", ( ARTICLE_CID, article["title"][:255], (article.get("desc") or "")[:255], (article.get("desc") or "")[:500], article.get("image", ""), (article.get("author") or "")[:64], (article.get("published_at") or "")[:64], (article.get("category") or "")[:128], source_url[:500], article.get("content") or "", ext_json, resolve_crawled_article_is_show(article.get("content") or ""), 0, 0, 0, create_ts, now, ) ) existing[source_url] = {"id": cur.lastrowid, "content": article.get("content") or "", "is_show": resolve_crawled_article_is_show(article.get("content") or "")} existing_titles.add(article["title"][:255]) new_count += 1 return len(articles), new_count, update_count # ─── 去重:两个 feed 可能有重叠 ─── def dedupe_articles(articles: list[dict]) -> list[dict]: """按 source_url 去重,保留先出现的""" seen = set() result = [] for a in articles: url = a.get("source_url", "") if url and url not in seen: seen.add(url) result.append(a) return result # ─── 主入口 ─── def run(db_config: dict): """主入口,db_config 从 dongqiudi-crawler 配置中传入""" session = requests.Session() session.headers.update({ "User-Agent": "Mozilla/5.0 (compatible; SportEra/1.0)", "Accept": "application/xml, text/xml, application/rss+xml, */*", }) all_articles = [] for feed in RSS_FEEDS: print(f"=== {feed['name']} 采集 ===") articles = fetch_rss(session, feed) print(f" {feed['name']} 返回 {len(articles)} 条") all_articles.extend(articles) all_articles = dedupe_articles(all_articles) print(f"去重后共 {len(all_articles)} 条") if not all_articles: print("所有数据源均无数据,跳过") return {"success": True, "candidate_count": 0, "saved_count": 0, "summary_text": "所有数据源均无数据,跳过"} conn = pymysql.connect(**db_config, cursorclass=pymysql.cursors.DictCursor) try: total, new_count, update_count = sync_to_db(conn, all_articles) conn.commit() summary = f"共 {total} 条, 新增 {new_count}, 更新 {update_count}" print(summary) return {"success": True, "candidate_count": total, "saved_count": new_count + update_count, "summary_text": summary} finally: conn.close()