Files
sbnews/docker/crawler/article_fetcher.py
T
2026-06-12 22:27:18 +08:00

464 lines
16 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""
文章正文抓取定时任务 - 独立运行
从 la_article 中查找指定分类下没有正文的文章,逐条抓取正文并回写
仅处理分类: 加密资讯(9), NBA(17), CBA(18), 彩票资讯(21)
每次最多抓取 BATCH_SIZE 条,每条之间随机间隔 DELAY 秒,避免被限制
在 main.py 中通过 article_content_fetch 命令调用
"""
import html as html_mod
import random
import re
import time
from concurrent.futures import ThreadPoolExecutor, as_completed
from urllib.parse import urlparse
import pymysql
import requests
from bs4 import BeautifulSoup
from article_publish_helper import has_publishable_article_content
from scrapling.fetchers import Fetcher
# ─── 配置 ───
TARGET_CIDS = [9, 17, 18, 21] # 需要抓正文的分类
BATCH_SIZE = 3 # 每个分类每次抓取条数
MAX_WORKERS = 4 # 并行抓取线程数
MAX_RETRY = 2 # 最大重试次数,达到后永久跳过
REQUESTS_FALLBACK_HEADERS = {
"User-Agent": (
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
"AppleWebKit/537.36 (KHTML, like Gecko) "
"Chrome/137.0.0.0 Safari/537.36"
),
"Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
"Referer": "https://www.google.com/",
}
def decode_google_news_url(url: str) -> str:
"""解码 Google News 跳转链接为真实文章 URL"""
if "news.google.com" not in url:
return url
try:
from googlenewsdecoder import new_decoderv1
decoded = new_decoderv1(url, interval=2)
if decoded and decoded.get("status") and decoded.get("decoded_url"):
return decoded["decoded_url"]
except Exception as e:
print(f" [WARN] Google News URL 解码失败: {e}")
return url
def _clean_media_url(url: str, base_url: str = "") -> str:
"""保留原始 URL,仅过滤 data: URI 并补全相对路径"""
if not url or url.startswith("data:"):
return ""
# 协议相对 URL//domain/path)直接补 https:
if url.startswith("//"):
return "https:" + url
# 站内相对路径(/path)补全域名
if url.startswith("/") and base_url:
parsed = urlparse(base_url)
return f"{parsed.scheme}://{parsed.netloc}{url}"
return url
def _extract_html_content(el, base_url: str = "") -> str:
"""从容器元素中提取正文 HTML
规则:
- 文字前的图片/视频不保留
- 文字后只保留第一张图片和第一个视频
- 文本去重
"""
items = [] # (type, value) type: text / img / video
seen_texts = set()
blocks = el.css("p, h2, h3, h4, blockquote, li, pre, img, video")
for block in blocks:
tag_name = getattr(block, 'tag', '') or ''
if tag_name == 'img':
src = block.attrib.get("src", "") or block.attrib.get("data-src", "")
cleaned = _clean_media_url(src, base_url)
if cleaned:
items.append(("img", cleaned))
continue
if tag_name == 'video':
src = block.attrib.get("src", "") or block.attrib.get("data-src", "")
poster = block.attrib.get("poster", "")
cleaned = _clean_media_url(src, base_url)
if cleaned:
items.append(("video", cleaned))
else:
cleaned = _clean_media_url(poster, base_url)
if cleaned:
items.append(("img", cleaned))
continue
# 文本段落
raw_parts = block.css("::text").getall()
text = "".join(t for t in raw_parts).strip()
if text and len(text) > 10 and text not in seen_texts:
seen_texts.add(text)
items.append(("text", html_mod.unescape(text)))
# 找到第一段文字的位置
first_text_idx = -1
for i, (kind, _) in enumerate(items):
if kind == "text":
first_text_idx = i
break
if first_text_idx < 0:
return ""
# 丢弃文字前的所有媒体
items = items[first_text_idx:]
# 找到最后一段文字的位置
last_text_idx = 0
for i, (kind, _) in enumerate(items):
if kind == "text":
last_text_idx = i
# 文字后的媒体只保留第一张图和第一个视频
result = items[:last_text_idx + 1]
got_img = False
got_video = False
for item in items[last_text_idx + 1:]:
if item[0] == "img" and not got_img:
result.append(item)
got_img = True
elif item[0] == "video" and not got_video:
result.append(item)
got_video = True
# 组装 HTML
parts = []
for kind, value in result:
if kind == "text":
parts.append(f"<p>{value}</p>")
elif kind == "img":
parts.append(f'<img src="{value}" />')
elif kind == "video":
parts.append(f'<video src="{value}" controls></video>')
return "".join(parts)
def _extract_html_content_bs4(el, base_url: str = "") -> str:
"""从 BeautifulSoup 容器中提取正文 HTML,规则与 scrapling 版本保持一致"""
items = []
seen_texts = set()
for block in el.select("p, h2, h3, h4, blockquote, li, pre, img, video"):
tag_name = (block.name or "").lower()
if tag_name == "img":
src = block.get("src", "") or block.get("data-src", "")
cleaned = _clean_media_url(src, base_url)
if cleaned:
items.append(("img", cleaned))
continue
if tag_name == "video":
src = block.get("src", "") or block.get("data-src", "")
poster = block.get("poster", "")
cleaned = _clean_media_url(src, base_url)
if cleaned:
items.append(("video", cleaned))
else:
cleaned = _clean_media_url(poster, base_url)
if cleaned:
items.append(("img", cleaned))
continue
text = block.get_text("", strip=True)
if text and len(text) > 10 and text not in seen_texts:
seen_texts.add(text)
items.append(("text", html_mod.unescape(text)))
first_text_idx = -1
for i, (kind, _) in enumerate(items):
if kind == "text":
first_text_idx = i
break
if first_text_idx < 0:
return ""
items = items[first_text_idx:]
last_text_idx = 0
for i, (kind, _) in enumerate(items):
if kind == "text":
last_text_idx = i
result = items[:last_text_idx + 1]
got_img = False
got_video = False
for item in items[last_text_idx + 1:]:
if item[0] == "img" and not got_img:
result.append(item)
got_img = True
elif item[0] == "video" and not got_video:
result.append(item)
got_video = True
parts = []
for kind, value in result:
if kind == "text":
parts.append(f"<p>{value}</p>")
elif kind == "img":
parts.append(f'<img src="{value}" />')
elif kind == "video":
parts.append(f'<video src="{value}" controls></video>')
return "".join(parts)
def _fetch_article_content_via_requests(url: str, timeout: int = 15) -> str:
"""requests 回退抓取,解决部分站点在 scrapling 下稳定返回 403 的问题"""
try:
resp = requests.get(url, timeout=timeout, headers=REQUESTS_FALLBACK_HEADERS)
except Exception as e:
print(f" [WARN] requests 回退抓取失败: {e}")
return ""
if resp.status_code != 200:
print(f" [WARN] requests 回退 HTTP {resp.status_code}")
return f"__HTTP_{resp.status_code}__"
soup = BeautifulSoup(resp.text, "html.parser")
result = _try_extract_bs4(soup, url)
if result:
print(" (requests 回退命中)")
return result
result = _fallback_p_tags_bs4(soup)
if result:
print(" (requests 回退p标签提取)")
return result
print(" [WARN] requests 回退提取未命中")
return ""
def fetch_article_content(url: str, timeout: int = 15) -> str:
"""抓取单篇文章正文,返回 HTML 格式(含 <p> 和 <img>"""
# 先解码 Google News 链接
real_url = decode_google_news_url(url)
if real_url != url:
print(f" 真实URL: {real_url[:100]}")
try:
page = Fetcher.get(
real_url,
follow_redirects=True,
timeout=timeout,
stealthy_headers=True,
impersonate="chrome",
)
if page.status != 200:
print(f" [WARN] HTTP {page.status}")
if page.status == 403:
fallback = _fetch_article_content_via_requests(real_url, timeout=timeout)
return fallback or "__HTTP_403__"
return f"__HTTP_{page.status}__"
result = _try_extract(page, real_url)
if result:
return result
# 回退:全页面 <p> 标签
result = _fallback_p_tags(page, real_url)
if result:
print(f" (回退p标签提取)")
return result
print(f" [WARN] 所有提取方式均未命中")
return ""
except Exception as e:
print(f" [WARN] 抓取失败 {real_url[:80]}: {e}")
fallback = _fetch_article_content_via_requests(real_url, timeout=timeout)
return fallback or ""
# 内容选择器列表
_CONTENT_SELECTORS = [
"article", # 标准 article 标签
"#article_content", # 新浪移动端
".article-content", # 通用
".art_content", # 新浪 sports.sina.cn
".article_body", # 新浪财经
"#artibody", # 新浪PC
"#mp-editor", # 搜狐号正文
"[data-role='original-title'] ~ div", # 搜狐文章体
".content-article", # 搜狐
".article-text", # 搜狐移动端
"#article-container", # 搜狐 m.sohu.com
".article-content-wrap", # 搜狐PC
".article", # 通用
"#content", # 通用
".post-content", # 通用博客
".main-content", # 通用
".TRS_Editor", # 网易/政府站
".text", # 腾讯新闻
".story-body", # USA Today / Gannett 系
".gnt_ar_b", # Gannett (tennessean 等)
"[class*='detail']", # 详情页通用
"[class*='news-body']", # 新闻正文
"[class*='rich-text']", # 富文本
"[class*='article']", # 模糊匹配含article的class
"[class*='content']", # 模糊匹配含content的class
]
def _try_extract(page, base_url: str) -> str:
"""用选择器列表尝试提取正文"""
for selector in _CONTENT_SELECTORS:
el = page.css(selector)
if not el:
continue
result = _extract_html_content(el, base_url=base_url)
text_len = len(re.sub(r"<[^>]+>", "", result))
if text_len > 50:
return result
return ""
def _try_extract_bs4(soup: BeautifulSoup, base_url: str) -> str:
"""BeautifulSoup 版本的正文提取"""
for selector in _CONTENT_SELECTORS:
try:
elements = soup.select(selector)
except Exception:
continue
for el in elements:
result = _extract_html_content_bs4(el, base_url=base_url)
text_len = len(re.sub(r"<[^>]+>", "", result))
if text_len > 50:
return result
return ""
def _fallback_p_tags(page, base_url: str) -> str:
"""回退:提取全页面所有 <p> 标签"""
blocks = page.css("p")
if not blocks:
return ""
seen = set()
parts = []
for block in blocks:
raw = block.css("::text").getall()
text = "".join(t for t in raw).strip()
if text and len(text) > 15 and text not in seen:
seen.add(text)
parts.append(f"<p>{html_mod.unescape(text)}</p>")
content = "".join(parts)
if len(re.sub(r"<[^>]+>", "", content)) > 50:
return content
return ""
def _fallback_p_tags_bs4(soup: BeautifulSoup) -> str:
"""BeautifulSoup 版本的全页 <p> 标签回退提取"""
blocks = soup.select("p")
if not blocks:
return ""
seen = set()
parts = []
for block in blocks:
text = block.get_text("", strip=True)
if text and len(text) > 15 and text not in seen:
seen.add(text)
parts.append(f"<p>{html_mod.unescape(text)}</p>")
content = "".join(parts)
if len(re.sub(r"<[^>]+>", "", content)) > 50:
return content
return ""
def run(db_config: dict):
"""主入口:查询无正文文章 → 逐条抓取 → 回写数据库"""
conn = pymysql.connect(**db_config, cursorclass=pymysql.cursors.DictCursor)
try:
# 每个分类各取 BATCH_SIZE 条,避免某个分类独占名额
rows = []
with conn.cursor() as cur:
for cid in TARGET_CIDS:
cur.execute(
"""SELECT id, title, source_url, content, is_show FROM la_article
WHERE cid = %s
AND ((content IS NULL OR content = '' OR CHAR_LENGTH(content) < 50) OR is_show = 0)
AND content_retry < %s
AND source_url IS NOT NULL AND source_url != ''
ORDER BY content_retry ASC, id DESC
LIMIT %s""",
(cid, MAX_RETRY, BATCH_SIZE * 12)
)
candidates = cur.fetchall()
selected = []
for row in candidates:
if has_publishable_article_content(row.get("content", "")):
continue
selected.append(row)
if len(selected) >= BATCH_SIZE:
break
rows.extend(selected)
if not rows:
print("没有需要抓取正文的文章")
return {"success": True, "candidate_count": 0, "saved_count": 0, "summary_text": "没有需要抓取正文的文章"}
print(f"待抓取正文: {len(rows)} 条")
# 并行抓取内容
def _fetch_one(row):
article_id = row["id"]
title = row["title"][:40]
url = row["source_url"]
print(f" [{article_id}] {title}...")
content = fetch_article_content(url)
return article_id, title, content
success = 0
fail = 0
skip = 0
futures = {}
with ThreadPoolExecutor(max_workers=MAX_WORKERS) as executor:
for row in rows:
fut = executor.submit(_fetch_one, row)
futures[fut] = row["id"]
for fut in as_completed(futures):
article_id, title, content = fut.result()
if content and not content.startswith("__") and len(content) > 50:
with conn.cursor() as cur:
cur.execute(
"UPDATE la_article SET content = %s, is_show = 1, content_retry = 0, update_time = %s WHERE id = %s",
(content, int(time.time()), article_id)
)
conn.commit()
success += 1
print(f" ✓ [{article_id}] 成功 ({len(content)}字)")
else:
# 失败,content_retry +1
with conn.cursor() as cur:
cur.execute(
"UPDATE la_article SET content_retry = content_retry + 1 WHERE id = %s",
(article_id,)
)
conn.commit()
fail += 1
reason = content if content and content.startswith("__") else "解析失败"
print(f" ✗ [{article_id}] {reason}retry+1")
summary = f"抓取完成: 成功 {success}, 失败 {fail}, 共 {len(rows)} 条"
print(f"\n{summary}")
return {"success": True, "candidate_count": len(rows), "saved_count": success, "summary_text": summary}
finally:
conn.close()