"""
FIFA 世界杯资讯采集模块 - 从 FIFA CXM API 获取世界杯相关新闻并写入 la_article 表
在 main.py 中通过 fifa_worldcup_news 命令调用
"""
import html
import json
import re
import time
import zlib
from datetime import datetime, timezone
from urllib.parse import urljoin
import pymysql
import requests
from article_publish_helper import has_publishable_article_content, resolve_crawled_article_is_show
BASE_URL = "https://cxm-api.fifa.com/fifaplusweb/api"
LOCALE = "en"
TOURNAMENT_PATH = "/en/tournaments/mens/worldcup/canadamexicousa2026"
TOURNAMENT_PAGE_URL = f"{BASE_URL}/pages{TOURNAMENT_PATH}"
NEWS_SECTION_URL = f"{BASE_URL}/sections/news/1aQDyhkYnKhkAW347zYi4Y?locale={LOCALE}"
ARTICLE_SECTION_URL = f"{BASE_URL}/sections/article/{{entry_id}}?locale={LOCALE}"
DEFAULT_CATEGORY_NAME = "世界杯"
DEFAULT_CATEGORY_SORT = 999
TITLE_MAX_LEN = 255
DESC_MAX_LEN = 255
ABSTRACT_MAX_LEN = 500
IMAGE_MAX_LEN = 128
AUTHOR_MAX_LEN = 255
PUBLISHED_AT_MAX_LEN = 30
CATEGORY_MAX_LEN = 50
SOURCE_URL_MAX_LEN = 500
def normalize_prefix(prefix: str) -> str:
prefix = prefix or "la_"
return prefix if re.match(r"^[A-Za-z0-9_]+$", prefix) else "la_"
def truncate(value: str, limit: int) -> str:
return (value or "")[:limit]
def console_print(message: str) -> None:
try:
print(message)
except UnicodeEncodeError:
print(message.encode("utf-8", "backslashreplace").decode("utf-8"))
def strip_html(text: str) -> str:
text = re.sub(r"<[^>]+>", "", text or "")
text = html.unescape(text)
text = text.replace("\xa0", " ")
return text.strip()
def parse_iso8601(date_str: str) -> tuple[int, str]:
if not date_str:
now = int(time.time())
return now, datetime.fromtimestamp(now).strftime("%Y-%m-%d %H:%M:%S")
try:
dt = datetime.fromisoformat(date_str.replace("Z", "+00:00"))
return int(dt.timestamp()), dt.strftime("%Y-%m-%d %H:%M:%S")
except Exception:
now = int(time.time())
return now, datetime.fromtimestamp(now).strftime("%Y-%m-%d %H:%M:%S")
def stable_article_id(entry_id: str) -> int:
"""将 FIFA entryId 映射成稳定的负整数,避开现有正整数 article_id。"""
if not entry_id:
return 0
return -int(zlib.crc32(f"fifa_worldcup:{entry_id}".encode("utf-8")) & 0x7FFFFFFF)
def get_headers() -> dict:
return {
"User-Agent": "Mozilla/5.0 (compatible; SportEra/1.0)",
"Accept": "application/json, text/plain, */*",
"Origin": "https://www.fifa.com",
"Referer": "https://www.fifa.com" + TOURNAMENT_PATH,
}
def node_text(node) -> str:
if node is None:
return ""
if isinstance(node, str):
return node
node_type = node.get("nodeType", "")
content = node.get("content", []) or []
if node_type == "text":
return html.escape(str(node.get("value", "")))
if node_type == "hyperlink":
href = node.get("data", {}).get("uri", "")
inner = "".join(node_text(child) for child in content)
if href:
return f'{inner}'
return inner
if node_type in ("paragraph", "heading-1", "heading-2", "heading-3", "heading-4", "heading-5", "heading-6"):
inner = "".join(node_text(child) for child in content).strip()
if not inner:
return ""
if node_type.startswith("heading-"):
level = node_type.split("-")[-1]
return f"{inner}"
return f"
{inner}
"
if node_type == "unordered-list":
items = [node_text(child) for child in content if node_text(child)]
if not items:
return ""
return ""
if node_type == "list-item":
inner = "".join(node_text(child) for child in content).strip()
return f"{inner}" if inner else ""
if node_type == "embedded-entry-block":
target = node.get("data", {}).get("target", {}) or {}
images = target.get("bynderImage") or []
if isinstance(images, list) and images:
image = images[0] or {}
src = image.get("src") or image.get("original") or image.get("transformBaseUrl") or ""
alt = image.get("title") or image.get("description") or ""
if src:
return f'})
'
url = target.get("url") or ""
if url:
text = target.get("title") or url
return f'{html.escape(str(text))}
'
return ""
if node_type == "document":
parts = [node_text(child) for child in content]
return "\n".join(part for part in parts if part)
parts = [node_text(child) for child in content]
return "".join(parts)
def html_to_text(value: str) -> str:
return strip_html(value)
def fetch_tournament_page(session: requests.Session) -> dict:
resp = session.get(TOURNAMENT_PAGE_URL, timeout=30)
resp.raise_for_status()
return resp.json()
def fetch_news_section(session: requests.Session) -> dict:
resp = session.get(NEWS_SECTION_URL, timeout=30)
resp.raise_for_status()
return resp.json()
def fetch_article_detail(session: requests.Session, entry_id: str) -> dict:
url = ARTICLE_SECTION_URL.format(entry_id=entry_id)
resp = session.get(url, timeout=30)
resp.raise_for_status()
return resp.json()
def extract_article_detail(entry_json: dict) -> dict:
if not isinstance(entry_json, dict):
return {}
props = entry_json.get("properties") or {}
richtext = props.get("richtext") or entry_json.get("richtext") or {}
hero_image = (
props.get("heroImage")
or props.get("previewImage")
or entry_json.get("heroImage")
or entry_json.get("previewImage")
or {}
)
article_title = props.get("articleTitle") or entry_json.get("articleTitle") or entry_json.get("title") or ""
article_published = props.get("articlePublishedDate") or entry_json.get("articlePublishedDate") or ""
preview_text = props.get("articlePreviewText") or entry_json.get("articlePreviewText") or ""
content_html = node_text(richtext) if richtext else ""
image = ""
if isinstance(hero_image, dict):
image = hero_image.get("src") or hero_image.get("url") or hero_image.get("transformBaseUrl") or ""
return {
"title": article_title,
"desc": html_to_text(preview_text)[:255],
"content": content_html or (html.escape(html_to_text(preview_text)) if preview_text else ""),
"image": image,
"published_at": article_published,
}
def fetch_articles(session: requests.Session, section: dict) -> list[dict]:
items = []
for raw in section.get("items", []) or []:
entry_id = (raw.get("entryId") or "").strip()
slug = (raw.get("slug") or "").strip()
title = (raw.get("title") or "").strip()
if not entry_id or not slug:
continue
page_url = urljoin("https://www.fifa.com", raw.get("articlePageUrl") or f"{TOURNAMENT_PATH}/articles/{slug}")
preview_text = html_to_text(raw.get("previewText") or "")[:255]
published_at = raw.get("publishedDate") or ""
image = ""
img = raw.get("image") or {}
if isinstance(img, dict):
image = img.get("src") or img.get("url") or img.get("transformBaseUrl") or ""
try:
detail_json = fetch_article_detail(session, entry_id)
detail = extract_article_detail(detail_json)
except Exception:
detail = {}
if detail.get("title"):
title = detail["title"]
article_image = detail.get("image") or image
publish_ts, publish_fmt = parse_iso8601(detail.get("published_at") or published_at)
content = detail.get("content") or html.escape(preview_text)
items.append({
"entry_id": entry_id,
"title": truncate(title, TITLE_MAX_LEN),
"desc": truncate(detail.get("desc") or preview_text, DESC_MAX_LEN),
"content": content,
"source_url": page_url,
"image": article_image,
"author": "FIFA",
"published_ts": publish_ts,
"published_at": publish_fmt,
"category": "World Cup",
"ext": {
"source": "fifa_cxm",
"entry_id": entry_id,
"slug": slug,
"article_page_url": page_url,
"tournament_path": TOURNAMENT_PATH,
},
})
return items
def ensure_worldcup_category(conn, prefix: str) -> int:
cate_table = f"{prefix}article_cate"
with conn.cursor() as cur:
cur.execute(
f"SELECT id FROM `{cate_table}` WHERE delete_time IS NULL AND name = %s LIMIT 1",
(DEFAULT_CATEGORY_NAME,)
)
row = cur.fetchone()
now = int(time.time())
if row:
cid = int(row["id"])
cur.execute(
f"UPDATE `{cate_table}` SET is_show = 1, sort = %s, update_time = %s WHERE id = %s",
(DEFAULT_CATEGORY_SORT, now, cid)
)
return cid
cur.execute(
f"""INSERT INTO `{cate_table}` (pid, name, sort, is_show, create_time, update_time)
VALUES (0, %s, %s, 1, %s, %s)""",
(DEFAULT_CATEGORY_NAME, DEFAULT_CATEGORY_SORT, now, now)
)
return cur.lastrowid
def sync_to_db(conn, articles: list[dict], cid: int, prefix: str) -> tuple[int, int, int]:
if not articles:
return 0, 0, 0
urls = [a["source_url"] for a in articles if a.get("source_url")]
if not urls:
return 0, 0, 0
article_table = f"{prefix}article"
with conn.cursor() as cur:
placeholders = ",".join(["%s"] * len(urls))
cur.execute(
f"SELECT id, source_url, content, is_show FROM `{article_table}` WHERE source_url IN ({placeholders}) AND cid = %s",
urls + [cid]
)
existing = {row["source_url"]: row for row in cur.fetchall()}
titles = [a["title"][:255] for a in articles if a.get("title")]
existing_titles = set()
if titles:
with conn.cursor() as cur:
t_placeholders = ",".join(["%s"] * len(titles))
cur.execute(
f"SELECT title FROM `{article_table}` WHERE title IN ({t_placeholders}) AND cid = %s",
titles + [cid]
)
existing_titles = {row["title"] for row in cur.fetchall()}
new_count = 0
update_count = 0
now = int(time.time())
for article in articles:
source_url = article.get("source_url", "")
title = (article.get("title") or "").strip()
if not source_url or not title:
continue
create_ts = article.get("published_ts") or now
article_id = stable_article_id(article.get("entry_id", ""))
ext_json = json.dumps(article.get("ext", {}), ensure_ascii=False) if article.get("ext") else None
image = truncate(article.get("image", ""), IMAGE_MAX_LEN)
author = truncate(article.get("author") or "", AUTHOR_MAX_LEN)
category = truncate(article.get("category") or "", CATEGORY_MAX_LEN)
published_at = truncate(article.get("published_at") or "", PUBLISHED_AT_MAX_LEN)
desc = truncate(article.get("desc") or "", DESC_MAX_LEN)
abstract = truncate(article.get("desc") or "", ABSTRACT_MAX_LEN)
source_url = truncate(source_url, SOURCE_URL_MAX_LEN)
if source_url in existing:
existing_row = existing[source_url]
next_is_show = existing_row["is_show"] if has_publishable_article_content(existing_row.get("content", "")) else 0
with conn.cursor() as cur:
cur.execute(
f"""UPDATE `{article_table}` SET
title = %s, `desc` = %s, image = %s, author = %s,
category = %s, ext = %s, is_show = %s, update_time = %s
WHERE id = %s""",
(
truncate(title, TITLE_MAX_LEN),
desc,
image,
author,
category,
ext_json,
next_is_show,
now,
existing_row["id"],
)
)
update_count += 1
else:
if title[:255] in existing_titles:
continue
with conn.cursor() as cur:
cur.execute(
f"""INSERT INTO `{article_table}`
(article_id, cid, title, `desc`, abstract, image, author, published_at,
category, source_url, content, ext, is_show, sort,
click_virtual, click_actual, create_time, update_time)
VALUES (%s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s)""",
(
article_id,
cid,
truncate(title, TITLE_MAX_LEN),
desc,
abstract,
image,
author,
published_at,
category,
source_url,
article.get("content") or "",
ext_json,
resolve_crawled_article_is_show(article.get("content") or ""),
0,
0,
0,
create_ts,
now,
)
)
existing[source_url] = {"id": cur.lastrowid, "content": article.get("content") or "", "is_show": resolve_crawled_article_is_show(article.get("content") or "")}
existing_titles.add(title[:255])
new_count += 1
return len(articles), new_count, update_count
def run(db_config: dict):
db_config = dict(db_config)
prefix = normalize_prefix(db_config.pop("prefix", "la_"))
session = requests.Session()
session.headers.update(get_headers())
try:
page = fetch_tournament_page(session)
news_section = fetch_news_section(session)
articles = fetch_articles(session, news_section)
console_print(f"世界杯页面: {page.get('meta', {}).get('title', '')}")
console_print(f"新闻区块: {news_section.get('entryId', '')}, 原始条数: {len(articles)}")
except Exception as e:
console_print(f"[WARN] FIFA 页面/新闻区块拉取失败: {e}")
return {"success": False, "candidate_count": 0, "saved_count": 0, "error_message": str(e), "summary_text": f"FIFA 页面/新闻区块拉取失败: {e}"}
if not articles:
console_print("世界杯资讯无数据,跳过")
return {"success": True, "candidate_count": 0, "saved_count": 0, "summary_text": "世界杯资讯无数据,跳过"}
conn = pymysql.connect(**db_config, cursorclass=pymysql.cursors.DictCursor)
try:
cid = ensure_worldcup_category(conn, prefix)
total, new_count, update_count = sync_to_db(conn, articles, cid, prefix)
conn.commit()
console_print(f"分类ID: {cid}")
summary = f"共 {total} 条, 新增 {new_count}, 更新 {update_count}"
console_print(summary)
return {"success": True, "candidate_count": total, "saved_count": new_count + update_count, "summary_text": summary}
finally:
conn.close()