迁移目录

This commit is contained in:
hajimi
2026-06-12 22:27:18 +08:00
parent 7381097582
commit cd44cd6e47
92 changed files with 1839 additions and 0 deletions
+158
View File
@@ -0,0 +1,158 @@
"""
Truth Social 采集模块 - 从 RapidAPI 获取帖子并入库
在 main.py 中通过 truth_social 命令调用
"""
import json
import re
import time
from datetime import datetime
import pymysql
import requests
# ─── 配置 ───
RAPIDAPI_KEY = "482d762c92mshfd65e1ec2ce81e4p103837jsna4df0033beb4"
RAPIDAPI_HOST = "truth-social-api.p.rapidapi.com"
RAPIDAPI_URL = f"https://{RAPIDAPI_HOST}"
SOURCE = "truthsocial"
TARGETS = [
{"username": "realDonaldTrump", "user_id": 8, "tag_id": 13},
]
def strip_html(html: str) -> str:
return re.sub(r"<[^>]+>", "", html or "").strip()
def parse_timestamp(ts_str: str) -> int:
if not ts_str:
return int(time.time())
ts_str = ts_str.replace("Z", "+00:00")
try:
return int(datetime.fromisoformat(ts_str).timestamp())
except Exception:
return int(time.time())
def fetch_feed(session: requests.Session, username: str, max_pages: int = 3) -> list[dict]:
all_posts = []
continue_id = None
for page in range(max_pages):
url = f"{RAPIDAPI_URL}/users/{username}/feed"
params = {}
if continue_id:
params["continue_from_id"] = continue_id
try:
resp = session.get(url, params=params, timeout=30)
except Exception as e:
print(f" [WARN] 请求失败: {e}")
break
if resp.status_code != 200:
print(f" [WARN] API {resp.status_code}: {resp.text[:200]}")
break
data = resp.json()
posts = data if isinstance(data, list) else data.get("data", data.get("statuses", []))
if not posts:
break
for p in posts:
if "content" in p and "content_plain" not in p:
p["content_plain"] = strip_html(p["content"])
all_posts.extend(posts)
continue_id = str(posts[-1].get("id", ""))
if page < max_pages - 1:
time.sleep(2)
return all_posts
def sync_to_db(conn, posts: list[dict], user_id: int, tag_id: int) -> tuple[int, int]:
with conn.cursor() as cur:
cur.execute(
"SELECT origin_id FROM la_community_post WHERE origin_id LIKE %s",
(f"{SOURCE}:%",)
)
existing_ids = {row["origin_id"] for row in cur.fetchall()}
new_count = 0
for post in posts:
post_id = str(post.get("id", ""))
origin_id = f"{SOURCE}:{post_id}"
if not post_id or origin_id in existing_ids:
continue
content_plain = post.get("content_plain") or strip_html(post.get("content", ""))
created_at = parse_timestamp(post.get("created_at", ""))
images = []
for media in post.get("media_attachments", []):
url = media.get("url", "")
if url and media.get("type") in ("image", "gifv"):
images.append(url)
with conn.cursor() as cur:
cur.execute(
"""INSERT INTO la_community_post
(origin_id, user_id, content, images, post_type, ext, status, create_time, update_time)
VALUES (%s, %s, %s, %s, %s, %s, %s, %s, %s)""",
(
origin_id, user_id, content_plain,
json.dumps(images) if images else "[]",
0, json.dumps(post, ensure_ascii=False), 1,
created_at, created_at,
)
)
new_post_id = cur.lastrowid
cur.execute(
"INSERT INTO la_community_post_tag (post_id, tag_id) VALUES (%s, %s)",
(new_post_id, tag_id)
)
for idx, img_url in enumerate(images):
cur.execute(
"INSERT INTO la_community_post_image (post_id, image_url, sort, create_time) VALUES (%s, %s, %s, %s)",
(new_post_id, img_url, idx, created_at)
)
existing_ids.add(origin_id)
new_count += 1
return len(existing_ids), new_count
def run(db_config: dict):
"""主入口,db_config 从 dongqiudi-crawler 配置中传入"""
session = requests.Session()
session.headers.update({
"Content-Type": "application/json",
"x-rapidapi-host": RAPIDAPI_HOST,
"x-rapidapi-key": RAPIDAPI_KEY,
})
conn = pymysql.connect(**db_config, cursorclass=pymysql.cursors.DictCursor)
try:
total_candidates = 0
total_saved = 0
for target in TARGETS:
username = target["username"]
print(f"=== @{username} ===")
posts = fetch_feed(session, username)
print(f" API返回 {len(posts)}")
total_candidates += len(posts)
if not posts:
continue
existing, new_count = sync_to_db(conn, posts, target["user_id"], target["tag_id"])
conn.commit()
print(f" 已入库 {existing} 条, 新增 {new_count}")
total_saved += new_count
return {"success": True, "candidate_count": total_candidates, "saved_count": total_saved, "summary_text": f"Truth Social 共拉取 {total_candidates} 条,新增 {total_saved}"}
finally:
conn.close()