Ship Web refresh cache/lock, mobile demand gate, matches 职业/国服 filter, and related site updates through 0.5.84. Co-authored-by: Cursor <cursoragent@cursor.com>
471 lines
14 KiB
Python
471 lines
14 KiB
Python
"""Fetch Douyin profile fields into data/streamers.json.
|
||
|
||
Manual seed fields (id / platform / live_url / profile_url / heroes / tagline)
|
||
are preserved. Profile enrichment (nickname, signature, counts, avatar) is
|
||
best-effort via Douyin HTML RENDER_DATA + text fallback; failures keep the
|
||
previous values.
|
||
|
||
Preview only — do not merge into relations/heroes or recommend.
|
||
Part of refresh_web ``daily`` (soft-fail: never aborts the tier).
|
||
|
||
Usage:
|
||
python fetch_streamers.py
|
||
python fetch_streamers.py --ids xiaowang
|
||
python fetch_streamers.py --out data/streamers.json
|
||
"""
|
||
|
||
from __future__ import annotations
|
||
|
||
import sys
|
||
from pathlib import Path
|
||
|
||
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
|
||
|
||
import argparse
|
||
import http.cookiejar
|
||
import json
|
||
import re
|
||
import time
|
||
import urllib.error
|
||
import urllib.parse
|
||
import urllib.request
|
||
from datetime import datetime, timezone
|
||
from typing import Any
|
||
|
||
from shared.http_utils import write_json_atomic
|
||
from shared.paths import DATA, STREAMER_AVATARS
|
||
|
||
OUT = DATA / "streamers.json"
|
||
AVATAR_DIR = STREAMER_AVATARS
|
||
BROWSER_UA = (
|
||
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
|
||
"AppleWebKit/537.36 (KHTML, like Gecko) "
|
||
"Chrome/120.0.0.0 Safari/537.36"
|
||
)
|
||
RENDER_DATA_RE = re.compile(
|
||
r'<script[^>]+id=["\']RENDER_DATA["\'][^>]*>([^<]+)</script>',
|
||
re.IGNORECASE,
|
||
)
|
||
HTML_COUNTS_RE = re.compile(
|
||
r"关注\s*([\d.]+万?亿?)\s*粉丝\s*([\d.]+万?亿?)\s*获赞\s*([\d.]+万?亿?)",
|
||
re.DOTALL,
|
||
)
|
||
HTML_UNIQUE_RE = re.compile(r"抖音号[::]\s*([A-Za-z0-9_.-]+)")
|
||
HTML_TITLE_RE = re.compile(r"<title>\s*([^<]+?)\s*的抖音", re.IGNORECASE)
|
||
HTML_AVATAR_RE = re.compile(
|
||
r'<img[^>]+alt="[^"]*头像"[^>]+src="([^"]+)"|'
|
||
r'<img[^>]+src="([^"]+)"[^>]+alt="[^"]*头像"',
|
||
re.IGNORECASE,
|
||
)
|
||
SEC_UID_RE = re.compile(r"/user/(MS4wLjABAAAA[A-Za-z0-9_-]+)")
|
||
# Fields fetch may overwrite; manual seed keys are never removed.
|
||
PROFILE_KEYS = (
|
||
"nickname",
|
||
"unique_id",
|
||
"signature",
|
||
"following_count",
|
||
"follower_count",
|
||
"total_favorited",
|
||
"avatar",
|
||
"profile_fetched_at",
|
||
)
|
||
|
||
|
||
def _now_iso() -> str:
|
||
return datetime.now(timezone.utc).isoformat()
|
||
|
||
|
||
def _load(path: Path) -> dict:
|
||
if not path.is_file():
|
||
return {
|
||
"fetched_at": None,
|
||
"source": "manual+douyin",
|
||
"platform_meta": {
|
||
"douyin": {
|
||
"label_zh": "抖音",
|
||
"icon": "ui-icon/platform_douyin.png",
|
||
}
|
||
},
|
||
"streamers": [],
|
||
}
|
||
raw = json.loads(path.read_text(encoding="utf-8"))
|
||
if not isinstance(raw, dict):
|
||
raise SystemExit(f"{path}: expected object")
|
||
return raw
|
||
|
||
|
||
def _save(path: Path, payload: dict) -> None:
|
||
write_json_atomic(path, payload)
|
||
|
||
|
||
def _opener() -> urllib.request.OpenerDirector:
|
||
jar = http.cookiejar.CookieJar()
|
||
return urllib.request.build_opener(urllib.request.HTTPCookieProcessor(jar))
|
||
|
||
|
||
def _get(
|
||
opener: urllib.request.OpenerDirector, url: str, *, timeout: int = 30
|
||
) -> tuple[str, str]:
|
||
"""Return (final_url, html)."""
|
||
req = urllib.request.Request(
|
||
url,
|
||
headers={
|
||
"User-Agent": BROWSER_UA,
|
||
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
|
||
"Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
|
||
"Referer": "https://www.douyin.com/",
|
||
},
|
||
)
|
||
with opener.open(req, timeout=timeout) as resp:
|
||
final = resp.geturl()
|
||
html = resp.read().decode("utf-8", "replace")
|
||
return final, html
|
||
|
||
|
||
def parse_cn_count(raw: str) -> int | None:
|
||
s = (raw or "").strip().replace(",", "").replace("\n", "")
|
||
if not s:
|
||
return None
|
||
try:
|
||
if s.endswith("亿"):
|
||
return int(float(s[:-1]) * 100_000_000)
|
||
if s.endswith("万"):
|
||
return int(float(s[:-1]) * 10_000)
|
||
return int(float(s))
|
||
except ValueError:
|
||
return None
|
||
|
||
|
||
def _parse_render_data(html: str) -> dict | None:
|
||
m = RENDER_DATA_RE.search(html)
|
||
if not m:
|
||
return None
|
||
try:
|
||
decoded = urllib.parse.unquote(m.group(1))
|
||
data = json.loads(decoded)
|
||
except (ValueError, json.JSONDecodeError):
|
||
return None
|
||
return data if isinstance(data, dict) else None
|
||
|
||
|
||
def _walk_user_nodes(obj: Any, found: list[dict]) -> None:
|
||
if isinstance(obj, dict):
|
||
if "nickname" in obj and (
|
||
"follower_count" in obj
|
||
or "followerCount" in obj
|
||
or "mplatform_followers_count" in obj
|
||
):
|
||
found.append(obj)
|
||
for v in obj.values():
|
||
_walk_user_nodes(v, found)
|
||
elif isinstance(obj, list):
|
||
for v in obj:
|
||
_walk_user_nodes(v, found)
|
||
|
||
|
||
def _first_int(*vals: Any) -> int | None:
|
||
for v in vals:
|
||
if isinstance(v, bool):
|
||
continue
|
||
if isinstance(v, int):
|
||
return v
|
||
if isinstance(v, float):
|
||
return int(v)
|
||
if isinstance(v, str):
|
||
n = parse_cn_count(v)
|
||
if n is not None:
|
||
return n
|
||
return None
|
||
|
||
|
||
def _first_str(*vals: Any) -> str | None:
|
||
for v in vals:
|
||
if isinstance(v, str) and v.strip():
|
||
return v.strip()
|
||
return None
|
||
|
||
|
||
def _avatar_url(user: dict) -> str | None:
|
||
for key in (
|
||
"avatar_larger",
|
||
"avatar_medium",
|
||
"avatar_thumb",
|
||
"avatarUrl",
|
||
"avatar_url",
|
||
):
|
||
cell = user.get(key)
|
||
if isinstance(cell, str) and cell.startswith("http"):
|
||
return cell
|
||
if isinstance(cell, dict):
|
||
urls = cell.get("url_list") or cell.get("urlList") or []
|
||
if isinstance(urls, list):
|
||
for u in urls:
|
||
if isinstance(u, str) and u.startswith("http"):
|
||
return u
|
||
return None
|
||
|
||
|
||
def extract_profile_from_render(render: dict) -> dict | None:
|
||
candidates: list[dict] = []
|
||
_walk_user_nodes(render, candidates)
|
||
if not candidates:
|
||
return None
|
||
|
||
def score(u: dict) -> int:
|
||
n = _first_int(
|
||
u.get("follower_count"),
|
||
u.get("followerCount"),
|
||
u.get("mplatform_followers_count"),
|
||
)
|
||
return n if n is not None else -1
|
||
|
||
user = max(candidates, key=score)
|
||
return {
|
||
"nickname": _first_str(user.get("nickname"), user.get("nickName")),
|
||
"unique_id": _first_str(
|
||
user.get("unique_id"),
|
||
user.get("uniqueId"),
|
||
user.get("short_id"),
|
||
user.get("shortId"),
|
||
),
|
||
"signature": _first_str(user.get("signature"), user.get("desc")),
|
||
"following_count": _first_int(
|
||
user.get("following_count"), user.get("followingCount")
|
||
),
|
||
"follower_count": _first_int(
|
||
user.get("follower_count"),
|
||
user.get("followerCount"),
|
||
user.get("mplatform_followers_count"),
|
||
),
|
||
"total_favorited": _first_int(
|
||
user.get("total_favorited"),
|
||
user.get("totalFavorited"),
|
||
user.get("favoriting_count"),
|
||
),
|
||
"avatar_url": _avatar_url(user),
|
||
}
|
||
|
||
|
||
def extract_profile_from_html(html: str) -> dict:
|
||
"""Best-effort parse when RENDER_DATA user blob is missing/blocked."""
|
||
out: dict[str, Any] = {}
|
||
m = HTML_COUNTS_RE.search(html)
|
||
if m:
|
||
out["following_count"] = parse_cn_count(m.group(1))
|
||
out["follower_count"] = parse_cn_count(m.group(2))
|
||
out["total_favorited"] = parse_cn_count(m.group(3))
|
||
uid = HTML_UNIQUE_RE.search(html)
|
||
if uid:
|
||
out["unique_id"] = uid.group(1)
|
||
title = HTML_TITLE_RE.search(html)
|
||
if title:
|
||
out["nickname"] = title.group(1).strip()
|
||
av = HTML_AVATAR_RE.search(html)
|
||
if av:
|
||
url = av.group(1) or av.group(2)
|
||
if url and url.startswith("http"):
|
||
out["avatar_url"] = url.replace("&", "&")
|
||
return out
|
||
|
||
|
||
def merge_profile_dicts(primary: dict | None, fallback: dict) -> dict:
|
||
out = dict(fallback)
|
||
if primary:
|
||
for k, v in primary.items():
|
||
if v is None or v == "":
|
||
continue
|
||
out[k] = v
|
||
return out
|
||
|
||
|
||
def canonicalize_douyin_user_url(url: str) -> str:
|
||
"""Map share / short-link destinations to www.douyin.com/user/<sec_uid>."""
|
||
m = SEC_UID_RE.search(url)
|
||
if m:
|
||
return f"https://www.douyin.com/user/{m.group(1)}"
|
||
return url
|
||
|
||
|
||
def resolve_profile_url(
|
||
opener: urllib.request.OpenerDirector, profile_url: str
|
||
) -> tuple[str, str | None]:
|
||
"""Follow redirects; return (canonical_url, html_or_None if not yet fetched)."""
|
||
url = profile_url.strip()
|
||
host = urllib.parse.urlparse(url).netloc.lower()
|
||
if "v.douyin.com" in host or "iesdouyin.com" in host:
|
||
final, html = _get(opener, url)
|
||
canon = canonicalize_douyin_user_url(final)
|
||
if canon != final and "douyin.com/user/" in canon:
|
||
# Re-fetch canonical profile for RENDER_DATA when possible.
|
||
try:
|
||
_, html2 = _get(opener, canon)
|
||
return canon, html2
|
||
except (urllib.error.URLError, TimeoutError, OSError):
|
||
return canon, html
|
||
return canon, html
|
||
return url, None
|
||
|
||
|
||
def download_avatar(url: str, dest: Path) -> bool:
|
||
dest.parent.mkdir(parents=True, exist_ok=True)
|
||
# Prefer a larger CDN variant when the URL embeds a size token.
|
||
candidates = [url]
|
||
if "/100x100/" in url:
|
||
candidates.insert(0, url.replace("/100x100/", "/720x720/"))
|
||
if "300x300" in url:
|
||
candidates.insert(0, url.replace("300x300", "720x720"))
|
||
for candidate in candidates:
|
||
req = urllib.request.Request(
|
||
candidate,
|
||
headers={
|
||
"User-Agent": BROWSER_UA,
|
||
"Referer": "https://www.douyin.com/",
|
||
},
|
||
)
|
||
try:
|
||
with urllib.request.urlopen(req, timeout=30) as resp:
|
||
data = resp.read()
|
||
except (urllib.error.URLError, TimeoutError, OSError) as e:
|
||
print(f" avatar download failed: {e}", flush=True)
|
||
continue
|
||
if not data or len(data) < 64:
|
||
continue
|
||
dest.write_bytes(data)
|
||
return True
|
||
print(" avatar download empty", flush=True)
|
||
return False
|
||
|
||
|
||
def fetch_douyin_profile(
|
||
opener: urllib.request.OpenerDirector, profile_url: str
|
||
) -> dict:
|
||
try:
|
||
_get(opener, "https://www.douyin.com/")
|
||
except (urllib.error.URLError, TimeoutError, OSError) as e:
|
||
print(f" douyin homepage warm-up failed: {e}", flush=True)
|
||
|
||
canon, html = resolve_profile_url(opener, profile_url)
|
||
if html is None:
|
||
_, html = _get(opener, canon)
|
||
|
||
primary = None
|
||
render = _parse_render_data(html)
|
||
if render is not None:
|
||
primary = extract_profile_from_render(render)
|
||
fallback = extract_profile_from_html(html)
|
||
profile = merge_profile_dicts(primary, fallback)
|
||
if not any(
|
||
profile.get(k) is not None
|
||
for k in (
|
||
"nickname",
|
||
"follower_count",
|
||
"following_count",
|
||
"total_favorited",
|
||
"unique_id",
|
||
"avatar_url",
|
||
)
|
||
):
|
||
raise ValueError("no profile fields parsed (blocked or layout changed)")
|
||
return profile
|
||
|
||
|
||
def merge_profile(row: dict, profile: dict, *, streamer_id: str) -> None:
|
||
for key in (
|
||
"nickname",
|
||
"unique_id",
|
||
"signature",
|
||
"following_count",
|
||
"follower_count",
|
||
"total_favorited",
|
||
):
|
||
val = profile.get(key)
|
||
if val is None or val == "":
|
||
continue
|
||
row[key] = val
|
||
avatar_url = profile.get("avatar_url")
|
||
if isinstance(avatar_url, str) and avatar_url:
|
||
dest = AVATAR_DIR / f"{streamer_id}.jpg"
|
||
if download_avatar(avatar_url, dest):
|
||
row["avatar"] = f"streamer_avatars/{streamer_id}.jpg"
|
||
print(f" avatar saved {dest.relative_to(ROOT)}", flush=True)
|
||
row["profile_fetched_at"] = _now_iso()
|
||
|
||
|
||
def enrich_streamers(
|
||
payload: dict, *, ids: set[str] | None = None
|
||
) -> tuple[int, int, int]:
|
||
rows = payload.get("streamers")
|
||
if not isinstance(rows, list):
|
||
raise SystemExit("streamers.json: missing streamers array")
|
||
opener = _opener()
|
||
ok = skip = fail = 0
|
||
for row in rows:
|
||
if not isinstance(row, dict):
|
||
continue
|
||
sid = str(row.get("id") or "").strip()
|
||
if not sid:
|
||
continue
|
||
if ids is not None and sid not in ids:
|
||
continue
|
||
platform = str(row.get("platform") or "").strip().lower()
|
||
if platform != "douyin":
|
||
print(f"skip {sid}: platform={platform!r} (only douyin supported)", flush=True)
|
||
skip += 1
|
||
continue
|
||
profile_url = str(row.get("profile_url") or "").strip()
|
||
if not profile_url:
|
||
print(f"skip {sid}: missing profile_url", flush=True)
|
||
skip += 1
|
||
continue
|
||
print(f"fetching {sid} ...", flush=True)
|
||
try:
|
||
profile = fetch_douyin_profile(opener, profile_url)
|
||
merge_profile(row, profile, streamer_id=sid)
|
||
nick = row.get("nickname") or "?"
|
||
print(
|
||
f" ok {nick} followers={row.get('follower_count')} "
|
||
f"likes={row.get('total_favorited')}",
|
||
flush=True,
|
||
)
|
||
ok += 1
|
||
except (urllib.error.URLError, TimeoutError, OSError, ValueError) as e:
|
||
print(f" FAIL {sid}: {e} (keeping previous values)", flush=True)
|
||
fail += 1
|
||
time.sleep(0.8)
|
||
if ok > 0:
|
||
payload["fetched_at"] = _now_iso()
|
||
payload["source"] = payload.get("source") or "manual+douyin"
|
||
meta = payload.get("platform_meta")
|
||
if not isinstance(meta, dict):
|
||
meta = {}
|
||
meta.setdefault(
|
||
"douyin",
|
||
{"label_zh": "抖音", "icon": "ui-icon/platform_douyin.png"},
|
||
)
|
||
payload["platform_meta"] = meta
|
||
_ = PROFILE_KEYS
|
||
return ok, skip, fail
|
||
|
||
|
||
def main() -> int:
|
||
ap = argparse.ArgumentParser(description="Enrich streamers.json from Douyin profiles")
|
||
ap.add_argument("--out", type=Path, default=OUT)
|
||
ap.add_argument(
|
||
"--ids",
|
||
nargs="+",
|
||
default=None,
|
||
help="only refresh these streamer ids",
|
||
)
|
||
args = ap.parse_args()
|
||
payload = _load(args.out)
|
||
id_set = set(args.ids) if args.ids else None
|
||
ok, skip, fail = enrich_streamers(payload, ids=id_set)
|
||
_save(args.out, payload)
|
||
print(f"wrote {args.out} ok={ok} skip={skip} fail={fail}", flush=True)
|
||
# Soft-fail for CI/refresh_web: always exit 0 after writing (keep old values).
|
||
return 0
|
||
|
||
|
||
if __name__ == "__main__":
|
||
raise SystemExit(main())
|