Reorganize repository into pc web shared monorepo

Separate the local recognition, web publishing, and shared data paths while preserving direct script execution and existing site content.

Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
voson
2026-07-29 14:29:08 +08:00
co-authored by Cursor
parent 96a9312194
commit 9c5aa5b610
280 changed files with 1451 additions and 450 deletions
+439
View File
@@ -0,0 +1,439 @@
"""Fetch recent league/tournament matches for OpenDota registered pros.
Pulls /proPlayers, then per player /players/{id}/matches (lobby practice +
tournament), enriches with /matches/{id} for final items + skill builds.
Output: data/pro_matches.json (Climperor web only; not used by recommend).
Usage:
python fetch_pro_matches.py --limit-pros 20 --limit 8
python fetch_pro_matches.py --players 1296625,117421467
python fetch_pro_matches.py --with-team --active-days 45
"""
from __future__ import annotations
import sys
from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
import argparse
import json
import time
import urllib.error
from datetime import datetime, timedelta, timezone
from shared.grid import hero_table
from shared.http_utils import http_json
from shared.paths import DATA
from fetch_hero_matches import (
collect_item_ids,
extract_player_row,
fetch_match,
load_ability_id_map,
)
from fetch_hero_items import load_item_catalog
from fetch_pro_builds import fetch_pro_index
OPENDOTA = "https://api.opendota.com/api"
OUT = DATA / "pro_matches.json"
DEFAULT_LIMIT = 8
DEFAULT_LIMIT_PROS = 40
# OpenDota lobby_type: 1=practice, 2=tournament (pro/league biased).
LOBBY_LEAGUE = (1, 2)
def parse_pro_filter(raw: str, pro_index: dict[int, dict]) -> list[int]:
"""Comma-separated account ids or registered pro names."""
if not raw.strip():
return []
name_to_id: dict[str, int] = {}
for aid, prof in pro_index.items():
for key in ("name",):
val = prof.get(key)
if isinstance(val, str) and val.strip():
name_to_id[val.strip().lower()] = aid
out: list[int] = []
for part in raw.split(","):
token = part.strip()
if not token:
continue
if token.isdigit():
out.append(int(token))
continue
aid = name_to_id.get(token.lower())
if aid:
out.append(aid)
else:
print(f" warn: unknown pro {token!r}", flush=True)
return out
def filter_pros(
pro_index: dict[int, dict],
*,
with_team: bool,
active_days: int | None,
limit_pros: int,
player_ids: list[int],
) -> list[tuple[int, dict]]:
if player_ids:
rows: list[tuple[int, dict]] = []
for aid in player_ids:
prof = pro_index.get(aid)
if prof:
rows.append((aid, prof))
return rows
cutoff = None
if active_days is not None and active_days > 0:
cutoff = datetime.now(timezone.utc) - timedelta(days=active_days)
candidates: list[tuple[int, dict, float]] = []
for aid, prof in pro_index.items():
if with_team and not prof.get("team_tag") and not prof.get("team_name"):
continue
last = prof.get("last_match_time")
score = 0.0
if isinstance(last, str) and last.strip():
try:
ts = datetime.fromisoformat(last.replace("Z", "+00:00"))
if cutoff and ts < cutoff:
continue
score = ts.timestamp()
except ValueError:
if cutoff:
continue
candidates.append((aid, prof, score))
candidates.sort(key=lambda t: (-t[2], str(t[1].get("name") or ""), t[0]))
picked = candidates[: max(1, limit_pros)]
return [(aid, prof) for aid, prof, _ in picked]
def player_match_metas(
account_id: int,
limit: int,
*,
lobby_types: tuple[int, ...] = LOBBY_LEAGUE,
) -> list[dict]:
"""Recent match list rows for a pro (deduped, newest first)."""
per_lt = max(limit, limit // max(1, len(lobby_types)) + 2)
by_id: dict[int, dict] = {}
for lt in lobby_types:
url = f"{OPENDOTA}/players/{account_id}/matches?limit={per_lt}&lobby_type={lt}"
try:
raw = http_json(url)
except (
urllib.error.HTTPError,
urllib.error.URLError,
TimeoutError,
json.JSONDecodeError,
OSError,
):
continue
if not isinstance(raw, list):
continue
for row in raw:
if not isinstance(row, dict):
continue
try:
mid = int(row.get("match_id") or 0)
except (TypeError, ValueError):
continue
if mid <= 0:
continue
prev = by_id.get(mid)
if prev is None:
by_id[mid] = row
continue
try:
st_new = int(row.get("start_time") or 0)
st_old = int(prev.get("start_time") or 0)
except (TypeError, ValueError):
st_new = st_old = 0
if st_new >= st_old:
by_id[mid] = row
ranked = sorted(
by_id.values(),
key=lambda r: (-int(r.get("start_time") or 0), -int(r.get("match_id") or 0)),
)
return ranked[:limit]
def fetch_player_matches(
account_id: int,
*,
limit: int,
id_map: dict[int, str],
catalog: dict[int, dict],
id_to_key: dict[int, str],
delay: float,
lobby_types: tuple[int, ...],
) -> list[dict]:
metas = player_match_metas(account_id, limit, lobby_types=lobby_types)
rows: list[dict] = []
for meta in metas:
try:
mid = int(meta.get("match_id") or 0)
hid = int(meta.get("hero_id") or 0)
except (TypeError, ValueError):
continue
if mid <= 0 or hid <= 0:
continue
detail = fetch_match(mid)
if delay > 0:
time.sleep(delay)
if not detail:
continue
slim = extract_player_row(
detail,
hid,
origin="pro",
id_map=id_map,
item_catalog=catalog,
list_meta=meta,
)
if not slim:
continue
row_aid = slim.get("account_id")
try:
row_aid_i = int(row_aid) if row_aid is not None else 0
except (TypeError, ValueError):
row_aid_i = 0
if row_aid_i and row_aid_i != account_id:
continue
slim["hero_id"] = hid
slim["hero_key"] = id_to_key.get(hid)
lt = meta.get("lobby_type")
try:
slim["lobby_type"] = int(lt) if lt is not None else None
except (TypeError, ValueError):
slim["lobby_type"] = None
rows.append(slim)
return rows
def build_indexes(
by_pro: dict[str, dict],
id_to_key: dict[int, str],
) -> dict[str, dict]:
by_hero: dict[str, list[dict]] = {}
for cell in by_pro.values():
for row in cell.get("matches") or []:
if not isinstance(row, dict):
continue
key = row.get("hero_key")
if not key:
hid = row.get("hero_id")
try:
key = id_to_key.get(int(hid)) if hid is not None else None
except (TypeError, ValueError):
key = None
if not key:
continue
by_hero.setdefault(str(key), []).append(row)
out: dict[str, dict] = {}
for key, rows in by_hero.items():
seen: set[int] = set()
deduped: list[dict] = []
for row in sorted(
rows,
key=lambda r: (-int(r.get("start_time") or 0), -int(r.get("match_id") or 0)),
):
try:
mid = int(row.get("match_id") or 0)
except (TypeError, ValueError):
continue
if mid in seen:
continue
seen.add(mid)
deduped.append(row)
out[key] = {"matches": deduped}
return out
def write_out(
path: Path,
*,
by_pro: dict[str, dict],
by_hero: dict[str, dict],
pros_meta: dict[str, dict],
item_catalog: dict[int, dict],
limit: int,
limit_pros: int,
lobby_types: tuple[int, ...],
) -> None:
used = collect_item_ids(by_hero)
items_out = {
str(iid): {
"key": item_catalog[iid]["key"],
"dname": item_catalog[iid]["dname"],
"name_loc": item_catalog[iid].get("name_loc") or item_catalog[iid]["dname"],
}
for iid in sorted(used)
if iid in item_catalog
}
match_count = sum(
len(cell.get("matches") or [])
for cell in by_pro.values()
if isinstance(cell, dict)
)
payload = {
"meta": {
"source": "opendota",
"attribution": "https://www.opendota.com",
"fetched_at": datetime.now(timezone.utc).isoformat(),
"limit_per_pro": limit,
"limit_pros": limit_pros,
"lobby_types": list(lobby_types),
"pro_count": len(by_pro),
"match_count": match_count,
"hero_count": len(by_hero),
"note_zh": (
"OpenDota 注册职业选手近期联赛/锦标赛对局;"
"lobby_type 1=训练/practice、2=tournament"
"含终局出装、加点与联赛名(若有)。"
),
},
"items": items_out,
"pros": pros_meta,
"by_pro": by_pro,
"by_hero": by_hero,
}
path.parent.mkdir(parents=True, exist_ok=True)
path.write_text(
json.dumps(payload, ensure_ascii=False, indent=2) + "\n", encoding="utf-8"
)
def _log(msg: str) -> None:
try:
print(msg, flush=True)
except UnicodeEncodeError:
print(msg.encode("ascii", "backslashreplace").decode("ascii"), flush=True)
def main() -> None:
ap = argparse.ArgumentParser(description=__doc__)
ap.add_argument("--out", type=Path, default=OUT)
ap.add_argument(
"--limit",
type=int,
default=DEFAULT_LIMIT,
help=f"Matches per pro (default: {DEFAULT_LIMIT})",
)
ap.add_argument(
"--limit-pros",
type=int,
default=DEFAULT_LIMIT_PROS,
help=f"Max pros when --players omitted (default: {DEFAULT_LIMIT_PROS})",
)
ap.add_argument(
"--players",
default="",
help="Comma-separated account_id or registered pro name (overrides --limit-pros)",
)
ap.add_argument(
"--with-team",
action="store_true",
help="Only pros with a team_tag/team_name when picking from /proPlayers",
)
ap.add_argument(
"--active-days",
type=int,
default=60,
help="Skip pros with no last_match_time within N days (0=disable; default: 60)",
)
ap.add_argument("--delay", type=float, default=0.35)
ap.add_argument(
"--include-pubs",
action="store_true",
help="Also include ranked pub lobby_type=7 (high-MMR scrims)",
)
args = ap.parse_args()
limit = max(1, int(args.limit))
limit_pros = max(1, int(args.limit_pros))
active_days = int(args.active_days) if args.active_days > 0 else None
lobby_types: tuple[int, ...] = LOBBY_LEAGUE
if args.include_pubs:
lobby_types = LOBBY_LEAGUE + (7,)
_log("loading pro players ...")
pro_index = fetch_pro_index()
_log(f" {len(pro_index)} registered pros")
player_ids = parse_pro_filter(args.players, pro_index)
picked = filter_pros(
pro_index,
with_team=args.with_team,
active_days=active_days,
limit_pros=limit_pros,
player_ids=player_ids,
)
if not picked:
raise SystemExit("No pros matched filters")
heroes = hero_table()
id_to_key = {int(h["id"]): h["key"] for h in heroes}
catalog, _ = load_item_catalog()
id_map = load_ability_id_map()
by_pro: dict[str, dict] = {}
pros_meta: dict[str, dict] = {}
total_matches = 0
for i, (aid, prof) in enumerate(picked, 1):
label = prof.get("name") or prof.get("team_tag") or str(aid)
_log(f"[{i}/{len(picked)}] {label} ({aid}) ...")
matches = fetch_player_matches(
aid,
limit=limit,
id_map=id_map,
catalog=catalog,
id_to_key=id_to_key,
delay=args.delay,
lobby_types=lobby_types,
)
sid = str(aid)
by_pro[sid] = {
"account_id": aid,
"name": prof.get("name"),
"team_tag": prof.get("team_tag"),
"team_name": prof.get("team_name"),
"country_code": prof.get("country_code"),
"match_count": len(matches),
"matches": matches,
}
pros_meta[sid] = {
"account_id": aid,
"name": prof.get("name"),
"team_tag": prof.get("team_tag"),
"team_name": prof.get("team_name"),
"country_code": prof.get("country_code"),
}
total_matches += len(matches)
_log(f" {len(matches)} matches")
by_hero = build_indexes(by_pro, id_to_key)
write_out(
args.out,
by_pro=by_pro,
by_hero=by_hero,
pros_meta=pros_meta,
item_catalog=catalog,
limit=limit,
limit_pros=len(picked),
lobby_types=lobby_types,
)
_log(
f"done pros={len(by_pro)} matches={total_matches} heroes={len(by_hero)}{args.out}"
)
if __name__ == "__main__":
main()