Files
dota-random-web/scripts/scrape_heroes.py
T
kwiialyssa edb4cd8613
CI/CD / Build (push) Successful in 5m59s
CI/CD / Frontend Build (push) Successful in 1m27s
CI/CD / Unit Tests (push) Failing after 1m12s
CI/CD / Docker (push) Skipped
CI/CD / Frontend Docker (push) Failing after 3m29s
feat(DRW-001) initial commit!
2026-09-13 20:37:48 +03:00

151 lines
5.4 KiB
Python

#!/usr/bin/env python3
"""
Scrapes the current Dota 2 hero roster from Liquipedia and writes an sqlx
migration that seeds the `heroes` table.
Requires:
pip install requests beautifulsoup4
Respects Liquipedia's API Terms of Use (https://liquipedia.net/api-terms-of-use):
- action=query requests: max 1 per 2 seconds (used for batched hero data)
- action=parse requests: max 1 per 30 seconds (used exactly once, for the hero grid)
- Requires a descriptive User-Agent with real contact info -- EDIT USER_AGENT BELOW.
"""
import re
import sys
import time
from pathlib import Path
import requests
from bs4 import BeautifulSoup
# --- EDIT THIS before running: Liquipedia requires real contact info in the User-Agent ---
USER_AGENT = "DotaRandomWebHeroScraper/1.0 (contact: YOUR_EMAIL_HERE)"
API_BASE = "https://liquipedia.net/dota2/api.php"
HEADERS = {"User-Agent": USER_AGENT, "Accept-Encoding": "gzip"}
# This script lives in <project_root>/scripts/, migrations live in <project_root>/migrations/
MIGRATIONS_DIR = Path(__file__).resolve().parent.parent / "migrations"
QUERY_BATCH_SIZE = 50 # MediaWiki's default action=query title batch limit
GENERAL_RATE_LIMIT_SECONDS = 2 # applies between action=query batches
def fetch_hero_list():
"""Get the current hero roster (name + image URL) from the Portal:Heroes grid."""
params = {"action": "parse", "page": "Portal:Heroes", "format": "json", "prop": "text"}
resp = requests.get(API_BASE, params=params, headers=HEADERS, timeout=30)
resp.raise_for_status()
html = resp.json()["parse"]["text"]["*"]
soup = BeautifulSoup(html, "html.parser")
heroes = []
for card in soup.select("div.heroes-panel__hero-card"):
link = card.select_one(".heroes-panel__hero-card__title a")
img = card.select_one("img")
if not link or not img:
continue
name = link.get("title") or link.text.strip()
heroes.append({"name": name, "image_url": "https://liquipedia.net" + img["src"]})
return heroes
def fetch_hero_infoboxes(names):
"""Batch-fetch raw wikitext for hero pages via action=query (cheap, not the expensive parse action)."""
infoboxes = {}
for i in range(0, len(names), QUERY_BATCH_SIZE):
batch = names[i:i + QUERY_BATCH_SIZE]
params = {
"action": "query",
"prop": "revisions",
"rvslots": "main",
"rvprop": "content",
"titles": "|".join(batch),
"format": "json",
}
resp = requests.get(API_BASE, params=params, headers=HEADERS, timeout=30)
resp.raise_for_status()
for page in resp.json()["query"]["pages"].values():
title = page.get("title")
revisions = page.get("revisions")
if not revisions:
print(f" WARNING: no content found for '{title}', skipping", file=sys.stderr)
continue
infoboxes[title] = revisions[0]["slots"]["main"]["*"]
if i + QUERY_BATCH_SIZE < len(names):
time.sleep(GENERAL_RATE_LIMIT_SECONDS)
return infoboxes
def parse_infobox_field(wikitext, field):
match = re.search(rf"\|\s*{re.escape(field)}\s*=\s*(.+)", wikitext)
return match.group(1).strip() if match else None
def sql_escape(value):
return value.replace("'", "''")
def write_migration(rows):
timestamp = time.strftime("%Y%m%d%H%M%S")
up_path = MIGRATIONS_DIR / f"{timestamp}_seed_heroes.up.sql"
down_path = MIGRATIONS_DIR / f"{timestamp}_seed_heroes.down.sql"
values = ",\n".join(
f" ('{sql_escape(r['name'])}', '{r['primary_attribute']}', '{r['attack_type']}', '{sql_escape(r['image_url'])}')"
for r in rows
)
up_sql = f"INSERT INTO heroes (name, primary_attribute, attack_type, image_url) VALUES\n{values};\n"
down_sql = "DELETE FROM heroes;\n"
MIGRATIONS_DIR.mkdir(exist_ok=True)
up_path.write_text(up_sql, encoding="utf-8")
down_path.write_text(down_sql, encoding="utf-8")
print(f"\nWrote migration:\n {up_path}\n {down_path}")
print("Run `sqlx migrate run` from the project root to apply it.")
def main():
if "YOUR_EMAIL_HERE" in USER_AGENT:
sys.exit("Edit USER_AGENT at the top of this script with real contact info before running (Liquipedia's API terms require it).")
print("Fetching current hero roster from Portal:Heroes ...")
heroes = fetch_hero_list()
print(f" Found {len(heroes)} heroes on the current grid.")
names = [h["name"] for h in heroes]
print("Fetching hero infoboxes in batches ...")
infoboxes = fetch_hero_infoboxes(names)
rows = []
skipped = []
for hero in heroes:
wikitext = infoboxes.get(hero["name"])
if not wikitext:
skipped.append(hero["name"])
continue
primary = parse_infobox_field(wikitext, "primary")
rangetype = parse_infobox_field(wikitext, "rangetype")
if not primary or not rangetype:
skipped.append(hero["name"])
continue
rows.append({
"name": hero["name"],
"primary_attribute": primary.lower(),
"attack_type": rangetype.lower(),
"image_url": hero["image_url"],
})
print(f" Parsed {len(rows)} heroes successfully.")
if skipped:
print(f" Skipped {len(skipped)} heroes (missing infobox data): {', '.join(skipped)}")
write_migration(rows)
if __name__ == "__main__":
main()