151 lines
5.4 KiB
Python
151 lines
5.4 KiB
Python
#!/usr/bin/env python3
|
|
"""
|
|
Scrapes the current Dota 2 hero roster from Liquipedia and writes an sqlx
|
|
migration that seeds the `heroes` table.
|
|
|
|
Requires:
|
|
pip install requests beautifulsoup4
|
|
|
|
Respects Liquipedia's API Terms of Use (https://liquipedia.net/api-terms-of-use):
|
|
- action=query requests: max 1 per 2 seconds (used for batched hero data)
|
|
- action=parse requests: max 1 per 30 seconds (used exactly once, for the hero grid)
|
|
- Requires a descriptive User-Agent with real contact info -- EDIT USER_AGENT BELOW.
|
|
"""
|
|
|
|
import re
|
|
import sys
|
|
import time
|
|
from pathlib import Path
|
|
|
|
import requests
|
|
from bs4 import BeautifulSoup
|
|
|
|
# --- EDIT THIS before running: Liquipedia requires real contact info in the User-Agent ---
|
|
USER_AGENT = "DotaRandomWebHeroScraper/1.0 (contact: YOUR_EMAIL_HERE)"
|
|
|
|
API_BASE = "https://liquipedia.net/dota2/api.php"
|
|
HEADERS = {"User-Agent": USER_AGENT, "Accept-Encoding": "gzip"}
|
|
|
|
# This script lives in <project_root>/scripts/, migrations live in <project_root>/migrations/
|
|
MIGRATIONS_DIR = Path(__file__).resolve().parent.parent / "migrations"
|
|
|
|
QUERY_BATCH_SIZE = 50 # MediaWiki's default action=query title batch limit
|
|
GENERAL_RATE_LIMIT_SECONDS = 2 # applies between action=query batches
|
|
|
|
|
|
def fetch_hero_list():
|
|
"""Get the current hero roster (name + image URL) from the Portal:Heroes grid."""
|
|
params = {"action": "parse", "page": "Portal:Heroes", "format": "json", "prop": "text"}
|
|
resp = requests.get(API_BASE, params=params, headers=HEADERS, timeout=30)
|
|
resp.raise_for_status()
|
|
html = resp.json()["parse"]["text"]["*"]
|
|
soup = BeautifulSoup(html, "html.parser")
|
|
|
|
heroes = []
|
|
for card in soup.select("div.heroes-panel__hero-card"):
|
|
link = card.select_one(".heroes-panel__hero-card__title a")
|
|
img = card.select_one("img")
|
|
if not link or not img:
|
|
continue
|
|
name = link.get("title") or link.text.strip()
|
|
heroes.append({"name": name, "image_url": "https://liquipedia.net" + img["src"]})
|
|
return heroes
|
|
|
|
|
|
def fetch_hero_infoboxes(names):
|
|
"""Batch-fetch raw wikitext for hero pages via action=query (cheap, not the expensive parse action)."""
|
|
infoboxes = {}
|
|
for i in range(0, len(names), QUERY_BATCH_SIZE):
|
|
batch = names[i:i + QUERY_BATCH_SIZE]
|
|
params = {
|
|
"action": "query",
|
|
"prop": "revisions",
|
|
"rvslots": "main",
|
|
"rvprop": "content",
|
|
"titles": "|".join(batch),
|
|
"format": "json",
|
|
}
|
|
resp = requests.get(API_BASE, params=params, headers=HEADERS, timeout=30)
|
|
resp.raise_for_status()
|
|
for page in resp.json()["query"]["pages"].values():
|
|
title = page.get("title")
|
|
revisions = page.get("revisions")
|
|
if not revisions:
|
|
print(f" WARNING: no content found for '{title}', skipping", file=sys.stderr)
|
|
continue
|
|
infoboxes[title] = revisions[0]["slots"]["main"]["*"]
|
|
if i + QUERY_BATCH_SIZE < len(names):
|
|
time.sleep(GENERAL_RATE_LIMIT_SECONDS)
|
|
return infoboxes
|
|
|
|
|
|
def parse_infobox_field(wikitext, field):
|
|
match = re.search(rf"\|\s*{re.escape(field)}\s*=\s*(.+)", wikitext)
|
|
return match.group(1).strip() if match else None
|
|
|
|
|
|
def sql_escape(value):
|
|
return value.replace("'", "''")
|
|
|
|
|
|
def write_migration(rows):
|
|
timestamp = time.strftime("%Y%m%d%H%M%S")
|
|
up_path = MIGRATIONS_DIR / f"{timestamp}_seed_heroes.up.sql"
|
|
down_path = MIGRATIONS_DIR / f"{timestamp}_seed_heroes.down.sql"
|
|
|
|
values = ",\n".join(
|
|
f" ('{sql_escape(r['name'])}', '{r['primary_attribute']}', '{r['attack_type']}', '{sql_escape(r['image_url'])}')"
|
|
for r in rows
|
|
)
|
|
up_sql = f"INSERT INTO heroes (name, primary_attribute, attack_type, image_url) VALUES\n{values};\n"
|
|
down_sql = "DELETE FROM heroes;\n"
|
|
|
|
MIGRATIONS_DIR.mkdir(exist_ok=True)
|
|
up_path.write_text(up_sql, encoding="utf-8")
|
|
down_path.write_text(down_sql, encoding="utf-8")
|
|
|
|
print(f"\nWrote migration:\n {up_path}\n {down_path}")
|
|
print("Run `sqlx migrate run` from the project root to apply it.")
|
|
|
|
|
|
def main():
|
|
if "YOUR_EMAIL_HERE" in USER_AGENT:
|
|
sys.exit("Edit USER_AGENT at the top of this script with real contact info before running (Liquipedia's API terms require it).")
|
|
|
|
print("Fetching current hero roster from Portal:Heroes ...")
|
|
heroes = fetch_hero_list()
|
|
print(f" Found {len(heroes)} heroes on the current grid.")
|
|
|
|
names = [h["name"] for h in heroes]
|
|
print("Fetching hero infoboxes in batches ...")
|
|
infoboxes = fetch_hero_infoboxes(names)
|
|
|
|
rows = []
|
|
skipped = []
|
|
for hero in heroes:
|
|
wikitext = infoboxes.get(hero["name"])
|
|
if not wikitext:
|
|
skipped.append(hero["name"])
|
|
continue
|
|
primary = parse_infobox_field(wikitext, "primary")
|
|
rangetype = parse_infobox_field(wikitext, "rangetype")
|
|
if not primary or not rangetype:
|
|
skipped.append(hero["name"])
|
|
continue
|
|
rows.append({
|
|
"name": hero["name"],
|
|
"primary_attribute": primary.lower(),
|
|
"attack_type": rangetype.lower(),
|
|
"image_url": hero["image_url"],
|
|
})
|
|
|
|
print(f" Parsed {len(rows)} heroes successfully.")
|
|
if skipped:
|
|
print(f" Skipped {len(skipped)} heroes (missing infobox data): {', '.join(skipped)}")
|
|
|
|
write_migration(rows)
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|