#!/usr/bin/env python3 """ Scrapes the current Dota 2 hero roster from Liquipedia and writes an sqlx migration that seeds the `heroes` table. Requires: pip install requests beautifulsoup4 Respects Liquipedia's API Terms of Use (https://liquipedia.net/api-terms-of-use): - action=query requests: max 1 per 2 seconds (used for batched hero data) - action=parse requests: max 1 per 30 seconds (used exactly once, for the hero grid) - Requires a descriptive User-Agent with real contact info -- EDIT USER_AGENT BELOW. """ import re import sys import time from pathlib import Path import requests from bs4 import BeautifulSoup # --- EDIT THIS before running: Liquipedia requires real contact info in the User-Agent --- USER_AGENT = "DotaRandomWebHeroScraper/1.0 (contact: YOUR_EMAIL_HERE)" API_BASE = "https://liquipedia.net/dota2/api.php" HEADERS = {"User-Agent": USER_AGENT, "Accept-Encoding": "gzip"} # This script lives in /scripts/, migrations live in /migrations/ MIGRATIONS_DIR = Path(__file__).resolve().parent.parent / "migrations" QUERY_BATCH_SIZE = 50 # MediaWiki's default action=query title batch limit GENERAL_RATE_LIMIT_SECONDS = 2 # applies between action=query batches def fetch_hero_list(): """Get the current hero roster (name + image URL) from the Portal:Heroes grid.""" params = {"action": "parse", "page": "Portal:Heroes", "format": "json", "prop": "text"} resp = requests.get(API_BASE, params=params, headers=HEADERS, timeout=30) resp.raise_for_status() html = resp.json()["parse"]["text"]["*"] soup = BeautifulSoup(html, "html.parser") heroes = [] for card in soup.select("div.heroes-panel__hero-card"): link = card.select_one(".heroes-panel__hero-card__title a") img = card.select_one("img") if not link or not img: continue name = link.get("title") or link.text.strip() heroes.append({"name": name, "image_url": "https://liquipedia.net" + img["src"]}) return heroes def fetch_hero_infoboxes(names): """Batch-fetch raw wikitext for hero pages via action=query (cheap, not the expensive parse action).""" infoboxes = {} for i in range(0, len(names), QUERY_BATCH_SIZE): batch = names[i:i + QUERY_BATCH_SIZE] params = { "action": "query", "prop": "revisions", "rvslots": "main", "rvprop": "content", "titles": "|".join(batch), "format": "json", } resp = requests.get(API_BASE, params=params, headers=HEADERS, timeout=30) resp.raise_for_status() for page in resp.json()["query"]["pages"].values(): title = page.get("title") revisions = page.get("revisions") if not revisions: print(f" WARNING: no content found for '{title}', skipping", file=sys.stderr) continue infoboxes[title] = revisions[0]["slots"]["main"]["*"] if i + QUERY_BATCH_SIZE < len(names): time.sleep(GENERAL_RATE_LIMIT_SECONDS) return infoboxes def parse_infobox_field(wikitext, field): match = re.search(rf"\|\s*{re.escape(field)}\s*=\s*(.+)", wikitext) return match.group(1).strip() if match else None def sql_escape(value): return value.replace("'", "''") def write_migration(rows): timestamp = time.strftime("%Y%m%d%H%M%S") up_path = MIGRATIONS_DIR / f"{timestamp}_seed_heroes.up.sql" down_path = MIGRATIONS_DIR / f"{timestamp}_seed_heroes.down.sql" values = ",\n".join( f" ('{sql_escape(r['name'])}', '{r['primary_attribute']}', '{r['attack_type']}', '{sql_escape(r['image_url'])}')" for r in rows ) up_sql = f"INSERT INTO heroes (name, primary_attribute, attack_type, image_url) VALUES\n{values};\n" down_sql = "DELETE FROM heroes;\n" MIGRATIONS_DIR.mkdir(exist_ok=True) up_path.write_text(up_sql, encoding="utf-8") down_path.write_text(down_sql, encoding="utf-8") print(f"\nWrote migration:\n {up_path}\n {down_path}") print("Run `sqlx migrate run` from the project root to apply it.") def main(): if "YOUR_EMAIL_HERE" in USER_AGENT: sys.exit("Edit USER_AGENT at the top of this script with real contact info before running (Liquipedia's API terms require it).") print("Fetching current hero roster from Portal:Heroes ...") heroes = fetch_hero_list() print(f" Found {len(heroes)} heroes on the current grid.") names = [h["name"] for h in heroes] print("Fetching hero infoboxes in batches ...") infoboxes = fetch_hero_infoboxes(names) rows = [] skipped = [] for hero in heroes: wikitext = infoboxes.get(hero["name"]) if not wikitext: skipped.append(hero["name"]) continue primary = parse_infobox_field(wikitext, "primary") rangetype = parse_infobox_field(wikitext, "rangetype") if not primary or not rangetype: skipped.append(hero["name"]) continue rows.append({ "name": hero["name"], "primary_attribute": primary.lower(), "attack_type": rangetype.lower(), "image_url": hero["image_url"], }) print(f" Parsed {len(rows)} heroes successfully.") if skipped: print(f" Skipped {len(skipped)} heroes (missing infobox data): {', '.join(skipped)}") write_migration(rows) if __name__ == "__main__": main()