feat(DRW-001) initial commit!
This commit is contained in:
@@ -0,0 +1,150 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
Scrapes the current Dota 2 hero roster from Liquipedia and writes an sqlx
|
||||
migration that seeds the `heroes` table.
|
||||
|
||||
Requires:
|
||||
pip install requests beautifulsoup4
|
||||
|
||||
Respects Liquipedia's API Terms of Use (https://liquipedia.net/api-terms-of-use):
|
||||
- action=query requests: max 1 per 2 seconds (used for batched hero data)
|
||||
- action=parse requests: max 1 per 30 seconds (used exactly once, for the hero grid)
|
||||
- Requires a descriptive User-Agent with real contact info -- EDIT USER_AGENT BELOW.
|
||||
"""
|
||||
|
||||
import re
|
||||
import sys
|
||||
import time
|
||||
from pathlib import Path
|
||||
|
||||
import requests
|
||||
from bs4 import BeautifulSoup
|
||||
|
||||
# --- EDIT THIS before running: Liquipedia requires real contact info in the User-Agent ---
|
||||
USER_AGENT = "DotaRandomWebHeroScraper/1.0 (contact: YOUR_EMAIL_HERE)"
|
||||
|
||||
API_BASE = "https://liquipedia.net/dota2/api.php"
|
||||
HEADERS = {"User-Agent": USER_AGENT, "Accept-Encoding": "gzip"}
|
||||
|
||||
# This script lives in <project_root>/scripts/, migrations live in <project_root>/migrations/
|
||||
MIGRATIONS_DIR = Path(__file__).resolve().parent.parent / "migrations"
|
||||
|
||||
QUERY_BATCH_SIZE = 50 # MediaWiki's default action=query title batch limit
|
||||
GENERAL_RATE_LIMIT_SECONDS = 2 # applies between action=query batches
|
||||
|
||||
|
||||
def fetch_hero_list():
|
||||
"""Get the current hero roster (name + image URL) from the Portal:Heroes grid."""
|
||||
params = {"action": "parse", "page": "Portal:Heroes", "format": "json", "prop": "text"}
|
||||
resp = requests.get(API_BASE, params=params, headers=HEADERS, timeout=30)
|
||||
resp.raise_for_status()
|
||||
html = resp.json()["parse"]["text"]["*"]
|
||||
soup = BeautifulSoup(html, "html.parser")
|
||||
|
||||
heroes = []
|
||||
for card in soup.select("div.heroes-panel__hero-card"):
|
||||
link = card.select_one(".heroes-panel__hero-card__title a")
|
||||
img = card.select_one("img")
|
||||
if not link or not img:
|
||||
continue
|
||||
name = link.get("title") or link.text.strip()
|
||||
heroes.append({"name": name, "image_url": "https://liquipedia.net" + img["src"]})
|
||||
return heroes
|
||||
|
||||
|
||||
def fetch_hero_infoboxes(names):
|
||||
"""Batch-fetch raw wikitext for hero pages via action=query (cheap, not the expensive parse action)."""
|
||||
infoboxes = {}
|
||||
for i in range(0, len(names), QUERY_BATCH_SIZE):
|
||||
batch = names[i:i + QUERY_BATCH_SIZE]
|
||||
params = {
|
||||
"action": "query",
|
||||
"prop": "revisions",
|
||||
"rvslots": "main",
|
||||
"rvprop": "content",
|
||||
"titles": "|".join(batch),
|
||||
"format": "json",
|
||||
}
|
||||
resp = requests.get(API_BASE, params=params, headers=HEADERS, timeout=30)
|
||||
resp.raise_for_status()
|
||||
for page in resp.json()["query"]["pages"].values():
|
||||
title = page.get("title")
|
||||
revisions = page.get("revisions")
|
||||
if not revisions:
|
||||
print(f" WARNING: no content found for '{title}', skipping", file=sys.stderr)
|
||||
continue
|
||||
infoboxes[title] = revisions[0]["slots"]["main"]["*"]
|
||||
if i + QUERY_BATCH_SIZE < len(names):
|
||||
time.sleep(GENERAL_RATE_LIMIT_SECONDS)
|
||||
return infoboxes
|
||||
|
||||
|
||||
def parse_infobox_field(wikitext, field):
|
||||
match = re.search(rf"\|\s*{re.escape(field)}\s*=\s*(.+)", wikitext)
|
||||
return match.group(1).strip() if match else None
|
||||
|
||||
|
||||
def sql_escape(value):
|
||||
return value.replace("'", "''")
|
||||
|
||||
|
||||
def write_migration(rows):
|
||||
timestamp = time.strftime("%Y%m%d%H%M%S")
|
||||
up_path = MIGRATIONS_DIR / f"{timestamp}_seed_heroes.up.sql"
|
||||
down_path = MIGRATIONS_DIR / f"{timestamp}_seed_heroes.down.sql"
|
||||
|
||||
values = ",\n".join(
|
||||
f" ('{sql_escape(r['name'])}', '{r['primary_attribute']}', '{r['attack_type']}', '{sql_escape(r['image_url'])}')"
|
||||
for r in rows
|
||||
)
|
||||
up_sql = f"INSERT INTO heroes (name, primary_attribute, attack_type, image_url) VALUES\n{values};\n"
|
||||
down_sql = "DELETE FROM heroes;\n"
|
||||
|
||||
MIGRATIONS_DIR.mkdir(exist_ok=True)
|
||||
up_path.write_text(up_sql, encoding="utf-8")
|
||||
down_path.write_text(down_sql, encoding="utf-8")
|
||||
|
||||
print(f"\nWrote migration:\n {up_path}\n {down_path}")
|
||||
print("Run `sqlx migrate run` from the project root to apply it.")
|
||||
|
||||
|
||||
def main():
|
||||
if "YOUR_EMAIL_HERE" in USER_AGENT:
|
||||
sys.exit("Edit USER_AGENT at the top of this script with real contact info before running (Liquipedia's API terms require it).")
|
||||
|
||||
print("Fetching current hero roster from Portal:Heroes ...")
|
||||
heroes = fetch_hero_list()
|
||||
print(f" Found {len(heroes)} heroes on the current grid.")
|
||||
|
||||
names = [h["name"] for h in heroes]
|
||||
print("Fetching hero infoboxes in batches ...")
|
||||
infoboxes = fetch_hero_infoboxes(names)
|
||||
|
||||
rows = []
|
||||
skipped = []
|
||||
for hero in heroes:
|
||||
wikitext = infoboxes.get(hero["name"])
|
||||
if not wikitext:
|
||||
skipped.append(hero["name"])
|
||||
continue
|
||||
primary = parse_infobox_field(wikitext, "primary")
|
||||
rangetype = parse_infobox_field(wikitext, "rangetype")
|
||||
if not primary or not rangetype:
|
||||
skipped.append(hero["name"])
|
||||
continue
|
||||
rows.append({
|
||||
"name": hero["name"],
|
||||
"primary_attribute": primary.lower(),
|
||||
"attack_type": rangetype.lower(),
|
||||
"image_url": hero["image_url"],
|
||||
})
|
||||
|
||||
print(f" Parsed {len(rows)} heroes successfully.")
|
||||
if skipped:
|
||||
print(f" Skipped {len(skipped)} heroes (missing infobox data): {', '.join(skipped)}")
|
||||
|
||||
write_migration(rows)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in New Issue
Block a user