feat(DRW-001) initial commit!
This commit is contained in:
@@ -0,0 +1,622 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
Scrapes the current Dota 2 item roster (shop items, active neutral artifacts,
|
||||
and active neutral enchantments) from Liquipedia and writes sqlx migrations
|
||||
that seed `items` / `tags` / `item_tags` and `enchantments` / `enchantment_attributes`.
|
||||
|
||||
Requires:
|
||||
pip install requests beautifulsoup4
|
||||
|
||||
Respects Liquipedia's API Terms of Use (https://liquipedia.net/api-terms-of-use):
|
||||
- action=query requests: max 1 per 2 seconds (used for batched item data)
|
||||
- action=parse requests: max 1 per 30 seconds (used exactly twice: the shop
|
||||
item grid, and the neutral items page)
|
||||
- Requires a descriptive User-Agent with real contact info -- EDIT USER_AGENT BELOW.
|
||||
|
||||
How "is this actually in the game right now" is decided: every item's own
|
||||
infobox carries a `game = Dota 2 Removed` / `game = Dota 2 Unreleased` field
|
||||
when it isn't in normal circulation, and omits the field entirely when it is.
|
||||
That single field is checked for every shop item, neutral artifact, and
|
||||
enchantment -- it's far more reliable than inferring status from which page
|
||||
section something is listed under (which is what earlier versions of this
|
||||
script did, and which doesn't cover shop items at all).
|
||||
|
||||
Note on "purpose" tags: Liquipedia doesn't have a structured "purpose" field
|
||||
for items, so tags are inferred heuristically from the item's stat bonuses
|
||||
and ability descriptions (see TAG_KEYWORDS below). This is a best-effort
|
||||
classifier, not ground truth -- review the generated migration and adjust
|
||||
TAG_KEYWORDS or the resulting SQL if something looks wrong.
|
||||
"""
|
||||
|
||||
import re
|
||||
import sys
|
||||
import time
|
||||
from pathlib import Path
|
||||
|
||||
import requests
|
||||
from bs4 import BeautifulSoup
|
||||
|
||||
# --- EDIT THIS before running: Liquipedia requires real contact info in the User-Agent ---
|
||||
USER_AGENT = "DotaRandomWebItemScraper/1.0 (contact: YOUR_EMAIL_HERE)"
|
||||
|
||||
API_BASE = "https://liquipedia.net/dota2/api.php"
|
||||
HEADERS = {"User-Agent": USER_AGENT, "Accept-Encoding": "gzip"}
|
||||
|
||||
MIGRATIONS_DIR = Path(__file__).resolve().parent.parent / "migrations"
|
||||
|
||||
QUERY_BATCH_SIZE = 50
|
||||
GENERAL_RATE_LIMIT_SECONDS = 2
|
||||
PARSE_RATE_LIMIT_SECONDS = 30
|
||||
|
||||
VALID_ATTRIBUTES = {"strength", "agility", "intelligence", "universal"}
|
||||
|
||||
# Real item pages that only exist as a side-effect of a specific hero
|
||||
# ability or a specific other neutral item (e.g. Ironwood Nut/Bag of Gold
|
||||
# come from particular neutral items, not from anything a random hero would
|
||||
# have) -- not something a "random build" should ever hand out as a normal
|
||||
# rolled item. Curated by hand; extend if more of these turn up.
|
||||
EXCLUDED_ITEM_NAMES = {
|
||||
"Tango (Shared)",
|
||||
"Ironwood Nut",
|
||||
"Bag of Gold",
|
||||
"Vital Toadstool",
|
||||
"Tomo'kan Ringcap",
|
||||
}
|
||||
|
||||
# Keyword -> tag mapping. Matched case-insensitively against a blob built from
|
||||
# each item's infobox stat-bonus field names, active/passive ability names,
|
||||
# ability descriptions, and lore text. An item can (and often will) get
|
||||
# several tags.
|
||||
TAG_KEYWORDS = {
|
||||
"Physical Damage": ["attack damage", "physical damage", "cleave", "critical strike"],
|
||||
"Magic Damage": ["spell amplification", "spell damage", "magical damage", "magic damage"],
|
||||
"Pure Damage": ["pure damage"],
|
||||
"Attack Speed": ["attack speed"],
|
||||
"Survivability": ["evasion", "evade", "damage block", "spell immunity", "invulnerable", "shield", "barrier", "armor"],
|
||||
"Sustain": ["health regeneration", "hp regeneration", "mana regeneration", "lifesteal", "life steal", "heal"],
|
||||
"Mobility": ["movement speed", "blink", "teleport", "dash", "leap"],
|
||||
"Disable": ["stun", "root", "hex", "fear", "sleep", "entangle"],
|
||||
"Silence": ["silence"],
|
||||
"Slow": ["slow"],
|
||||
"Vision": ["true sight", "vision", "invisible", "invisibility"],
|
||||
"Dispel": ["dispel", "purge"],
|
||||
"Aura": ["aura"],
|
||||
"Illusion / Summon": ["illusion", "summon"],
|
||||
"Economy": ["bounty", "gold "],
|
||||
"Spell Resistance": ["magic resistance", "spell resistance"],
|
||||
}
|
||||
|
||||
|
||||
def fetch_shop_candidates():
|
||||
"""Get all shop/purchasable item candidates (name + image) from the Item Grid page."""
|
||||
params = {"action": "parse", "page": "Item_Grid", "format": "json", "prop": "text"}
|
||||
resp = requests.get(API_BASE, params=params, headers=HEADERS, timeout=30)
|
||||
resp.raise_for_status()
|
||||
html = resp.json()["parse"]["text"]["*"]
|
||||
soup = BeautifulSoup(html, "html.parser")
|
||||
|
||||
candidates = []
|
||||
for li in soup.select("li"):
|
||||
a = li.find("a", recursive=False)
|
||||
if not a or not a.get("href", "").startswith("/dota2/"):
|
||||
continue
|
||||
img = a.find("img")
|
||||
if not img:
|
||||
continue
|
||||
title = a.get("title", "").strip()
|
||||
name = re.sub(r"\s*\(\d+\)$", "", title) # strip trailing " (cost)"
|
||||
if not name:
|
||||
continue
|
||||
candidates.append({"name": name, "image_url": "https://liquipedia.net" + img["src"]})
|
||||
return candidates
|
||||
|
||||
|
||||
def fetch_neutral_candidates():
|
||||
"""
|
||||
Get every neutral item/enchantment candidate (name + image) from the
|
||||
Neutral Items page, regardless of section. Active-vs-removed status,
|
||||
tier, and artifact-vs-enchantment category all come from each
|
||||
candidate's own infobox later (see parse_item_fields) rather than from
|
||||
page layout, so we don't need to filter by heading here.
|
||||
"""
|
||||
params = {"action": "parse", "page": "Neutral_Items", "format": "json", "prop": "text"}
|
||||
resp = requests.get(API_BASE, params=params, headers=HEADERS, timeout=30)
|
||||
resp.raise_for_status()
|
||||
html = resp.json()["parse"]["text"]["*"]
|
||||
soup = BeautifulSoup(html, "html.parser")
|
||||
|
||||
candidates = []
|
||||
seen = set()
|
||||
for itemlist in soup.select("div.itemlist"):
|
||||
for a in itemlist.select("a:has(img)"):
|
||||
img = a.find("img")
|
||||
if not img:
|
||||
continue
|
||||
name = a.get("title", "").strip()
|
||||
if not name or name in seen:
|
||||
continue
|
||||
seen.add(name)
|
||||
candidates.append({"name": name, "image_url": "https://liquipedia.net" + img["src"]})
|
||||
return candidates
|
||||
|
||||
|
||||
def fetch_category_item_names():
|
||||
"""The complete, authoritative list of every item page on the wiki.
|
||||
Item_Grid and Neutral_Items are curated listing pages and can simply omit
|
||||
something real (e.g. Block of Cheese isn't on Item_Grid at all, despite
|
||||
being a normal active item) -- this category membership list can't miss
|
||||
anything, since every item page is tagged with it regardless of which
|
||||
curated list remembered to mention it.
|
||||
|
||||
cmnamespace=0 restricts to the main article namespace -- without it,
|
||||
miscategorized non-item pages sneak in too (e.g. Template:Item infobox/doc,
|
||||
the template's own documentation page, is a member of this category despite
|
||||
obviously not being an item)."""
|
||||
params = {
|
||||
"action": "query",
|
||||
"list": "categorymembers",
|
||||
"cmtitle": "Category:Items",
|
||||
"cmlimit": "500",
|
||||
"cmnamespace": "0",
|
||||
"format": "json",
|
||||
}
|
||||
resp = requests.get(API_BASE, params=params, headers=HEADERS, timeout=30)
|
||||
resp.raise_for_status()
|
||||
return [m["title"] for m in resp.json()["query"]["categorymembers"]]
|
||||
|
||||
|
||||
def resolve_image_urls(filenames):
|
||||
"""Batch-resolve wiki image filenames (as found in an infobox's `image`
|
||||
field) to their real hosted URLs. Used only for candidates recovered via
|
||||
fetch_category_item_names(), since Item_Grid/Neutral_Items already give
|
||||
us a usable <img> src directly."""
|
||||
urls = {}
|
||||
unique = list(dict.fromkeys(f for f in filenames if f))
|
||||
for i in range(0, len(unique), QUERY_BATCH_SIZE):
|
||||
batch = unique[i:i + QUERY_BATCH_SIZE]
|
||||
params = {
|
||||
"action": "query",
|
||||
"titles": "|".join(f"File:{f}" for f in batch),
|
||||
"prop": "imageinfo",
|
||||
"iiprop": "url",
|
||||
"format": "json",
|
||||
}
|
||||
resp = requests.get(API_BASE, params=params, headers=HEADERS, timeout=30)
|
||||
resp.raise_for_status()
|
||||
for page in resp.json()["query"]["pages"].values():
|
||||
title = page.get("title", "")
|
||||
imageinfo = page.get("imageinfo")
|
||||
if imageinfo and title.startswith("File:"):
|
||||
urls[title[len("File:"):]] = imageinfo[0]["url"]
|
||||
if i + QUERY_BATCH_SIZE < len(unique):
|
||||
time.sleep(GENERAL_RATE_LIMIT_SECONDS)
|
||||
return urls
|
||||
|
||||
|
||||
def fetch_wikitext(names):
|
||||
"""Batch-fetch raw wikitext for pages via action=query (cheap, not the expensive parse action)."""
|
||||
pages = {}
|
||||
for i in range(0, len(names), QUERY_BATCH_SIZE):
|
||||
batch = names[i:i + QUERY_BATCH_SIZE]
|
||||
params = {
|
||||
"action": "query",
|
||||
"prop": "revisions",
|
||||
"rvslots": "main",
|
||||
"rvprop": "content",
|
||||
"titles": "|".join(batch),
|
||||
"format": "json",
|
||||
}
|
||||
resp = requests.get(API_BASE, params=params, headers=HEADERS, timeout=30)
|
||||
resp.raise_for_status()
|
||||
for page in resp.json()["query"]["pages"].values():
|
||||
title = page.get("title")
|
||||
revisions = page.get("revisions")
|
||||
if not revisions:
|
||||
print(f" WARNING: no content found for '{title}', skipping", file=sys.stderr)
|
||||
continue
|
||||
pages[title] = revisions[0]["slots"]["main"]["*"]
|
||||
if i + QUERY_BATCH_SIZE < len(names):
|
||||
time.sleep(GENERAL_RATE_LIMIT_SECONDS)
|
||||
return pages
|
||||
|
||||
|
||||
def extract_infobox(wikitext):
|
||||
match = re.search(r"\{\{Item infobox([\s\S]*?)\n\}\}", wikitext)
|
||||
return match.group(1) if match else None
|
||||
|
||||
|
||||
INACTIVE_GAME_VALUES = {"dota 2 removed", "dota 2 unreleased"}
|
||||
|
||||
|
||||
def is_active(infobox):
|
||||
"""Most currently-active items omit the `game` field entirely. Genuinely
|
||||
gone items set it to 'Dota 2 Removed' / 'Dota 2 Unreleased'. There's also
|
||||
a 'Dota 2 Hidden' value (seen on e.g. Power Treads, Boots of Travel) that
|
||||
does NOT mean inactive -- treating any non-empty `game` field as inactive
|
||||
wrongly excluded those. Only the known removal/unreleased values count."""
|
||||
match = re.search(r"^\s*\|\s*game\s*=\s*(.+)$", infobox, re.MULTILINE)
|
||||
if not match:
|
||||
return True
|
||||
return match.group(1).strip().lower() not in INACTIVE_GAME_VALUES
|
||||
|
||||
|
||||
def parse_tier(infobox):
|
||||
"""Returns (None, None) for anything that isn't a real, obtainable 1-5
|
||||
tier -- including the literal 'tier = 0' Liquipedia uses on a handful of
|
||||
non-independently-obtainable derived states (e.g. Disgraced Regalia, an
|
||||
automatic broken-item replacement, not something that drops on its own)."""
|
||||
match = re.search(r"^\s*\|\s*tier\s*=\s*(\d+)(?:-(\d+))?", infobox, re.MULTILINE)
|
||||
if not match:
|
||||
return None, None
|
||||
min_tier = int(match.group(1))
|
||||
max_tier = int(match.group(2)) if match.group(2) else min_tier
|
||||
if not (1 <= min_tier <= 5 and 1 <= max_tier <= 5):
|
||||
return None, None
|
||||
return min_tier, max_tier
|
||||
|
||||
|
||||
def parse_type(infobox):
|
||||
match = re.search(r"^\s*\|\s*type\s*=\s*(.+)$", infobox, re.MULTILINE)
|
||||
return match.group(1).strip() if match else ""
|
||||
|
||||
|
||||
def parse_neutral_flag(infobox):
|
||||
match = re.search(r"^\s*\|\s*neutral\s*=\s*(\w+)", infobox, re.MULTILINE)
|
||||
return bool(match and match.group(1).strip().lower() == "true")
|
||||
|
||||
|
||||
def normalize_filename(filename):
|
||||
"""MediaWiki collapses repeated whitespace in file titles server-side
|
||||
(confirmed via the API's own "normalized" field, e.g. Boots of Travel 2's
|
||||
infobox has a literal double space in its filename) -- without matching
|
||||
that here, a lookup keyed by the raw wikitext value silently misses the
|
||||
resolved URL and falls back to the wrong image."""
|
||||
return re.sub(r"\s+", " ", filename).strip()
|
||||
|
||||
|
||||
def parse_image_filename(infobox):
|
||||
match = re.search(r"^\s*\|\s*image\s*=\s*(.+)$", infobox, re.MULTILINE)
|
||||
return normalize_filename(match.group(1)) if match else None
|
||||
|
||||
|
||||
def parse_grade_count(infobox):
|
||||
"""Some shop items are a single purchase that gets upgraded in place
|
||||
(Dagon 1-5, Boots of Travel 1-2) rather than being bought fresh each
|
||||
time. Liquipedia encodes the grade count as N slash-separated parts in
|
||||
the `intern` field (e.g. `item_dagon/item_dagon_2/.../item_dagon_5` for
|
||||
5 grades) -- normal single-grade items just have one part."""
|
||||
match = re.search(r"^\s*\|\s*intern\s*=\s*(.+)$", infobox, re.MULTILINE)
|
||||
if not match:
|
||||
return 1
|
||||
return max(1, len(match.group(1).split("/")))
|
||||
|
||||
|
||||
def parse_all_image_filenames(infobox):
|
||||
"""A multi-grade item numbers its images image1, image2, ... (one per
|
||||
grade, in order); a normal single-grade item just has one plain `image`
|
||||
field. Returns them in grade order."""
|
||||
numbered = sorted(
|
||||
(int(n), normalize_filename(fname)) for n, fname in re.findall(r"^\s*\|\s*image(\d+)\s*=\s*(.+)$", infobox, re.MULTILINE)
|
||||
)
|
||||
if numbered:
|
||||
return [fname for _, fname in numbered]
|
||||
plain = parse_image_filename(infobox)
|
||||
return [plain] if plain else []
|
||||
|
||||
|
||||
def build_shop_item_rows(name, infobox, wikitext, fallback_image_url):
|
||||
"""Returns one row for a normal item, or one row per grade for an
|
||||
upgrade-in-place item like Dagon or Boots of Travel -- grade 1 keeps the
|
||||
base name, later grades are named '{name} {grade}' (matching Liquipedia's
|
||||
own convention, e.g. 'Dagon 2', 'Boots of Travel 2')."""
|
||||
grade_count = parse_grade_count(infobox)
|
||||
tags = infer_tags(infobox, wikitext)
|
||||
|
||||
if grade_count == 1:
|
||||
return [{
|
||||
"name": name, "source": "shop", "min_tier": None, "max_tier": None,
|
||||
"min_grade": 1, "max_grade": 1, "image_url": fallback_image_url, "tags": tags,
|
||||
}]
|
||||
|
||||
filenames = parse_all_image_filenames(infobox)
|
||||
resolved = resolve_image_urls(filenames)
|
||||
rows = []
|
||||
for grade in range(1, grade_count + 1):
|
||||
grade_name = name if grade == 1 else f"{name} {grade}"
|
||||
filename = filenames[grade - 1] if grade - 1 < len(filenames) else None
|
||||
image_url = resolved.get(filename) or fallback_image_url
|
||||
rows.append({
|
||||
"name": grade_name, "source": "shop", "min_tier": None, "max_tier": None,
|
||||
"min_grade": grade, "max_grade": grade, "image_url": image_url, "tags": tags,
|
||||
})
|
||||
return rows
|
||||
|
||||
|
||||
def parse_enchantment_attributes(wikitext, infobox):
|
||||
"""Attribute lock isn't a structured infobox field -- it's stated in the
|
||||
intro prose right before the infobox, e.g. '... for {{Attribute ID|Strength}}
|
||||
heroes only.' Search only that intro portion to avoid false hits from
|
||||
unrelated attribute mentions elsewhere on the page.
|
||||
|
||||
Not every enchantment is attribute-locked -- about a third of them say
|
||||
'... enchantment for all heroes.' instead of naming an attribute. Those
|
||||
are compatible with every attribute, so they're stored with all four
|
||||
rather than zero (zero would make them match nothing at all, since the
|
||||
backend looks enchantments up by a specific requested attribute)."""
|
||||
infobox_pos = wikitext.find("{{Item infobox")
|
||||
intro = wikitext[:infobox_pos] if infobox_pos != -1 else wikitext
|
||||
if re.search(r"for all heroes", intro, re.IGNORECASE):
|
||||
return sorted(VALID_ATTRIBUTES)
|
||||
found = {m.lower() for m in re.findall(r"\{\{Attribute ID\|(\w+)\}\}", intro)}
|
||||
return sorted(found & VALID_ATTRIBUTES)
|
||||
|
||||
|
||||
def build_search_blob(infobox, wikitext):
|
||||
"""Concatenate everything useful for tag-matching into one lowercase blob."""
|
||||
parts = []
|
||||
if infobox:
|
||||
parts += re.findall(r"^\s*\|\s*([a-z0-9 ]+?)\s*=", infobox, re.MULTILINE | re.IGNORECASE)
|
||||
parts += re.findall(r"^\s*\|\s*(?:active|passive)\s*=\s*(.+)$", infobox, re.MULTILINE | re.IGNORECASE)
|
||||
lore_match = re.search(r"^\s*\|\s*lore\s*=\s*(.+)$", infobox, re.MULTILINE | re.IGNORECASE)
|
||||
if lore_match:
|
||||
parts.append(lore_match.group(1))
|
||||
parts += re.findall(r"\bdesc\d*\s*=\s*(.+)", wikitext)
|
||||
return " ".join(parts).lower()
|
||||
|
||||
|
||||
def infer_tags(infobox, wikitext):
|
||||
blob = build_search_blob(infobox, wikitext)
|
||||
return sorted(tag for tag, keywords in TAG_KEYWORDS.items() if any(kw in blob for kw in keywords))
|
||||
|
||||
|
||||
def sql_escape(value):
|
||||
return value.replace("'", "''")
|
||||
|
||||
|
||||
def write_items_migration(rows, item_tags, timestamp):
|
||||
up_path = MIGRATIONS_DIR / f"{timestamp}_seed_items.up.sql"
|
||||
down_path = MIGRATIONS_DIR / f"{timestamp}_seed_items.down.sql"
|
||||
|
||||
item_values = ",\n".join(
|
||||
" ('{}', '{}', {}, {}, {}, {}, '{}')".format(
|
||||
sql_escape(r["name"]),
|
||||
r["source"],
|
||||
r["min_tier"] if r["min_tier"] is not None else "NULL",
|
||||
r["max_tier"] if r["max_tier"] is not None else "NULL",
|
||||
r["min_grade"] if r["min_grade"] is not None else "NULL",
|
||||
r["max_grade"] if r["max_grade"] is not None else "NULL",
|
||||
sql_escape(r["image_url"]),
|
||||
)
|
||||
for r in rows
|
||||
)
|
||||
|
||||
all_tags = sorted({tag for tags in item_tags.values() for tag in tags})
|
||||
tag_values = ",\n".join(f" ('{sql_escape(t)}')" for t in all_tags)
|
||||
|
||||
pair_values = ",\n".join(
|
||||
f" ('{sql_escape(name)}', '{sql_escape(tag)}')"
|
||||
for name, tags in item_tags.items()
|
||||
for tag in tags
|
||||
)
|
||||
|
||||
# DELETE-first so this migration is safe to apply regardless of whether
|
||||
# an earlier seed_items run's data is still sitting in the table --
|
||||
# useful since this script tends to get rerun a lot during development.
|
||||
up_sql = (
|
||||
"DELETE FROM item_tags;\nDELETE FROM tags;\nDELETE FROM items;\n\n"
|
||||
f"INSERT INTO items (name, source, min_tier, max_tier, min_grade, max_grade, image_url) VALUES\n{item_values};\n\n"
|
||||
f"INSERT INTO tags (name) VALUES\n{tag_values};\n\n"
|
||||
"INSERT INTO item_tags (item_id, tag_id)\n"
|
||||
"SELECT i.id, t.id FROM (VALUES\n"
|
||||
f"{pair_values}\n"
|
||||
") AS pairs(item_name, tag_name)\n"
|
||||
"JOIN items i ON i.name = pairs.item_name\n"
|
||||
"JOIN tags t ON t.name = pairs.tag_name;\n"
|
||||
)
|
||||
down_sql = "DELETE FROM item_tags;\nDELETE FROM tags;\nDELETE FROM items;\n"
|
||||
|
||||
MIGRATIONS_DIR.mkdir(exist_ok=True)
|
||||
up_path.write_text(up_sql, encoding="utf-8")
|
||||
down_path.write_text(down_sql, encoding="utf-8")
|
||||
print(f"\nWrote items migration:\n {up_path}\n {down_path}")
|
||||
|
||||
|
||||
def write_enchantments_migration(rows, timestamp):
|
||||
up_path = MIGRATIONS_DIR / f"{timestamp}_seed_enchantments.up.sql"
|
||||
down_path = MIGRATIONS_DIR / f"{timestamp}_seed_enchantments.down.sql"
|
||||
|
||||
enchant_values = ",\n".join(
|
||||
" ('{}', {}, {}, '{}')".format(sql_escape(r["name"]), r["min_tier"], r["max_tier"], sql_escape(r["image_url"]))
|
||||
for r in rows
|
||||
)
|
||||
pair_values = ",\n".join(
|
||||
f" ('{sql_escape(r['name'])}', '{attr}')" for r in rows for attr in r["attributes"]
|
||||
)
|
||||
|
||||
up_sql = (
|
||||
"DELETE FROM enchantment_attributes;\nDELETE FROM enchantments;\n\n"
|
||||
f"INSERT INTO enchantments (name, min_tier, max_tier, image_url) VALUES\n{enchant_values};\n\n"
|
||||
"INSERT INTO enchantment_attributes (enchantment_id, attribute)\n"
|
||||
"SELECT e.id, pairs.attribute::primary_attribute FROM (VALUES\n"
|
||||
f"{pair_values}\n"
|
||||
") AS pairs(enchantment_name, attribute)\n"
|
||||
"JOIN enchantments e ON e.name = pairs.enchantment_name;\n"
|
||||
)
|
||||
down_sql = "DELETE FROM enchantment_attributes;\nDELETE FROM enchantments;\n"
|
||||
|
||||
MIGRATIONS_DIR.mkdir(exist_ok=True)
|
||||
up_path.write_text(up_sql, encoding="utf-8")
|
||||
down_path.write_text(down_sql, encoding="utf-8")
|
||||
print(f"Wrote enchantments migration:\n {up_path}\n {down_path}")
|
||||
|
||||
|
||||
def main():
|
||||
if "YOUR_EMAIL_HERE" in USER_AGENT:
|
||||
sys.exit("Edit USER_AGENT at the top of this script with real contact info before running (Liquipedia's API terms require it).")
|
||||
|
||||
print("Fetching shop item candidates from Item Grid ...")
|
||||
shop_candidates = fetch_shop_candidates()
|
||||
print(f" Found {len(shop_candidates)} shop item candidates.")
|
||||
|
||||
time.sleep(PARSE_RATE_LIMIT_SECONDS) # second action=parse call, must respect the 30s limit
|
||||
|
||||
print("Fetching neutral item/enchantment candidates ...")
|
||||
neutral_candidates = fetch_neutral_candidates()
|
||||
print(f" Found {len(neutral_candidates)} neutral candidates (active + removed + unreleased).")
|
||||
|
||||
all_candidates = shop_candidates + neutral_candidates
|
||||
names = [c["name"] for c in all_candidates]
|
||||
|
||||
print("Fetching complete item category listing to catch anything Item Grid / Neutral Items missed ...")
|
||||
category_names = fetch_category_item_names()
|
||||
known_names = {c["name"] for c in all_candidates}
|
||||
recovered_names = [n for n in category_names if n not in known_names]
|
||||
if recovered_names:
|
||||
print(f" Found {len(recovered_names)} item(s) missing from the curated listings: {', '.join(recovered_names)}")
|
||||
|
||||
print("Fetching wikitext in batches (for status/tier/tags/attribute checks) ...")
|
||||
wikitext_by_name = fetch_wikitext(names + recovered_names)
|
||||
|
||||
if recovered_names:
|
||||
print("Resolving image URLs for recovered items ...")
|
||||
recovered_filenames = {
|
||||
name: parse_image_filename(extract_infobox(wikitext_by_name[name]) or "")
|
||||
for name in recovered_names
|
||||
if wikitext_by_name.get(name)
|
||||
}
|
||||
recovered_image_urls = resolve_image_urls(recovered_filenames.values())
|
||||
|
||||
item_rows = []
|
||||
item_tags = {}
|
||||
enchantment_rows = []
|
||||
skipped_missing = []
|
||||
skipped_inactive = []
|
||||
skipped_excluded = []
|
||||
|
||||
for candidate in shop_candidates:
|
||||
if candidate["name"] in EXCLUDED_ITEM_NAMES:
|
||||
skipped_excluded.append(candidate["name"])
|
||||
continue
|
||||
wikitext = wikitext_by_name.get(candidate["name"])
|
||||
if not wikitext:
|
||||
skipped_missing.append(candidate["name"])
|
||||
continue
|
||||
infobox = extract_infobox(wikitext)
|
||||
if not infobox or not is_active(infobox):
|
||||
skipped_inactive.append(candidate["name"])
|
||||
continue
|
||||
for row in build_shop_item_rows(candidate["name"], infobox, wikitext, candidate["image_url"]):
|
||||
item_rows.append({k: v for k, v in row.items() if k != "tags"})
|
||||
item_tags[row["name"]] = row["tags"]
|
||||
|
||||
for candidate in neutral_candidates:
|
||||
if candidate["name"] in EXCLUDED_ITEM_NAMES:
|
||||
skipped_excluded.append(candidate["name"])
|
||||
continue
|
||||
wikitext = wikitext_by_name.get(candidate["name"])
|
||||
if not wikitext:
|
||||
skipped_missing.append(candidate["name"])
|
||||
continue
|
||||
infobox = extract_infobox(wikitext)
|
||||
if not infobox or not is_active(infobox):
|
||||
skipped_inactive.append(candidate["name"])
|
||||
continue
|
||||
|
||||
min_tier, max_tier = parse_tier(infobox)
|
||||
if min_tier is None:
|
||||
skipped_missing.append(candidate["name"])
|
||||
continue
|
||||
|
||||
if "enchantment" in parse_type(infobox).lower():
|
||||
attributes = parse_enchantment_attributes(wikitext, infobox)
|
||||
enchantment_rows.append({
|
||||
"name": candidate["name"],
|
||||
"min_tier": min_tier,
|
||||
"max_tier": max_tier,
|
||||
"image_url": candidate["image_url"],
|
||||
"attributes": attributes,
|
||||
})
|
||||
else:
|
||||
item_rows.append({
|
||||
"name": candidate["name"],
|
||||
"source": "neutral",
|
||||
"min_tier": min_tier,
|
||||
"max_tier": max_tier,
|
||||
"min_grade": None,
|
||||
"max_grade": None,
|
||||
"image_url": candidate["image_url"],
|
||||
})
|
||||
item_tags[candidate["name"]] = infer_tags(infobox, wikitext)
|
||||
|
||||
for name in recovered_names:
|
||||
if name in EXCLUDED_ITEM_NAMES:
|
||||
skipped_excluded.append(name)
|
||||
continue
|
||||
wikitext = wikitext_by_name.get(name)
|
||||
if not wikitext:
|
||||
skipped_missing.append(name)
|
||||
continue
|
||||
infobox = extract_infobox(wikitext)
|
||||
if not infobox or not is_active(infobox):
|
||||
skipped_inactive.append(name)
|
||||
continue
|
||||
|
||||
image_url = recovered_image_urls.get(recovered_filenames.get(name))
|
||||
if not image_url:
|
||||
skipped_missing.append(name)
|
||||
continue
|
||||
|
||||
if parse_neutral_flag(infobox):
|
||||
min_tier, max_tier = parse_tier(infobox)
|
||||
if min_tier is None:
|
||||
skipped_missing.append(name)
|
||||
continue
|
||||
if "enchantment" in parse_type(infobox).lower():
|
||||
enchantment_rows.append({
|
||||
"name": name,
|
||||
"min_tier": min_tier,
|
||||
"max_tier": max_tier,
|
||||
"image_url": image_url,
|
||||
"attributes": parse_enchantment_attributes(wikitext, infobox),
|
||||
})
|
||||
else:
|
||||
item_rows.append({
|
||||
"name": name, "source": "neutral", "min_tier": min_tier, "max_tier": max_tier,
|
||||
"min_grade": None, "max_grade": None, "image_url": image_url,
|
||||
})
|
||||
item_tags[name] = infer_tags(infobox, wikitext)
|
||||
else:
|
||||
for row in build_shop_item_rows(name, infobox, wikitext, image_url):
|
||||
item_rows.append({k: v for k, v in row.items() if k != "tags"})
|
||||
item_tags[row["name"]] = row["tags"]
|
||||
|
||||
print(f"\n {len(item_rows)} items and {len(enchantment_rows)} enchantments are currently active.")
|
||||
if skipped_excluded:
|
||||
print(f" Deliberately excluded {len(skipped_excluded)} hero/item-specific entries: {', '.join(skipped_excluded)}")
|
||||
if skipped_inactive:
|
||||
print(f" Excluded {len(skipped_inactive)} removed/unreleased entries: {', '.join(skipped_inactive)}")
|
||||
if skipped_missing:
|
||||
print(f" Skipped {len(skipped_missing)} entries with no usable data: {', '.join(skipped_missing)}")
|
||||
|
||||
untagged = [name for name, tags in item_tags.items() if not tags]
|
||||
if untagged:
|
||||
print(f" {len(untagged)} items got zero inferred tags (may need a manual look): {', '.join(untagged)}")
|
||||
|
||||
no_attribute = [r["name"] for r in enchantment_rows if not r["attributes"]]
|
||||
if no_attribute:
|
||||
print(f" {len(no_attribute)} enchantments got no attribute lock detected (may need a manual look): {', '.join(no_attribute)}")
|
||||
|
||||
# sqlx uses this timestamp as the migration's primary key, so the two
|
||||
# migrations below must not share one -- offsetting by a second
|
||||
# guarantees uniqueness regardless of how fast this function runs.
|
||||
now = time.time()
|
||||
items_timestamp = time.strftime("%Y%m%d%H%M%S", time.localtime(now))
|
||||
enchantments_timestamp = time.strftime("%Y%m%d%H%M%S", time.localtime(now + 1))
|
||||
|
||||
write_items_migration(item_rows, item_tags, items_timestamp)
|
||||
write_enchantments_migration(enchantment_rows, enchantments_timestamp)
|
||||
print("\nRun `sqlx migrate run` from the project root to apply both.")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in New Issue
Block a user