diff --git a/.gitignore b/.gitignore
index 7edce68..7158012 100644
--- a/.gitignore
+++ b/.gitignore
@@ -39,4 +39,4 @@ Thumbs.db
*.swp
# Cached libretro directory listings (regenerated on demand)
-tools/cover-art/.cache/
+tools/library/.cache/
diff --git a/README.md b/README.md
index 6453a85..e8e723b 100644
--- a/README.md
+++ b/README.md
@@ -77,7 +77,7 @@ cd backend && dotnet build # 0 warnings expected
### Cover art
-`tools/cover-art/fetch_art.py` fills in box art, pushing each image through the
+`tools/library/fetch_art.py` fills in box art, pushing each image through the
app's own `POST /api/images` so it gets the same validation and WebP re-encoding
as a manual upload. Standard library only — no virtualenv, and neither source
needs an account.
@@ -95,7 +95,7 @@ It tries two sources in order:
one per second and cached; the API returns 429 if pushed harder.
```bash
-cd tools/cover-art
+cd tools/library
python3 fetch_art.py --password '...' --dry-run # report matches, change nothing
python3 fetch_art.py --password '...' # download and attach
python3 fetch_art.py --password '...' --overwrite # also replace existing art
@@ -123,6 +123,52 @@ Fix the system field and re-run with `--overwrite` to correct them.
Art is publisher copyright. Fetching it for a private collection is ordinary
practice for library software; redistributing it is a different question.
+### Metadata enrichment
+
+`tools/library/enrich_metadata.py` fills developer, publisher, year and
+description from the same Wikipedia articles the cover fetcher locates. Only
+empty fields are touched unless `--overwrite` is passed — anything typed by hand
+outranks anything derived here.
+
+```bash
+cd tools/library
+python3 enrich_metadata.py --password '...' --dry-run
+python3 enrich_metadata.py --password '...'
+python3 enrich_metadata.py --password '...' --fields developer,publisher
+```
+
+Coverage went from this to this:
+
+| Field | Before | After |
+| --- | --- | --- |
+| year | 79% | 96% |
+| developer | 3.8% | 95% |
+| publisher | 2.9% | 96% |
+| description | 0% | 96% |
+
+Developer, publisher and year are facts, written verbatim. Descriptions are
+article summaries, which are CC BY-SA, so each is stored with an attribution
+line naming the source article.
+
+Parsing an infobox is messier than it looks, and the guards matter:
+
+- **Series articles are rejected.** A substring test for `Infobox video game`
+ also matches `Infobox video game series`, which resolved Banjo-Kazooie to the
+ series overview instead of the 1998 game.
+- **Year is only filled when the article covers that platform.** An article
+ spans every release, and its date block leads with the original — so a DS
+ port would otherwise be dated to the SNES original.
+- **A search hit that neither covers the platform nor closely matches the title
+ is discarded.** Our "Dragon Ball Z Budokai" surfaces "Dragon Ball Z: Shin
+ Budokai", a different game on a different console. A blank field beats a
+ confidently wrong one.
+- Platform headings (`'''PlayStation'''`), region codes (`JP`, `NA`), trailing
+ platform annotations (`Rare (N64)`) and named template parameters (`title=`)
+ are all stripped, since each one otherwise reads as the value itself.
+
+Four games have no usable article: a typo'd title (`Brett Hull Hocky 95`),
+`Dragon Ball Z Budokai`, and two niche releases.
+
### Database changes
```bash
diff --git a/tools/cover-art/__pycache__/fetch_art.cpython-314.pyc b/tools/cover-art/__pycache__/fetch_art.cpython-314.pyc
deleted file mode 100644
index 25af11d..0000000
Binary files a/tools/cover-art/__pycache__/fetch_art.cpython-314.pyc and /dev/null differ
diff --git a/tools/library/__pycache__/enrich_metadata.cpython-314.pyc b/tools/library/__pycache__/enrich_metadata.cpython-314.pyc
new file mode 100644
index 0000000..71da68e
Binary files /dev/null and b/tools/library/__pycache__/enrich_metadata.cpython-314.pyc differ
diff --git a/tools/library/__pycache__/fetch_art.cpython-314.pyc b/tools/library/__pycache__/fetch_art.cpython-314.pyc
new file mode 100644
index 0000000..879d252
Binary files /dev/null and b/tools/library/__pycache__/fetch_art.cpython-314.pyc differ
diff --git a/tools/library/enrich_metadata.py b/tools/library/enrich_metadata.py
new file mode 100644
index 0000000..2151e22
--- /dev/null
+++ b/tools/library/enrich_metadata.py
@@ -0,0 +1,436 @@
+#!/usr/bin/env python3
+"""Fill in developer, publisher, year and description from Wikipedia.
+
+The library's own data is thin in exactly these columns — the 2018 dump carried
+titles, systems and genres, but almost no developer or publisher and no
+descriptions at all. This reads the values off the same English Wikipedia
+articles the cover fetcher already locates.
+
+Developer, publisher and release year are facts and are written verbatim.
+Descriptions come from the article summary, which is CC BY-SA text, so each one
+is stored with an attribution line naming the source article.
+
+Only empty fields are filled unless --overwrite is given: anything typed by hand
+outranks anything guessed here.
+
+ python3 enrich_metadata.py --password '...' --dry-run
+ python3 enrich_metadata.py --password '...'
+"""
+
+from __future__ import annotations
+
+import argparse
+import difflib
+import json
+import re
+import sys
+import urllib.parse
+from pathlib import Path
+
+# The cover fetcher already owns the throttled, cached Wikipedia client.
+from fetch_art import (
+ WIKI_UA, api_json, http, is_game_article, normalise, wiki_api, wiki_article,
+)
+
+FIELDS = ("developer", "publisher", "year", "description")
+
+# Infobox values for multi-platform games are frequently grouped under a bold
+# platform heading ("'''PlayStation'''{{vgrelease|JP|Square|...}}") or annotated
+# with the platform in parentheses ("Rare (N64)
Nintendo (DS)"). Both would
+# otherwise be read as the developer or publisher itself.
+PLATFORM_WORDS = (
+ "nes", "snes", "super nes", "super nintendo", "nintendo entertainment system",
+ "n64", "nintendo 64", "gamecube", "gc", "wii", "wii u", "switch",
+ "nintendo switch", "game boy", "game boy color", "game boy advance", "gb",
+ "gbc", "gba", "nintendo ds", "ds", "3ds", "nintendo 3ds",
+ "playstation", "ps1", "ps2", "ps3", "ps4", "ps5", "playstation 2",
+ "playstation 3", "playstation 4", "playstation 5", "psp",
+ "playstation portable", "playstation vita", "vita",
+ "xbox", "xbox 360", "xbox one", "xbox series x", "x360",
+ "windows", "win", "pc", "ms-dos", "dos", "mac", "macos", "osx", "linux",
+ "ios", "android", "arcade", "arc", "genesis", "mega drive", "md",
+ "dreamcast", "saturn", "amiga", "psx", "xbla", "wiiu", "ngc", "3do",
+ "steam", "microsoft windows",
+)
+
+# Whether an article covers the platform a row claims. Used to keep a port's
+# entry from inheriting the original release's year.
+PLATFORM_PATTERNS = {
+ "NES": r"Nintendo Entertainment System|\bNES\b",
+ "SNES": r"Super Nintendo|Super NES|\bSNES\b",
+ "N64": r"Nintendo 64",
+ "GB": r"Game Boy(?! Advance)", # plain Game Boy or Game Boy Color
+ "GBA": r"Game Boy Advance",
+ "DS": r"Nintendo DS",
+ "GC": r"GameCube",
+ "WII": r"\bWii\b(?! ?U)",
+ "PS1": r"PlayStation(?!\s*(?:2|3|4|5|Portable|Vita))",
+ "PS2": r"PlayStation 2",
+ "PSP": r"PlayStation Portable|\bPSP\b",
+ "360": r"Xbox 360",
+}
+
+
+# Release templates tag each entry with a region, so those codes sit between the
+# heading and the value we actually want.
+REGION_CODES = {
+ "jp", "na", "eu", "au", "ww", "uk", "us", "ca", "kr", "cn", "br", "ru", "in",
+ "sea", "int", "pal", "ntsc", "row", "as", "hk", "tw", "mx", "nz", "fr", "de",
+ "es", "it", "nl", "se", "no", "dk", "fi", "pt", "pl",
+}
+
+DATE_LIKE = re.compile(
+ r"^\s*(?:\d{1,2}\s+)?(?:January|February|March|April|May|June|July|August|"
+ r"September|October|November|December)\b|^\s*\d{1,2}[/-]\d{1,2}[/-]\d{2,4}\s*$"
+ r"|^\s*(?:19|20)\d{2}\s*$",
+ re.I,
+)
+
+
+def is_platform_label(text: str) -> bool:
+ return text.strip().lower().strip(":;,") in PLATFORM_WORDS
+
+
+def is_noise_entry(text: str) -> bool:
+ """Region tags, dates and platform headings are scaffolding, not values."""
+ stripped = text.strip().lower().strip(":;,")
+ return (
+ not stripped
+ or stripped in REGION_CODES
+ or is_platform_label(stripped)
+ or bool(DATE_LIKE.match(text))
+ )
+
+
+def article_covers_system(wikitext: str, system: str | None) -> bool:
+ """True when the article's platform list mentions the row's system."""
+ if not system:
+ return False
+ pattern = PLATFORM_PATTERNS.get(system.upper())
+ if not pattern:
+ return False
+
+ platforms = infobox_field(wikitext, "platforms") or infobox_field(wikitext, "platform") or ""
+ return re.search(pattern, platforms, flags=re.I) is not None
+
+
+# --------------------------------------------------------------------------
+# Article resolution
+# --------------------------------------------------------------------------
+
+def resolve_article(game: dict, cache_dir: Path) -> tuple[str, str] | None:
+ """Find the article about this specific game.
+
+ Direct title guesses first, then a search. The search matters for titles
+ whose bare name is the series overview rather than a game — "Mortal Kombat"
+ and "SimCity" both are, with the games at "(1992 video game)" and
+ "(1989 video game)" — and for small punctuation differences, such as our
+ "Dragon Ball Z Budokai" against the catalogued "Dragon Ball Z: Budokai".
+ """
+ title, year, system = game["title"], game.get("year"), game.get("system")
+
+ attempts = [title, f"{title} (video game)"]
+ if year:
+ attempts.append(f"{title} ({year} video game)")
+
+ for candidate in attempts:
+ found = wiki_article(candidate, cache_dir)
+ if found and is_game_article(found[1]):
+ return found
+
+ try:
+ results = wiki_api(list="search", srsearch=f"{title} video game", srlimit=5)
+ titles = [r["title"] for r in results.get("query", {}).get("search", [])]
+ except Exception: # noqa: BLE001
+ return None
+
+ target = normalise(title)
+ scored: list[tuple[bool, float, tuple[str, str]]] = []
+
+ for candidate in titles:
+ found = wiki_article(candidate, cache_dir)
+ if not found or not is_game_article(found[1]):
+ continue
+
+ # Strip the disambiguator before comparing: "Mortal Kombat (1992 video
+ # game)" should read as "Mortal Kombat".
+ bare = re.sub(r"\s*\([^)]*\)\s*$", "", found[0])
+ similarity = difflib.SequenceMatcher(None, target, normalise(bare)).ratio()
+ scored.append((article_covers_system(found[1], system), similarity, found))
+
+ if not scored:
+ return None
+
+ scored.sort(key=lambda s: (s[0], s[1]), reverse=True)
+ covers, similarity, article = scored[0]
+
+ # A search hit that neither covers the platform nor closely matches the name
+ # is more likely a sibling in the series than this game — our "Dragon Ball Z
+ # Budokai" surfaces "Dragon Ball Z: Shin Budokai", a different game on a
+ # different console. A blank field beats a confidently wrong one.
+ if not covers and similarity < 0.90:
+ return None
+
+ return article
+
+
+# --------------------------------------------------------------------------
+# Wikitext cleaning
+# --------------------------------------------------------------------------
+
+def clean_value(value: str) -> str:
+ """Reduce an infobox field to plain text.
+
+ Infobox values are wikitext, so a developer might arrive as
+ ``[[Rare (company)|Rare]]`` or as a ``{{ubl|...}}`` list with references
+ attached. This unwraps the common forms and keeps the first entry.
+ """
+ text = value
+
+ # Footnotes and comments carry nothing we want.
+ text = re.sub(r"[]*/>", "", text)
+ text = re.sub(r"", "", text, flags=re.S | re.I)
+ text = re.sub(r"", "", text, flags=re.S)
+
+ # List and release templates carry the value we want as their arguments —
+ # {{vgrelease|JP|Square|NA|Square Electronic Arts}} holds the publisher —
+ # so keep the payload and drop only the wrapper.
+ text = re.sub(
+ r"\{\{\s*(?:ubl|unbulleted list|plainlist|flatlist|hlist|"
+ r"vgrelease|vgr|video game release|collapsible list|nobold|nowrap)\s*\|",
+ "|", text, flags=re.I)
+ # Anything still wrapped in braces is scaffolding for these fields.
+ text = re.sub(r"\{\{[^{}]*\}\}", "|", text)
+ text = text.replace("{{", "|").replace("}}", "|")
+
+ # [[Target|Label]] -> Label, [[Target]] -> Target
+ text = re.sub(r"\[\[([^\]|]+)\|([^\]]+)\]\]", r"\2", text)
+ text = re.sub(r"\[\[([^\]]+)\]\]", r"\1", text)
+
+ text = re.sub(r"]
", "|", text, flags=re.I)
+ text = re.sub(r"<[^>]+>", "", text)
+ text = text.replace("'''", "").replace("''", "")
+
+ # Take the first real entry of whatever list survived, skipping any leading
+ # platform heading that was grouping the list.
+ parts = [p.strip(" *\n\t") for p in re.split(r"[|\n]", text)]
+ parts = [re.sub(r"\s+", " ", p).strip(" ,;") for p in parts if p.strip(" *\n\t")]
+
+ # A named template parameter leaks through unwrapping as "title=Foo";
+ # keep the value, drop the key.
+ parts = [re.sub(r"^[a-z_][a-z0-9_]*\s*=\s*", "", p, flags=re.I) for p in parts]
+ parts = [p for p in parts if p]
+
+ first = next((p for p in parts if not is_noise_entry(p)), "")
+
+ # Drop a trailing platform annotation: "Rare (N64)" -> "Rare".
+ match = re.match(r"^(.*?)\s*\(([^()]*)\)$", first)
+ if match and is_platform_label(match.group(2)):
+ first = match.group(1).strip()
+
+ return first.strip(" ,;")
+
+
+def infobox_field(wikitext: str, field: str) -> str | None:
+ """Raw value of one infobox field, brace- and bracket-aware."""
+ match = re.search(rf"^\s*\|\s*{field}\s*=", wikitext, flags=re.M | re.I)
+ if not match:
+ return None
+
+ index = match.end()
+ depth_brace = depth_bracket = 0
+ out: list[str] = []
+
+ while index < len(wikitext):
+ two = wikitext[index:index + 2]
+ if two == "{{":
+ depth_brace += 1
+ elif two == "}}":
+ depth_brace -= 1
+ elif two == "[[":
+ depth_bracket += 1
+ elif two == "]]":
+ depth_bracket -= 1
+
+ char = wikitext[index]
+ # A pipe or newline-pipe at depth zero ends this field.
+ if depth_brace <= 0 and depth_bracket <= 0 and char == "|":
+ break
+ if depth_brace <= 0 and depth_bracket <= 0 and char == "\n" and \
+ re.match(r"\s*[|}]", wikitext[index:index + 3]):
+ break
+
+ out.append(char)
+ index += 1
+
+ return "".join(out).strip() or None
+
+
+def extract_year(value: str) -> str | None:
+ """First plausible release year in a {{Video game release}} block."""
+ years = re.findall(r"\b(19[5-9]\d|20[0-4]\d)\b", value)
+ return years[0] if years else None
+
+
+# --------------------------------------------------------------------------
+# Summary
+# --------------------------------------------------------------------------
+
+def wiki_summary(page_title: str) -> str | None:
+ """Short article summary via the REST endpoint, which is built for reuse."""
+ quoted = urllib.parse.quote(page_title.replace(" ", "_"), safe="")
+ raw = http(f"https://en.wikipedia.org/api/rest_v1/page/summary/{quoted}",
+ headers={"User-Agent": WIKI_UA})
+ extract = (json.loads(raw) or {}).get("extract")
+ if not extract:
+ return None
+
+ text = re.sub(r"\s+", " ", extract).strip()
+ # CC BY-SA requires attribution, so it travels with the text.
+ return f"{text}\n\nSource: Wikipedia — {page_title}"
+
+
+# --------------------------------------------------------------------------
+# Main
+# --------------------------------------------------------------------------
+
+def main() -> int:
+ parser = argparse.ArgumentParser(
+ description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
+ parser.add_argument("--api", default="http://localhost:8080")
+ parser.add_argument("--user", default="ckoch")
+ parser.add_argument("--password", required=True)
+ parser.add_argument("--dry-run", action="store_true")
+ parser.add_argument("--overwrite", action="store_true",
+ help="replace fields that already have a value")
+ parser.add_argument("--fields", default=",".join(FIELDS),
+ help=f"comma-separated subset of: {','.join(FIELDS)}")
+ parser.add_argument("--cache", default=str(Path(__file__).parent / ".cache"))
+ args = parser.parse_args()
+
+ wanted = [f.strip() for f in args.fields.split(",") if f.strip() in FIELDS]
+ if not wanted:
+ print("No valid fields requested.", file=sys.stderr)
+ return 2
+
+ base = args.api.rstrip("/")
+ cache_dir = Path(args.cache)
+
+ print("Signing in…")
+ auth = api_json(base, "/api/auth/login",
+ data={"userName": args.user, "password": args.password})
+ token = auth["token"]
+
+ print("Fetching library…")
+ games: list[dict] = []
+ page = 1
+ while True:
+ result = api_json(base, f"/api/games?page={page}&pageSize=100", token)
+ games.extend(result["items"])
+ if page >= result["totalPages"] or not result["items"]:
+ break
+ page += 1
+ print(f" {len(games)} games\n")
+
+ filled = {field: 0 for field in wanted}
+ touched = failed = 0
+ no_article: list[dict] = []
+ skipped_year: list[dict] = []
+
+ for game in sorted(games, key=lambda g: g["title"].lower()):
+ missing = [f for f in wanted if args.overwrite or not game.get(f)]
+ if not missing:
+ continue
+
+ try:
+ article = resolve_article(game, cache_dir)
+ except Exception as exc: # noqa: BLE001
+ print(f" ! {game['title'][:44]:46} lookup failed: {exc}")
+ article = None
+
+ if not article:
+ no_article.append(game)
+ continue
+
+ page_title, wikitext = article
+ updates: dict[str, str] = {}
+
+ for field in missing:
+ if field == "description":
+ try:
+ summary = wiki_summary(page_title)
+ except Exception: # noqa: BLE001
+ summary = None
+ if summary:
+ updates["description"] = summary
+ continue
+
+ if field == "year":
+ # An article covers every platform a game shipped on, and its
+ # release block leads with the original. Taking that year for a
+ # later port would date a DS re-release to the SNES original, so
+ # only fill it when the article actually covers this platform.
+ if not article_covers_system(wikitext, game.get("system")):
+ skipped_year.append(game)
+ continue
+ raw = infobox_field(wikitext, "released")
+ value = extract_year(raw) if raw else None
+ else:
+ raw = infobox_field(wikitext, field)
+ value = clean_value(raw) if raw else None
+
+ if value:
+ updates[field] = value
+
+ if not updates:
+ continue
+
+ summary_line = ", ".join(
+ f"{k}={v[:26]}…" if len(v) > 26 else f"{k}={v}"
+ for k, v in updates.items() if k != "description")
+ if "description" in updates:
+ summary_line = (summary_line + ", " if summary_line else "") + "description"
+
+ print(f" + {game['system'] or '-':4} {game['title'][:40]:42} {summary_line[:60]}")
+
+ for field in updates:
+ filled[field] += 1
+ touched += 1
+
+ if args.dry_run:
+ continue
+
+ payload = {k: game.get(k) for k in (
+ "title", "system", "genre", "year", "developer", "publisher",
+ "art", "description", "own", "dumped", "played", "finished")}
+ payload.update(updates)
+
+ try:
+ api_json(base, f"/api/games/{game['id']}", token, data=payload, method="PUT")
+ except Exception as exc: # noqa: BLE001
+ print(f" -> FAILED: {exc}")
+ failed += 1
+ touched -= 1
+
+ print("\n" + "=" * 72)
+ verb = "would update" if args.dry_run else "updated"
+ print(f"{verb}: {touched} games failed: {failed}")
+ for field in wanted:
+ print(f" {field:12} {filled[field]:3}")
+
+ if skipped_year:
+ print(f"\nYear left alone — the article does not cover this platform, so its "
+ f"release date belongs to a different version ({len(skipped_year)}):")
+ for game in skipped_year[:15]:
+ print(f" {game['system'] or '-':4} {game['title']}")
+
+ if no_article:
+ print(f"\nNo Wikipedia article found ({len(no_article)}):")
+ for game in no_article[:20]:
+ print(f" {game['system'] or '-':4} {game['title']}")
+
+ return 0
+
+
+if __name__ == "__main__":
+ sys.exit(main())
diff --git a/tools/cover-art/fetch_art.py b/tools/library/fetch_art.py
similarity index 96%
rename from tools/cover-art/fetch_art.py
rename to tools/library/fetch_art.py
index 5703524..363d5e2 100644
--- a/tools/cover-art/fetch_art.py
+++ b/tools/library/fetch_art.py
@@ -1,10 +1,13 @@
#!/usr/bin/env python3
"""Fetch box art for the library and attach it to each game.
-Art comes from libretro-thumbnails (https://thumbnails.libretro.com), a
-community archive of boxart named to the No-Intro / Redump conventions. It needs
-no API key, but covers retro consoles only — Xbox 360 has no thumbnail set, so
-those titles are reported as unsupported rather than mismatched.
+Two sources, tried in order, neither needing an account:
+
+ 1. libretro-thumbnails (https://thumbnails.libretro.com) — scanned retail
+ boxes named to the No-Intro / Redump conventions. Best art where it has
+ any, but its coverage is the retro consoles.
+ 2. English Wikipedia — a cover on essentially every notable game article,
+ which is what fills the Xbox 360 shelf.
Images are pushed through the app's own POST /api/images endpoint, so they get
the same validation, WebP re-encoding and per-user filing as a manual upload.
@@ -33,7 +36,6 @@ from pathlib import Path
THUMBNAIL_HOST = "https://thumbnails.libretro.com"
# Our system codes to libretro's DAT-derived directory names.
-# Xbox 360 is deliberately absent: libretro-thumbnails has no set for it.
# Several directories may back one system code. Our "GB" covers both Game Boy
# and Game Boy Color titles — Pokemon Trading Card Game, for instance, is only in
# the Color set — so both are searched and the better match wins.
@@ -230,6 +232,16 @@ def wiki_api(**params):
return json.loads(raw)
+def is_game_article(wikitext: str) -> bool:
+ """True for an article about a single game.
+
+ A substring test for "Infobox video game" also matches "Infobox video game
+ series", which would resolve Banjo-Kazooie to the series overview rather
+ than the 1998 game — and give the wrong cover and the wrong summary.
+ """
+ return re.search(r"\{\{\s*Infobox video game\b(?!\s*series)", wikitext, re.I) is not None
+
+
def wiki_article(title: str, cache_dir: Path) -> tuple[str, str] | None:
"""(resolved title, wikitext) for a page, or None when it does not exist."""
safe = re.sub(r"[^A-Za-z0-9._ -]", "_", title)[:120]
@@ -265,7 +277,7 @@ def wiki_cover_url(game: dict, cache_dir: str | Path) -> tuple[str, str] | None:
article = None
for candidate in attempts:
found = wiki_article(candidate, cache_dir)
- if found and "Infobox video game" in found[1]:
+ if found and is_game_article(found[1]):
article = found
break