#!/usr/bin/env python3 """Fetch box art for the library and attach it to each game. Art comes from libretro-thumbnails (https://thumbnails.libretro.com), a community archive of boxart named to the No-Intro / Redump conventions. It needs no API key, but covers retro consoles only — Xbox 360 has no thumbnail set, so those titles are reported as unsupported rather than mismatched. Images are pushed through the app's own POST /api/images endpoint, so they get the same validation, WebP re-encoding and per-user filing as a manual upload. Standard library only, so it runs without a virtualenv. python3 fetch_art.py --password '...' --dry-run # report, change nothing python3 fetch_art.py --password '...' # download and attach """ from __future__ import annotations import argparse import difflib import json import re import sys import time import unicodedata import urllib.error import urllib.parse import urllib.request from dataclasses import dataclass from pathlib import Path THUMBNAIL_HOST = "https://thumbnails.libretro.com" # Our system codes to libretro's DAT-derived directory names. # Xbox 360 is deliberately absent: libretro-thumbnails has no set for it. # Several directories may back one system code. Our "GB" covers both Game Boy # and Game Boy Color titles — Pokemon Trading Card Game, for instance, is only in # the Color set — so both are searched and the better match wins. SYSTEM_DIRS = { "NES": ["Nintendo - Nintendo Entertainment System"], "SNES": ["Nintendo - Super Nintendo Entertainment System"], "N64": ["Nintendo - Nintendo 64"], "GB": ["Nintendo - Game Boy", "Nintendo - Game Boy Color"], "GBA": ["Nintendo - Game Boy Advance"], "DS": ["Nintendo - Nintendo DS"], "GC": ["Nintendo - GameCube"], "WII": ["Nintendo - Wii"], "PS1": ["Sony - PlayStation"], "PS2": ["Sony - PlayStation 2"], "PSP": ["Sony - PlayStation Portable"], } # Preferred release region, best first. The library is a US collection, so a USA # release wins; a European box is better than nothing. REGION_RANK = ["usa", "world", "usa, europe", "europe", "japan, usa", "japan"] # Tags that mark a variant we would rather not pick when a plain release exists. UNDESIRABLE_TAGS = ( "beta", "proto", "demo", "sample", "virtual console", "switch online", "classic mini", "rev ", "alt", "unl", "aftermarket", "competition", ) ROMAN = { "i": 1, "ii": 2, "iii": 3, "iv": 4, "v": 5, "vi": 6, "vii": 7, "viii": 8, "ix": 9, "x": 10, "xi": 11, "xii": 12, "xiii": 13, "xiv": 14, "xv": 15, } # -------------------------------------------------------------------------- # Title normalisation # -------------------------------------------------------------------------- def strip_accents(text: str) -> str: """'Pokémon' -> 'Pokemon', so our un-accented rows still match.""" return "".join( c for c in unicodedata.normalize("NFKD", text) if not unicodedata.combining(c) ) def normalise(title: str) -> str: """Reduce a title to a comparable form. Roman numerals become digits, which is what makes our SNES 'Final Fantasy 2' line up with the catalogued 'Final Fantasy II', and the PS1 'Final Fantasy V' with 'Final Fantasy V' rather than drifting to a different entry. """ text = strip_accents(title).lower() text = text.replace("&", " and ") # The catalogue moves a leading article to the end: "Sims 2, The", # "Legend of Zelda, The - Ocarina of Time". Drop it before anything else, # while the comma that marks it is still there to find. text = re.sub(r",\s*(the|a|an)\b", " ", text) # libretro renders a subtitle colon as " - "; flatten both to a space. text = re.sub(r"\s+-\s+", " ", text) text = re.sub(r"[^a-z0-9]+", " ", text) words = [str(ROMAN.get(w, w)) for w in text.split()] # Leading articles carry no signal and differ between catalogues. while words and words[0] in ("the", "a", "an"): words.pop(0) return " ".join(words).strip() @dataclass class Candidate: filename: str base: str # title with all parenthetical tags removed tags: str # the tags, lowercased, for region and variant ranking directory: str # libretro set it came from, needed to build the download URL @property def region_rank(self) -> int: for i, region in enumerate(REGION_RANK): if region in self.tags: return i return len(REGION_RANK) @property def variant_penalty(self) -> int: return sum(1 for tag in UNDESIRABLE_TAGS if tag in self.tags) def parse_candidate(filename: str, directory: str) -> Candidate: stem = filename[:-4] if filename.lower().endswith(".png") else filename tags = " ".join(re.findall(r"\(([^)]*)\)", stem)).lower() base = re.sub(r"\s*\([^)]*\)", "", stem).strip() return Candidate(filename=filename, base=base, tags=tags, directory=directory) # -------------------------------------------------------------------------- # HTTP # -------------------------------------------------------------------------- def http(url: str, *, data=None, headers=None, method=None, timeout=90) -> bytes: request = urllib.request.Request(url, data=data, method=method) for key, value in (headers or {}).items(): request.add_header(key, value) last_error: Exception | None = None for attempt in range(3): try: with urllib.request.urlopen(request, timeout=timeout) as response: return response.read() except urllib.error.HTTPError as exc: # 4xx will not improve on retry; surface it immediately. if exc.code < 500: raise last_error = exc except (urllib.error.URLError, TimeoutError) as exc: last_error = exc time.sleep(1.5 * (attempt + 1)) raise RuntimeError(f"GET {url} failed after 3 attempts: {last_error}") def api_json(base: str, path: str, token: str | None = None, *, data=None, method=None): headers = {"Accept": "application/json"} if token: headers["Authorization"] = f"Bearer {token}" body = None if data is not None: body = json.dumps(data).encode() headers["Content-Type"] = "application/json" raw = http(base + path, data=body, headers=headers, method=method) return json.loads(raw) if raw else None def upload_image(base: str, token: str, filename: str, blob: bytes) -> dict: """multipart/form-data POST, hand-rolled to avoid a requests dependency.""" boundary = "----LudosArt" + str(int(time.time() * 1000)) body = b"".join([ f'--{boundary}\r\n'.encode(), f'Content-Disposition: form-data; name="file"; filename="{filename}"\r\n'.encode(), b"Content-Type: image/png\r\n\r\n", blob, f"\r\n--{boundary}--\r\n".encode(), ]) raw = http( base + "/api/images", data=body, headers={ "Authorization": f"Bearer {token}", "Content-Type": f"multipart/form-data; boundary={boundary}", }, ) return json.loads(raw) # -------------------------------------------------------------------------- # Listings # -------------------------------------------------------------------------- def load_listing(system: str, cache_dir: Path) -> list[Candidate]: """Every candidate for a system, across all of its libretro sets. Listings are cached on disk, so reruns and dry-runs cost nothing. """ candidates: list[Candidate] = [] for directory in SYSTEM_DIRS[system]: cache = cache_dir / f"{directory}.json" if cache.exists(): names = json.loads(cache.read_text()) else: quoted = urllib.parse.quote(directory) html = http(f"{THUMBNAIL_HOST}/{quoted}/Named_Boxarts/").decode( "utf-8", errors="replace" ) names = sorted( {urllib.parse.unquote(m) for m in re.findall(r'href="([^"]+\.png)"', html)} ) cache.parent.mkdir(parents=True, exist_ok=True) cache.write_text(json.dumps(names, indent=0)) candidates.extend(parse_candidate(n, directory) for n in names) return candidates def contains_tokens(haystack: list[str], needle: list[str]) -> bool: """True when `needle` appears as a contiguous run inside `haystack`.""" if not needle or len(needle) > len(haystack): return False return any( haystack[i:i + len(needle)] == needle for i in range(len(haystack) - len(needle) + 1) ) def best_match(title: str, candidates: list[Candidate]) -> tuple[Candidate | None, float]: """Highest-scoring candidate, with region and variant used as tie-breakers.""" target = normalise(title) if not target: return None, 0.0 target_tokens = target.split() scored: list[tuple[float, int, int, int, Candidate]] = [] for candidate in candidates: base = normalise(candidate.base) score = difflib.SequenceMatcher(None, target, base).ratio() base_tokens = base.split() extra = len(base_tokens) - len(target_tokens) # Subtitles go missing in both directions. Our rows sometimes omit what # the catalogue carries ("Wave Race 64" vs "Wave Race 64 - Kawasaki Jet # Ski") and sometimes carry what it omits (our "Donkey Kong Country 2: # Diddy's Kong Quest" vs the GBA set's "Donkey Kong Country 2"). Either # way a clean token-run containment is a strong signal, so compare # whichever is shorter against whichever is longer. # # Capped below 0.95 so a genuine exact title always outranks it, and # scaled by coverage so the closest-length candidate wins among several # ("Donkey Kong Country 3" beats a bare "Donkey Kong Country"). # Guard against collapsing a sequel onto its base game. If we are asking # for a number the candidate does not have — "Donkey Kong Country 2" # against a plain "Donkey Kong Country" — containment would happily match # the wrong box. Extra numbers on the candidate side are fine, since that # is just a series prefix ("Super Mario World 2 - Yoshi's Island"). target_numbers = {t for t in target_tokens if t.isdigit()} base_numbers = {t for t in base_tokens if t.isdigit()} sequel_mismatch = bool(target_numbers - base_numbers) shorter, longer = sorted((target_tokens, base_tokens), key=len) if extra != 0 and not sequel_mismatch and contains_tokens(longer, shorter): coverage = len(shorter) / len(longer) score = max(score, 0.88 + 0.06 * coverage) if score >= 0.80: scored.append( (score, -candidate.region_rank, -abs(extra), -candidate.variant_penalty, candidate) ) if not scored: return None, 0.0 # Similarity first, then the US release, the closest-length title, and the # plain variant over a revision or re-release. scored.sort(key=lambda t: (round(t[0], 3), t[1], t[2], t[3]), reverse=True) score, _, _, _, candidate = scored[0] return candidate, score # -------------------------------------------------------------------------- # Main # -------------------------------------------------------------------------- def main() -> int: parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) parser.add_argument("--api", default="http://localhost:8080", help="API base URL") parser.add_argument("--user", default="ckoch") parser.add_argument("--password", required=True) parser.add_argument("--dry-run", action="store_true", help="report matches, change nothing") parser.add_argument("--overwrite", action="store_true", help="replace art that is already set") parser.add_argument("--min-score", type=float, default=0.80, help="similarity below which a match is not applied (0-1)") parser.add_argument("--cache", default=str(Path(__file__).parent / ".cache")) args = parser.parse_args() base = args.api.rstrip("/") cache_dir = Path(args.cache) print("Signing in…") auth = api_json(base, "/api/auth/login", data={"userName": args.user, "password": args.password}) token = auth["token"] print("Fetching library…") games: list[dict] = [] page = 1 while True: result = api_json(base, f"/api/games?page={page}&pageSize=100", token) games.extend(result["items"]) if page >= result["totalPages"] or not result["items"]: break page += 1 print(f" {len(games)} games\n") systems = sorted({g["system"] for g in games if g["system"]}) listings: dict[str, list[Candidate]] = {} for system in systems: if system not in SYSTEM_DIRS: continue print(f"Loading {system} catalogue…", end=" ", flush=True) listings[system] = load_listing(system, cache_dir) print(f"{len(listings[system])} covers") print() applied = skipped = failed = 0 unsupported: list[dict] = [] weak: list[tuple[dict, str, float]] = [] for game in sorted(games, key=lambda g: g["title"].lower()): title, system = game["title"], game["system"] if game.get("art") and not args.overwrite: skipped += 1 continue if not system or system not in SYSTEM_DIRS: unsupported.append(game) continue candidate, score = best_match(title, listings[system]) if not candidate or score < args.min_score: print(f" ? {system:4} {title[:46]:48} no match") failed += 1 continue flag = " " if score >= 0.95 else "~" print(f" {flag} {system:4} {title[:46]:48} {score:.2f} {candidate.filename[:52]}") if score < 0.95: weak.append((game, candidate.filename, score)) if args.dry_run: applied += 1 continue try: directory = urllib.parse.quote(candidate.directory) name = urllib.parse.quote(candidate.filename) blob = http(f"{THUMBNAIL_HOST}/{directory}/Named_Boxarts/{name}") uploaded = upload_image(base, token, candidate.filename, blob) payload = {k: game.get(k) for k in ( "title", "system", "genre", "year", "developer", "publisher", "description", "own", "dumped", "played", "finished")} payload["art"] = uploaded["fileName"] api_json(base, f"/api/games/{game['id']}", token, data=payload, method="PUT") applied += 1 except Exception as exc: # noqa: BLE001 - report and continue the batch print(f" -> FAILED: {exc}") failed += 1 # ---- report ---------------------------------------------------------- print("\n" + "=" * 72) verb = "would attach" if args.dry_run else "attached" print(f"{verb}: {applied} already had art: {skipped} no match: {failed} " f"unsupported system: {len(unsupported)}") if weak: print(f"\nWorth eyeballing in the UI — matched below 0.95 similarity ({len(weak)}):") for game, filename, score in sorted(weak, key=lambda w: w[2]): print(f" {score:.2f} {game['system']:4} {game['title'][:40]:42} -> {filename[:50]}") if unsupported: systems_missing = sorted({g['system'] or '(none)' for g in unsupported}) print(f"\nNo libretro thumbnail set for: {', '.join(systems_missing)} " f"({len(unsupported)} games). These need IGDB:") for game in unsupported[:15]: print(f" {game['system'] or '-':4} {game['title']}") return 0 if __name__ == "__main__": sys.exit(main())