diff --git a/.gitattributes b/.gitattributes index dfe07704..7d4cf3d1 100644 --- a/.gitattributes +++ b/.gitattributes @@ -1,2 +1,3 @@ # Auto detect text files and perform LF normalization * text=auto +db/static/content/** -text diff --git a/.github/workflows/generate-database.yml b/.github/workflows/generate-database.yml deleted file mode 100644 index 85736fed..00000000 --- a/.github/workflows/generate-database.yml +++ /dev/null @@ -1,140 +0,0 @@ -name: Generate ROM Database - -on: - schedule: - - cron: '0 0 * * 0' # Weekly on Sunday at midnight UTC - workflow_dispatch: # Allow manual trigger - -permissions: - contents: write - -jobs: - generate: - runs-on: ubuntu-latest - timeout-minutes: 120 - defaults: - run: - working-directory: db - - steps: - - name: Checkout repository - uses: actions/checkout@v7 - with: - ssh-key: ${{ secrets.DB_DEPLOY_KEY }} - - - name: Set up Python - uses: actions/setup-python@v7 - with: - python-version: '3.11' - cache: 'pip' - cache-dependency-path: db/requirements.txt - - - name: Cache scraped data - uses: actions/cache@v6 - with: - path: | - db/cache - db/data - key: scraper-cache-${{ github.run_number }} - restore-keys: | - scraper-cache- - - # MiNERVA artefacts are mirrored inside workflow.py via - # scripts/download_minerva_artefacts.py — same code path as a - # local `python workflow.py` run. ETag-cached, resumable. - - - name: Install Python dependencies - run: pip install -r requirements.txt - - - name: Install Playwright browsers - run: playwright install chromium - - - name: Create Internet Archive credentials - env: - IA_USERNAME: ${{ secrets.IA_USERNAME }} - IA_PASSWORD: ${{ secrets.IA_PASSWORD }} - run: | - if [ -n "$IA_USERNAME" ]; then - mkdir -p secrets - python -c "import json,os; json.dump({'username':os.environ['IA_USERNAME'],'password':os.environ['IA_PASSWORD']}, open('secrets/internet_archive_creds.json','w'))" - fi - - - name: Run database generation workflow - env: - RA_API_USER: ${{ secrets.RA_API_USER }} - RA_API_KEY: ${{ secrets.RA_API_KEY }} - run: python workflow.py - - - name: Compress database - run: gzip -f -k romdb.db - - - name: Get database stats - id: stats - run: | - DB_SIZE=$(stat -c%s romdb.db) - GZ_SIZE=$(stat -c%s romdb.db.gz) - ENTRY_COUNT=$(sqlite3 romdb.db "SELECT COUNT(*) FROM entries") - LINK_COUNT=$(sqlite3 romdb.db "SELECT COUNT(*) FROM links") - PLATFORM_COUNT=$(sqlite3 romdb.db "SELECT COUNT(DISTINCT platform) FROM entries") - SOURCE_COUNT=$(sqlite3 romdb.db "SELECT COUNT(*) FROM sources") - RA_COUNT=$(sqlite3 romdb.db "SELECT COUNT(*) FROM entries WHERE ra_game_id IS NOT NULL") - SCHEMA_VERSION=$(python -c "from database import db_manager; print(db_manager.SCHEMA_VERSION)") - echo "db_size=$DB_SIZE" >> $GITHUB_OUTPUT - echo "gz_size=$GZ_SIZE" >> $GITHUB_OUTPUT - echo "entry_count=$ENTRY_COUNT" >> $GITHUB_OUTPUT - echo "link_count=$LINK_COUNT" >> $GITHUB_OUTPUT - echo "platform_count=$PLATFORM_COUNT" >> $GITHUB_OUTPUT - echo "source_count=$SOURCE_COUNT" >> $GITHUB_OUTPUT - echo "ra_count=$RA_COUNT" >> $GITHUB_OUTPUT - echo "schema_version=$SCHEMA_VERSION" >> $GITHUB_OUTPUT - - - name: Refuse to publish a gutted catalog - env: - ALLOW_EMPTY_SOURCES: ${{ vars.ALLOW_EMPTY_SOURCES }} - run: | - OLD=$(python -c "import json; print(json.load(open('version.json'))['entries'])") - NEW=${{ steps.stats.outputs.entry_count }} - if [ "$NEW" -lt $((OLD * 7 / 10)) ]; then - echo "Entry count dropped from $OLD to $NEW - refusing to publish" - exit 1 - fi - EMPTY=$(sqlite3 romdb.db "SELECT s.id FROM sources s WHERE NOT EXISTS (SELECT 1 FROM links l WHERE l.source_id = s.id)") - BLOCKED="" - for src in $EMPTY; do - case ",$ALLOW_EMPTY_SOURCES," in - *",$src,"*) echo "Source $src has zero links (allowed by ALLOW_EMPTY_SOURCES)" ;; - *) BLOCKED="$BLOCKED $src" ;; - esac - done - if [ -n "$BLOCKED" ]; then - echo "Sources with zero links:$BLOCKED - refusing to publish" - exit 1 - fi - - - name: Create version.json - run: | - cat > version.json << EOF - { - "version": "$(date +%Y%m%d)", - "generated_at": "$(date -u +%Y-%m-%dT%H:%M:%SZ)", - "schema_version": ${{ steps.stats.outputs.schema_version }}, - "min_app_version": "1.2.0", - "size": ${{ steps.stats.outputs.gz_size }}, - "uncompressed_size": ${{ steps.stats.outputs.db_size }}, - "entries": ${{ steps.stats.outputs.entry_count }}, - "links": ${{ steps.stats.outputs.link_count }}, - "platforms": ${{ steps.stats.outputs.platform_count }}, - "sources": ${{ steps.stats.outputs.source_count }}, - "retroachievements": ${{ steps.stats.outputs.ra_count }} - } - EOF - - - name: Commit database to repository - working-directory: . - run: | - git config user.name github-actions - git config user.email github-actions@github.com - git add db/romdb.db.gz db/version.json - git commit -m "Update database $(date +%Y%m%d) - ${{ steps.stats.outputs.entry_count }} entries" || exit 0 - git pull --rebase - git push diff --git a/.github/workflows/pr-checks.yml b/.github/workflows/pr-checks.yml index f628616c..6e84154f 100644 --- a/.github/workflows/pr-checks.yml +++ b/.github/workflows/pr-checks.yml @@ -10,30 +10,6 @@ concurrency: cancel-in-progress: true jobs: - python-tests: - name: Python tests (db/) - runs-on: ubuntu-latest - steps: - - uses: actions/checkout@v7 - - - name: Set up Python - uses: actions/setup-python@v7 - with: - python-version: '3.11' - cache: 'pip' - cache-dependency-path: db/requirements.txt - - - name: Install runtime + test deps - run: pip install -r db/requirements.txt pytest - - - name: Build MiNERVA test fixture - working-directory: db - run: python tests/fixtures/build_minerva_fixture.py - - - name: Run pytest - working-directory: db - run: python -m pytest tests -v - flutter-analyze: name: Flutter analyze runs-on: ubuntu-latest diff --git a/README.md b/README.md index dc7f3690..640623db 100644 --- a/README.md +++ b/README.md @@ -121,19 +121,9 @@ The built APK will be located at `build/app/outputs/flutter-apk/app-release.apk` Release builds are filtered to **arm64-v8a** to keep the APK size sane after bundling the BitTorrent runtime. Every modern Android phone and retro handheld is arm64. To produce a 4-ABI build, edit `abiFilters` in `android/app/build.gradle.kts`. -### Rebuild the catalog database (optional) +### The catalog database -The app pulls a pre-built SQLite catalog from this repo. CI rebuilds it weekly. If you want to run the build locally: - -```bash -cd db -pip install -r requirements.txt -python workflow.py # full rebuild -python workflow.py --use-cached # reuse cached HTTP responses -python workflow.py --skip-minerva # skip the ~1.7 GB MiNERVA mirror download -``` - -Output is `db/romdb.db`. Adding a new source = drop a folder under `db/sources//` with `source.yml` + `scraper.py` and add the platform routes in `db/platforms.yml`. +The app pulls a pre-built SQLite catalog from this repo, updated weekly. `db/CATALOG.md` documents the artifact format. The catalog is provided for use in romgi; forks and derivative tools are not supported. ## Technical Details diff --git a/db/.gitignore b/db/.gitignore index e6e3e654..fbe56afa 100644 --- a/db/.gitignore +++ b/db/.gitignore @@ -1,34 +1,10 @@ -# Cached HTTP responses +# Local pipeline leftovers, not part of the published artifacts cache/ - -# Downloaded metadata files data/ - -# Generated static files (PS3/PSV) -static/ - -# Uncompressed database (only commit the .gz) +secrets/ *.db -!romdb.db.gz - -# Temp files *.db-journal -*_temp.db -*_old.db - -# Python __pycache__/ -*.py[cod] -*$py.class -.Python -*.so -.env -venv/ -ENV/ - -# Config files (use examples instead) -config.json - -# Credentials (never commit) *_creds.json -secrets/ +scrapers/ +static/ diff --git a/db/CATALOG.md b/db/CATALOG.md new file mode 100644 index 00000000..5b50fc59 --- /dev/null +++ b/db/CATALOG.md @@ -0,0 +1,45 @@ +# Catalog format + +The app downloads these artifacts from this directory on `main`: + +| File | Purpose | +|---|---| +| `version.json` | Catalog metadata, checked before download | +| `romdb.db.gz` | Gzipped SQLite database | +| `static/content/ps3/raps/*.rap` | PS3 license files, referenced by catalog URLs | +| `static/content/psv/zrifs/*.txt` | PSVita zRIF keys, referenced by catalog URLs | + +## version.json + +```json +{ + "version": "20260904", + "generated_at": "2026-09-04T15:00:00Z", + "schema_version": 4, + "min_app_version": "1.2.0", + "size": 55424693, + "uncompressed_size": 345706496, + "entries": 241126, + "links": 400298, + "platforms": 50, + "sources": 4, + "retroachievements": 9576 +} +``` + +`schema_version` bumps trigger a wipe-and-redownload in the app. `min_app_version` gates catalogs that need newer app features. + +## Database schema (v4) + +- `platforms(id, brand, name)` +- `entries(slug PK, rom_id, search_key, title, platform, boxart_url, ra_game_id, ra_num_achievements)` +- `entries_fts` — FTS4 over `search_key`, `tokenize=unicode61` +- `regions(id, name)` / `regions_entries(entry, region)` +- `sources(id PK, name, homepage, kind, auth_required, priority, manifest_json)` +- `source_health(source_id PK, status, last_checked, reason, entry_count, link_count)` +- `user_sources(id, name, kind, config_json, created_at)` — app-owned, empty in published catalogs +- `torrents(infohash PK, source_id, name, magnet, torrent_blob, total_size, piece_length, file_count, trackers_json, added_at)` +- `links(entry, name, type, format, url, filename, host, size, size_str, source_url, source_id, requires_auth, torrent_infohash, torrent_file_index, torrent_file_path)` +- `entry_groups(id PK, kind, title, platform, member_count, metadata_json)` / `entry_group_members(group_id, entry, member_index, member_label)` — multi-disc grouping + +Indexes: `idx_entries_platform` on `entries(platform)`, `idx_entry_groups_kind` on `entry_groups(kind)`. diff --git a/db/core/__init__.py b/db/core/__init__.py deleted file mode 100644 index 6b9ced26..00000000 --- a/db/core/__init__.py +++ /dev/null @@ -1,23 +0,0 @@ -""" -Core build infrastructure: source plugin contract and registry. - -This package owns the *abstractions* the per-source plugins implement -and the *discovery mechanism* that wires them into the build pipeline. -Plugin code itself lives under db/sources//. -""" -from .contract import ( - BuildContext, - PlatformConfig, - Source, - SourceManifest, -) -from .registry import Registry, load_registry - -__all__ = [ - "BuildContext", - "PlatformConfig", - "Source", - "SourceManifest", - "Registry", - "load_registry", -] diff --git a/db/core/contract.py b/db/core/contract.py deleted file mode 100644 index 52ae24cf..00000000 --- a/db/core/contract.py +++ /dev/null @@ -1,96 +0,0 @@ -""" -Source plugin contract. - -Each source lives in db/sources// as: - source.yml static manifest - __init__.py exposes SOURCE - scraper.py the implementation -""" -from __future__ import annotations - -from dataclasses import dataclass, field -from typing import Any, Protocol, runtime_checkable - - -@dataclass(frozen=True) -class SourceManifest: - """Parsed source.yml. `raw` round-trips into sources.manifest_json.""" - - id: str - name: str - kind: str # 'catalog' | 'host' | 'hybrid' - homepage: str | None = None - auth_required: bool = False - priority: int = 0 - capabilities: tuple[str, ...] = () - platforms: tuple[str, ...] = () - raw: dict[str, Any] = field(default_factory=dict) - - -@dataclass -class PlatformConfig: - """One per-platform routing entry from platforms.yml.""" - - format: str - regions: list[str] - urls: list[str] - type: str - parsers: dict[str, dict] - filter: str | None = None - extras: dict[str, Any] = field(default_factory=dict) - - def to_legacy_dict(self) -> dict[str, Any]: - """Dict form expected by module-level scrape() functions.""" - return { - "format": self.format, - "regions": list(self.regions), - "urls": list(self.urls), - "type": self.type, - "parsers": dict(self.parsers), - "filter": self.filter or "", - **self.extras, - } - - -@dataclass -class BuildContext: - """Shared state for one build invocation.""" - - use_cached: bool = False - - -@runtime_checkable -class Source(Protocol): - """Plugin contract. - - Plugins expose a `SOURCE` symbol at db/sources//__init__.py - pointing at an instance (or class) satisfying this protocol. - """ - - manifest: SourceManifest - - def scrape( - self, - platform: str, - config: PlatformConfig, - ctx: BuildContext, - ) -> list[dict[str, Any]]: - """Return entry dicts. Entry shape: - { - 'title': str, - 'platform': str, - 'regions': list[str], - 'links': [ - { - 'name': str, 'type': str, 'format': str, - 'url': str, 'filename': str, 'host': str, - 'size': int, 'size_str': str, 'source_url': str, - }, - ... - ], - # optional: - 'rom_id': str, - 'boxart_url': str, - } - """ - ... diff --git a/db/core/registry.py b/db/core/registry.py deleted file mode 100644 index 369ea3f5..00000000 --- a/db/core/registry.py +++ /dev/null @@ -1,158 +0,0 @@ -""" -Source plugin discovery. - -Walks db/sources// folders, loads each source.yml manifest, imports -the package, and exposes the {id: Source} map make.py uses. Plugin authors -add a folder; nothing else in the pipeline needs editing. -""" -from __future__ import annotations - -import importlib -from dataclasses import dataclass -from pathlib import Path -from typing import Any - -import yaml - -from .contract import Source, SourceManifest - - -SOURCES_PACKAGE = "sources" # importable as `sources.` from db/ -SOURCES_DIR_NAME = "sources" - - -# -- manifest parsing -------------------------------------------------------- - -REQUIRED_MANIFEST_KEYS = ("id", "name", "kind") -VALID_KINDS = {"catalog", "host", "hybrid"} - - -def _parse_manifest(raw: dict[str, Any], source_id: str) -> SourceManifest: - """Validate a parsed source.yml dict and return a SourceManifest.""" - for key in REQUIRED_MANIFEST_KEYS: - if key not in raw: - raise ManifestError( - f"source.yml for '{source_id}' is missing required key: {key!r}" - ) - - if raw["id"] != source_id: - raise ManifestError( - f"source.yml id {raw['id']!r} does not match folder name {source_id!r}" - ) - - if raw["kind"] not in VALID_KINDS: - raise ManifestError( - f"source.yml for '{source_id}': kind {raw['kind']!r} not in " - f"{sorted(VALID_KINDS)}" - ) - - auth = raw.get("auth") or {} - capabilities = tuple(raw.get("capabilities") or ()) - platforms = tuple(raw.get("platforms") or ()) - - return SourceManifest( - id=raw["id"], - name=raw["name"], - kind=raw["kind"], - homepage=raw.get("homepage"), - auth_required=bool(auth.get("required", False)), - priority=int(raw.get("priority", 0)), - capabilities=capabilities, - platforms=platforms, - raw=raw, - ) - - -class ManifestError(ValueError): - """Raised when a source.yml file is missing keys or has invalid values.""" - - -class RegistryError(RuntimeError): - """Raised when plugin discovery or loading fails.""" - - -# -- discovery --------------------------------------------------------------- - -@dataclass -class Registry: - """Loaded set of source plugins, keyed by manifest id.""" - - sources: dict[str, Source] - manifests: dict[str, SourceManifest] - - def get(self, source_id: str) -> Source | None: - return self.sources.get(source_id) - - def ids(self) -> list[str]: - return sorted(self.sources.keys()) - - -def _sources_root(db_root: Path) -> Path: - return db_root / SOURCES_DIR_NAME - - -def _discover_source_dirs(db_root: Path) -> list[Path]: - root = _sources_root(db_root) - if not root.is_dir(): - raise RegistryError(f"Sources directory not found: {root}") - return sorted(p for p in root.iterdir() if p.is_dir() and not p.name.startswith("_")) - - -def _load_manifest(source_dir: Path) -> SourceManifest: - manifest_path = source_dir / "source.yml" - if not manifest_path.is_file(): - raise ManifestError(f"Missing source.yml in {source_dir}") - with manifest_path.open("r", encoding="utf-8") as f: - raw = yaml.safe_load(f) or {} - if not isinstance(raw, dict): - raise ManifestError(f"{manifest_path} did not parse to a mapping") - return _parse_manifest(raw, source_dir.name) - - -def _import_source_module(source_id: str): - """Import the plugin package. Expects db/ on sys.path (set by make.py).""" - return importlib.import_module(f"{SOURCES_PACKAGE}.{source_id}") - - -def _resolve_source_obj(module, manifest: SourceManifest) -> Source: - obj = getattr(module, "SOURCE", None) - if obj is None: - raise RegistryError( - f"Source '{manifest.id}' module {module.__name__} does not " - f"expose a SOURCE attribute" - ) - - # If the module exports a class, instantiate it with the manifest. - if isinstance(obj, type): - obj = obj(manifest) - - if not isinstance(obj, Source): - raise RegistryError( - f"Source '{manifest.id}' SOURCE does not satisfy the Source " - f"protocol (missing manifest or scrape())" - ) - - if obj.manifest.id != manifest.id: - raise RegistryError( - f"Source '{manifest.id}' SOURCE.manifest.id is " - f"{obj.manifest.id!r}, expected {manifest.id!r}" - ) - - return obj - - -def load_registry(db_root: Path | str) -> Registry: - """Discover and load every plugin under db/sources/.""" - db_root = Path(db_root) - sources: dict[str, Source] = {} - manifests: dict[str, SourceManifest] = {} - - for source_dir in _discover_source_dirs(db_root): - manifest = _load_manifest(source_dir) - if manifest.id in sources: - raise RegistryError(f"Duplicate source id: {manifest.id}") - module = _import_source_module(manifest.id) - sources[manifest.id] = _resolve_source_obj(module, manifest) - manifests[manifest.id] = manifest - - return Registry(sources=sources, manifests=manifests) diff --git a/db/database/db_manager.py b/db/database/db_manager.py deleted file mode 100644 index 35efd099..00000000 --- a/db/database/db_manager.py +++ /dev/null @@ -1,497 +0,0 @@ -""" -Catalog DB manager. Owns schema + write path. - -Bumps to SCHEMA_VERSION trigger an app-side wipe-and-redownload on -the next launch; we don't run live migrations. -""" -import json -import os -import sqlite3 -import time -from typing import Any - -from utils.parse_utils import create_slug, create_search_key - -DB_NAME = 'romdb.db' -DB_TEMP_NAME = 'romdb_temp.db' -DB_OLD_NAME = 'romdb_old.db' - -# Bump on any schema change. The app reads version.json#schema_version -# and wipes its local catalog DB on mismatch. -# Mirror of: lib/services/rom_database_service.dart `kAppExpectedSchemaVersion`. -SCHEMA_VERSION = 4 - -con: sqlite3.Connection | None = None -cur: sqlite3.Cursor | None = None - -PLATFORMS = { - 'nes': {'brand': 'Nintendo', 'name': 'Nintendo Entertainment System'}, - 'fds': {'brand': 'Nintendo', 'name': 'Famicom Disk System'}, - 'snes': {'brand': 'Nintendo', 'name': 'Super Nintendo Entertainment System'}, - 'gb': {'brand': 'Nintendo', 'name': 'Game Boy'}, - 'gbc': {'brand': 'Nintendo', 'name': 'Game Boy Color'}, - 'gba': {'brand': 'Nintendo', 'name': 'Game Boy Advance'}, - 'min': {'brand': 'Nintendo', 'name': 'Pokemon Mini'}, - 'vb': {'brand': 'Nintendo', 'name': 'Virtual Boy'}, - 'n64': {'brand': 'Nintendo', 'name': 'Nintendo 64'}, - 'ndd': {'brand': 'Nintendo', 'name': 'Nintendo 64DD'}, - 'gc': {'brand': 'Nintendo', 'name': 'GameCube'}, - 'nds': {'brand': 'Nintendo', 'name': 'Nintendo DS'}, - 'dsi': {'brand': 'Nintendo', 'name': 'Nintendo DSi'}, - 'wii': {'brand': 'Nintendo', 'name': 'Wii'}, - '3ds': {'brand': 'Nintendo', 'name': 'Nintendo 3DS'}, - 'n3ds': {'brand': 'Nintendo', 'name': 'New Nintendo 3DS'}, - 'wiiu': {'brand': 'Nintendo', 'name': 'Wii U'}, - 'ps1': {'brand': 'Sony', 'name': 'PlayStation'}, - 'ps2': {'brand': 'Sony', 'name': 'PlayStation 2'}, - 'psp': {'brand': 'Sony', 'name': 'PlayStation Portable'}, - 'ps3': {'brand': 'Sony', 'name': 'PlayStation 3'}, - 'psv': {'brand': 'Sony', 'name': 'PlayStation Vita'}, - 'xbox': {'brand': 'Microsoft', 'name': 'Xbox'}, - 'x360': {'brand': 'Microsoft', 'name': 'Xbox 360'}, - 'sms': {'brand': 'Sega', 'name': 'Master System - Mark III'}, - 'gg': {'brand': 'Sega', 'name': 'Game Gear'}, - 'smd': {'brand': 'Sega', 'name': 'Mega Drive - Genesis'}, - 'scd': {'brand': 'Sega', 'name': 'Mega-CD - Sega CD'}, - '32x': {'brand': 'Sega', 'name': '32X'}, - 'sat': {'brand': 'Sega', 'name': 'Sega Saturn'}, - 'dc': {'brand': 'Sega', 'name': 'Dreamcast'}, - 'mame': {'brand': 'Arcade', 'name': 'MAME'}, - 'fbneo': {'brand': 'Arcade', 'name': 'FinalBurn Neo'}, - 'a26': {'brand': 'Atari', 'name': 'Atari 2600'}, - 'a52': {'brand': 'Atari', 'name': 'Atari 5200'}, - 'a78': {'brand': 'Atari', 'name': 'Atari 7800'}, - 'lynx': {'brand': 'Atari', 'name': 'Atari Lynx'}, - 'jag': {'brand': 'Atari', 'name': 'Atari Jaguar'}, - 'jcd': {'brand': 'Atari', 'name': 'Atari Jaguar CD'}, - 'tg16': {'brand': 'NEC', 'name': 'PC Engine - TurboGrafx-16'}, - 'tgcd': {'brand': 'NEC', 'name': 'PC Engine CD - TurboGrafx-CD'}, - 'pcfx': {'brand': 'NEC', 'name': 'PC-FX'}, - 'pc98': {'brand': 'NEC', 'name': 'PC-98'}, - 'intv': {'brand': 'Mattel', 'name': 'Intellivision'}, - 'cv': {'brand': 'Coleco', 'name': 'ColecoVision'}, - '3do': {'brand': 'The 3DO Company', 'name': '3DO Interactive Multiplayer'}, - 'cdi': {'brand': 'Philips', 'name': 'CD-i'}, - 'fmt': {'brand': 'Fujitsu', 'name': 'FM Towns'}, - 'ngcd': {'brand': 'SNK', 'name': 'Neo Geo CD'}, - 'pip': {'brand': 'Apple-Bandai', 'name': 'Pippin'} -} - -REGIONS = { - 'eu': 'Europe', - 'us': 'USA', - 'jp': 'Japan', - 'other': 'Other' -} - - -def init_database() -> None: - """Initialize the database: create tables, indexes, seed static data.""" - global con, cur - - if os.path.exists(DB_TEMP_NAME): - os.remove(DB_TEMP_NAME) - - con = sqlite3.connect(DB_TEMP_NAME) - cur = con.cursor() - - cur.execute('PRAGMA foreign_keys = ON;') - - cur.execute(''' - CREATE TABLE platforms ( - id TEXT PRIMARY KEY, - brand TEXT, - name TEXT - ) - ''') - - cur.execute(''' - CREATE TABLE entries ( - slug TEXT PRIMARY KEY, - rom_id TEXT, - search_key TEXT, - title TEXT, - platform TEXT, - boxart_url TEXT, - ra_game_id INTEGER, - ra_num_achievements INTEGER, - FOREIGN KEY (platform) REFERENCES platforms (id) - ) - ''') - - cur.execute(''' - CREATE VIRTUAL TABLE entries_fts USING fts4( - search_key, - content='entries', - tokenize=unicode61 - ) - ''') - - cur.execute(''' - CREATE TABLE regions ( - id TEXT PRIMARY KEY, - name TEXT - ) - ''') - - cur.execute(''' - CREATE TABLE regions_entries ( - entry TEXT, - region TEXT, - FOREIGN KEY (entry) REFERENCES entries (slug), - FOREIGN KEY (region) REFERENCES regions (id) - ) - ''') - - cur.execute(''' - CREATE TABLE sources ( - id TEXT PRIMARY KEY, - name TEXT NOT NULL, - homepage TEXT, - kind TEXT NOT NULL, - auth_required INTEGER NOT NULL DEFAULT 0, - priority INTEGER NOT NULL DEFAULT 0, - manifest_json TEXT NOT NULL - ) - ''') - - cur.execute(''' - CREATE TABLE source_health ( - source_id TEXT PRIMARY KEY REFERENCES sources(id), - status TEXT NOT NULL, - last_checked INTEGER NOT NULL, - reason TEXT, - entry_count INTEGER, - link_count INTEGER - ) - ''') - - # Reserved for user-supplied sources. The build pipeline never writes - # here; the app owns it. - cur.execute(''' - CREATE TABLE user_sources ( - id TEXT PRIMARY KEY, - name TEXT NOT NULL, - kind TEXT NOT NULL, - config_json TEXT NOT NULL, - created_at INTEGER NOT NULL - ) - ''') - - # Torrent metadata, deduped by infohash. Each links row that - # represents a file inside a torrent points here via source_id. - cur.execute(''' - CREATE TABLE torrents ( - infohash TEXT PRIMARY KEY, - source_id TEXT NOT NULL REFERENCES sources(id), - name TEXT, - magnet TEXT, - torrent_blob BLOB, - total_size INTEGER, - piece_length INTEGER, - file_count INTEGER, - trackers_json TEXT, - added_at INTEGER NOT NULL - ) - ''') - - cur.execute(''' - CREATE TABLE links ( - entry TEXT, - name TEXT, - type TEXT, - format TEXT, - url TEXT, - filename TEXT, - host TEXT, - size INTEGER, - size_str TEXT, - source_url TEXT, - source_id TEXT REFERENCES sources(id), - requires_auth INTEGER NOT NULL DEFAULT 0, - torrent_infohash TEXT REFERENCES torrents(infohash), - torrent_file_index INTEGER, - torrent_file_path TEXT, - FOREIGN KEY (entry) REFERENCES entries (slug) - ) - ''') - - # Logical groups relating entries (multi-disc games today; the `kind` - # discriminator + metadata_json leave room for revisions/bundles later). - cur.execute(''' - CREATE TABLE entry_groups ( - id TEXT PRIMARY KEY, - kind TEXT NOT NULL, - title TEXT, - platform TEXT, - member_count INTEGER NOT NULL, - metadata_json TEXT - ) - ''') - - cur.execute(''' - CREATE TABLE entry_group_members ( - group_id TEXT NOT NULL REFERENCES entry_groups (id), - entry TEXT NOT NULL REFERENCES entries (slug), - member_index INTEGER, - member_label TEXT, - PRIMARY KEY (group_id, entry) - ) - ''') - - cur.execute('CREATE INDEX idx_entries_platform ON entries (platform);') - cur.execute('CREATE INDEX idx_entry_groups_kind ON entry_groups (kind);') - cur.execute( - 'CREATE INDEX idx_group_members_entry ON entry_group_members (entry);') - cur.execute( - 'CREATE INDEX idx_regions_entries_entry ON regions_entries (entry);') - cur.execute( - 'CREATE INDEX idx_regions_entries_region ON regions_entries (region);') - cur.execute('CREATE INDEX idx_links_entry ON links (entry);') - cur.execute('CREATE INDEX idx_links_source ON links (source_id);') - cur.execute('CREATE INDEX idx_links_torrent ON links (torrent_infohash);') - - for id, info in PLATFORMS.items(): - cur.execute('INSERT INTO platforms (id, brand, name) VALUES (?, ?, ?)', - (id, info['brand'], info['name'])) - - for id, name in REGIONS.items(): - cur.execute('INSERT INTO regions (id, name) VALUES (?, ?)', (id, name)) - - -def register_source(manifest: Any) -> None: - """Insert a source row from a SourceManifest. Called once per source.""" - assert cur is not None - cur.execute( - 'INSERT INTO sources ' - '(id, name, homepage, kind, auth_required, priority, manifest_json) ' - 'VALUES (?, ?, ?, ?, ?, ?, ?)', - ( - manifest.id, - manifest.name, - manifest.homepage, - manifest.kind, - int(bool(manifest.auth_required)), - int(manifest.priority), - json.dumps(manifest.raw, sort_keys=True), - ), - ) - - -def register_torrent( - *, - infohash: str, - source_id: str, - name: str | None = None, - magnet: str | None = None, - torrent_blob: bytes | None = None, - total_size: int | None = None, - piece_length: int | None = None, - file_count: int | None = None, - trackers: list[str] | None = None, - added_at: int | None = None, -) -> None: - """Insert a torrent row idempotently (no-op if infohash already present).""" - assert cur is not None - ts = added_at if added_at is not None else int(time.time()) - cur.execute( - 'INSERT OR IGNORE INTO torrents ' - '(infohash, source_id, name, magnet, torrent_blob, ' - ' total_size, piece_length, file_count, trackers_json, added_at) ' - 'VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?)', - ( - infohash.lower(), - source_id, - name, - magnet, - torrent_blob, - total_size, - piece_length, - file_count, - json.dumps(trackers) if trackers is not None else None, - ts, - ), - ) - - -def record_source_health( - source_id: str, - status: str, - *, - reason: str | None = None, - entry_count: int = 0, - link_count: int = 0, - last_checked: int | None = None, -) -> None: - """Upsert a source_health row.""" - assert cur is not None - ts = last_checked if last_checked is not None else int(time.time()) - cur.execute( - 'INSERT OR REPLACE INTO source_health ' - '(source_id, status, last_checked, reason, entry_count, link_count) ' - 'VALUES (?, ?, ?, ?, ?, ?)', - (source_id, status, ts, reason, entry_count, link_count), - ) - - -def insert_entry(entry: dict[str, Any]) -> None: - """Insert a new entry into the database or update it if it exists.""" - assert cur is not None - entry['slug'] = create_slug(entry) - entry['search_key'] = create_search_key(entry['title']) - - cur.execute("SELECT slug FROM entries WHERE slug = ?", (entry['slug'],)) - existing_entry = cur.fetchone() - - if existing_entry: - cur.execute(''' - UPDATE entries - SET rom_id = COALESCE(rom_id, ?), - search_key = COALESCE(search_key, ?), - title = COALESCE(title, ?), - platform = COALESCE(platform, ?), - boxart_url = COALESCE(boxart_url, ?), - ra_game_id = COALESCE(ra_game_id, ?), - ra_num_achievements = COALESCE(ra_num_achievements, ?) - WHERE slug = ? - ''', ( - entry.get('rom_id'), - entry.get('search_key'), - entry.get('title'), - entry.get('platform'), - entry.get('boxart_url'), - entry.get('ra_game_id'), - entry.get('ra_num_achievements'), - entry['slug'] - )) - - for link in entry.get('links', []): - _insert_link(entry['slug'], link, ignore_duplicates=True) - else: - cur.execute(''' - INSERT INTO entries (slug, rom_id, search_key, title, platform, boxart_url, - ra_game_id, ra_num_achievements) - VALUES (?, ?, ?, ?, ?, ?, ?, ?) - ''', ( - entry.get('slug'), - entry.get('rom_id'), - entry.get('search_key'), - entry.get('title'), - entry.get('platform'), - entry.get('boxart_url'), - entry.get('ra_game_id'), - entry.get('ra_num_achievements') - )) - - cur.execute(''' - INSERT INTO entries_fts (docid, search_key) - VALUES (last_insert_rowid(), ?) - ''', (entry['search_key'],)) - - for region in entry.get('regions', []): - cur.execute(''' - INSERT OR IGNORE INTO regions_entries (entry, region) - VALUES (?, ?) - ''', (entry.get('slug'), region)) - - for link in entry.get('links', []): - _insert_link(entry['slug'], link, ignore_duplicates=False) - - -def _insert_link(entry_slug: str, link: dict[str, Any], *, ignore_duplicates: bool) -> None: - """Insert one link row. - - If `_torrent_meta` is set on the link, the corresponding torrents - row is upserted first so scrapers don't have to call register_torrent. - """ - assert cur is not None - meta = link.get('_torrent_meta') - if meta is not None: - register_torrent(**meta) - - sql = ( - 'INSERT OR IGNORE INTO links (' if ignore_duplicates - else 'INSERT INTO links (' - ) + ( - 'entry, name, type, format, url, filename, host, size, size_str, ' - 'source_url, source_id, requires_auth, ' - 'torrent_infohash, torrent_file_index, torrent_file_path' - ') VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)' - ) - cur.execute(sql, ( - entry_slug, - link.get('name'), - link.get('type'), - link.get('format'), - link.get('url'), - link.get('filename'), - link.get('host'), - link.get('size'), - link.get('size_str'), - link.get('source_url'), - link.get('source_id'), - int(bool(link.get('requires_auth'))), - link.get('torrent_infohash'), - link.get('torrent_file_index'), - link.get('torrent_file_path'), - )) - - -def fetch_entries_for_grouping() -> list[tuple[str, str, str, tuple[str, ...]]]: - """Return (slug, title, platform, regions) for every entry. - - Regions are sorted+deduped so grouping keys are order-independent. Called - by the build driver after all entries are inserted, before close. - """ - assert cur is not None - cur.execute(''' - SELECT e.slug, e.title, e.platform, GROUP_CONCAT(re.region) - FROM entries e - LEFT JOIN regions_entries re ON re.entry = e.slug - GROUP BY e.slug - ''') - out: list[tuple[str, str, str, tuple[str, ...]]] = [] - for slug, title, platform, regions_csv in cur.fetchall(): - regions = tuple(sorted({r for r in (regions_csv or '').split(',') if r})) - out.append((slug, title, platform, regions)) - return out - - -def store_entry_groups(groups: Any) -> None: - """Persist grouping results into entry_groups / entry_group_members.""" - assert cur is not None - for group in groups: - cur.execute(''' - INSERT INTO entry_groups - (id, kind, title, platform, member_count, metadata_json) - VALUES (?, ?, ?, ?, ?, ?) - ''', ( - group.id, - group.kind, - group.title, - group.platform, - len(group.members), - json.dumps(group.metadata, sort_keys=True) if group.metadata else None, - )) - for member in group.members: - cur.execute(''' - INSERT OR IGNORE INTO entry_group_members - (group_id, entry, member_index, member_label) - VALUES (?, ?, ?, ?) - ''', (group.id, member.slug, member.index, member.label)) - - -def close_database() -> None: - """Close the database connection and finalize changes.""" - assert con is not None - assert cur is not None - con.commit() - - cur.close() - con.close() - - if os.path.exists(DB_NAME): - if os.path.exists(DB_OLD_NAME): - os.remove(DB_OLD_NAME) - os.rename(DB_NAME, DB_OLD_NAME) - os.rename(DB_TEMP_NAME, DB_NAME) diff --git a/db/grouping/__init__.py b/db/grouping/__init__.py deleted file mode 100644 index 5e2903d0..00000000 --- a/db/grouping/__init__.py +++ /dev/null @@ -1,21 +0,0 @@ -"""Entry grouping: pluggable strategies that relate catalog entries. - -Public surface used by the build pipeline: - load_strategies() -> discover strategy plugins - build_groups(...) -> bucket + promote memberships into groups -""" -from __future__ import annotations - -from .base import EntryRef, GroupingStrategy, Membership -from .build import EntryGroup, GroupMember, build_groups -from .registry import load_strategies - -__all__ = [ - "EntryRef", - "GroupingStrategy", - "Membership", - "EntryGroup", - "GroupMember", - "build_groups", - "load_strategies", -] diff --git a/db/grouping/base.py b/db/grouping/base.py deleted file mode 100644 index 8296ea47..00000000 --- a/db/grouping/base.py +++ /dev/null @@ -1,63 +0,0 @@ -""" -Entry-grouping contract. - -A grouping strategy decides whether a single catalog entry participates in -a larger logical group (multi-disc games today; revisions, bundles, and -multi-part releases are natural future additions). Each strategy is pure: -it inspects one entry at a time and never depends on the others. The build -pass (see `build.py`) is what buckets memberships and promotes buckets with -enough members into real groups. - -Strategies are discovered as plugins under `db/grouping/strategies/` — drop -a module exposing a `STRATEGY` symbol and nothing else in the pipeline needs -editing. This mirrors the source-plugin idiom in `db/sources/`. -""" -from __future__ import annotations - -from dataclasses import dataclass, field -from typing import Any, Protocol, runtime_checkable - - -@dataclass(frozen=True) -class EntryRef: - """Minimal, read-only view of one catalog entry the grouping layer sees. - - `regions` is normalized (sorted, deduped) by the caller so that two - members of the same product always hash to the same group key. - """ - - slug: str - title: str - platform: str - regions: tuple[str, ...] - - -@dataclass(frozen=True) -class Membership: - """A strategy's verdict that one entry belongs to a group. - - `key` is a stable bucket identifier shared by every member of the same - group within a build. `index` orders members (e.g. disc number). `title` - is the canonical group title with the grouping token stripped out. - """ - - key: str - index: int - label: str - title: str - metadata: dict[str, Any] = field(default_factory=dict) - - -@runtime_checkable -class GroupingStrategy(Protocol): - """Plugin contract. Expose an instance (or class) as `STRATEGY`.""" - - #: Discriminator persisted on every group this strategy produces - #: (e.g. 'disc'). Lets the app and future strategies coexist. - kind: str - - def membership(self, entry: EntryRef) -> Membership | None: - """Return a Membership if `entry` participates in a group of this - kind, otherwise None. Must be pure — no dependence on other entries. - """ - ... diff --git a/db/grouping/build.py b/db/grouping/build.py deleted file mode 100644 index be5cdc2d..00000000 --- a/db/grouping/build.py +++ /dev/null @@ -1,96 +0,0 @@ -""" -Grouping build pass. - -Pure orchestration: given the full entry list and the loaded strategies, -bucket every membership by key and promote only the buckets that look like -genuine groups (enough members, more than one distinct position). Kept free -of SQL so it is trivially unit-testable; `make.py` feeds it rows and hands -the result to `db_manager` for persistence. -""" -from __future__ import annotations - -from collections import defaultdict -from dataclasses import dataclass, field -from typing import Any, Iterable, Sequence - -from .base import EntryRef, GroupingStrategy - - -@dataclass(frozen=True) -class GroupMember: - slug: str - index: int - label: str - - -@dataclass(frozen=True) -class EntryGroup: - id: str - kind: str - title: str - platform: str - members: list[GroupMember] - metadata: dict[str, Any] = field(default_factory=dict) - - -#: A bucket is only a real group when at least this many entries land in it. -DEFAULT_MIN_MEMBERS = 2 - - -def _group_id(kind: str, key: str) -> str: - """Namespace the bucket key by kind so ids stay unique across strategies.""" - return f"{kind}:{key}" - - -def build_groups( - entries: Iterable[EntryRef], - strategies: Sequence[GroupingStrategy], - *, - min_members: int = DEFAULT_MIN_MEMBERS, -) -> list[EntryGroup]: - """Bucket memberships per strategy and return the promoted groups. - - A bucket is promoted only when it holds >= `min_members` entries AND - spans more than one distinct index. The index guard is what keeps - unrelated same-titled dumps (or a lone ``(Disc 1)``) from forming a - bogus one-position "group". - """ - entries = list(entries) - groups: list[EntryGroup] = [] - - for strategy in strategies: - buckets: dict[str, list[tuple[EntryRef, Any]]] = defaultdict(list) - for entry in entries: - member = strategy.membership(entry) - if member is not None: - buckets[member.key].append((entry, member)) - - for key, items in buckets.items(): - # Collapse to one entry per position. The same physical - # disc is sometimes catalogued twice under different labels - by_index: dict[int, tuple[EntryRef, Any]] = {} - for entry, member in items: - current = by_index.get(member.index) - if current is None or entry.slug < current[0].slug: - by_index[member.index] = (entry, member) - - if len(by_index) < min_members: - continue - - ordered = sorted(by_index.values(), key=lambda im: im[1].index) - canonical = ordered[0][1] - groups.append( - EntryGroup( - id=_group_id(strategy.kind, key), - kind=strategy.kind, - title=canonical.title, - platform=ordered[0][0].platform, - members=[ - GroupMember(slug=e.slug, index=m.index, label=m.label) - for e, m in ordered - ], - metadata=dict(canonical.metadata), - ) - ) - - return groups diff --git a/db/grouping/registry.py b/db/grouping/registry.py deleted file mode 100644 index d2b698e0..00000000 --- a/db/grouping/registry.py +++ /dev/null @@ -1,36 +0,0 @@ -""" -Grouping strategy discovery. - -Walks `db/grouping/strategies/`, imports each module, and collects the -`STRATEGY` symbol it exposes. Adding a strategy is a one-file drop — nothing -here or in `make.py` needs editing. -""" -from __future__ import annotations - -import importlib -import pkgutil - -from .base import GroupingStrategy - - -def load_strategies() -> list[GroupingStrategy]: - """Discover and instantiate every strategy plugin, ordered by kind.""" - from . import strategies as pkg - - found: list[GroupingStrategy] = [] - for mod in pkgutil.iter_modules(pkg.__path__): - module = importlib.import_module(f"{pkg.__name__}.{mod.name}") - strategy = getattr(module, "STRATEGY", None) - if strategy is None: - continue - if isinstance(strategy, type): - strategy = strategy() - if not isinstance(strategy, GroupingStrategy): - raise TypeError( - f"{module.__name__}.STRATEGY does not satisfy GroupingStrategy " - f"(needs a `kind` attr and a `membership()` method)" - ) - found.append(strategy) - - found.sort(key=lambda s: s.kind) - return found diff --git a/db/grouping/strategies/__init__.py b/db/grouping/strategies/__init__.py deleted file mode 100644 index 1c9fd565..00000000 --- a/db/grouping/strategies/__init__.py +++ /dev/null @@ -1 +0,0 @@ -"""Grouping strategy plugins. Each module exposes a `STRATEGY` symbol.""" diff --git a/db/grouping/strategies/disc.py b/db/grouping/strategies/disc.py deleted file mode 100644 index 1af2daff..00000000 --- a/db/grouping/strategies/disc.py +++ /dev/null @@ -1,101 +0,0 @@ -""" -Multi-disc grouping strategy. - -Recognizes the ``(Disc N)`` family of tokens that Redump/No-Intro style -titles use for multi-disc games, strips the token to derive a canonical -group title, and combines it with platform + region into a stable key. -""" -from __future__ import annotations - -import re - -from utils.parse_utils import create_slug - -from ..base import EntryRef, Membership - -_DISC_TOKEN = re.compile( - r"\s*\(\s*(?:Disc|Disk|Disco)\s+" - r"(?P\d+|[IVXLCDM]+|[A-Za-z])" - r"(?:\s+of\s+(?:\d+|[IVXLCDM]+|[A-Za-z]))?" - r"\s*\)", - re.IGNORECASE, -) - -_ROMAN = {"I": 1, "V": 5, "X": 10, "L": 50, "C": 100, "D": 500, "M": 1000} - - -def _roman_to_int(raw: str) -> int | None: - """Parse a roman numeral, or None if it is not well-formed.""" - total = 0 - prev = 0 - for ch in reversed(raw.upper()): - value = _ROMAN.get(ch) - if value is None: - return None - if value < prev: - total -= value - else: - total += value - prev = value - return total or None - - -def _parse_index(raw: str) -> int | None: - """Turn a disc token's index into an orderable integer. - - Numbers win outright. A lone ``I`` is read as roman 1 (so ``I``/``II`` - order as 1/2), while any other single letter is read as an A=1 series - position (so ``A``/``B``/``C`` order as 1/2/3). This resolves the - letter-vs-roman ambiguity in favor of how these labels are used on disc. - """ - if raw.isdigit(): - return int(raw) - upper = raw.upper() - if len(upper) == 1 and upper != "I": - if "A" <= upper <= "Z": - return ord(upper) - ord("A") + 1 - return None - return _roman_to_int(upper) - - -def _clean_title(title: str) -> str: - """Strip every disc token and tidy the leftover whitespace/separators.""" - stripped = _DISC_TOKEN.sub(" ", title) - stripped = re.sub(r"\s+", " ", stripped).strip() - return stripped.strip(" -") - - -class DiscStrategy: - kind = "disc" - - def membership(self, entry: EntryRef) -> Membership | None: - match = _DISC_TOKEN.search(entry.title) - if match is None: - return None - - index = _parse_index(match.group("idx")) - if index is None: - return None - - base_title = _clean_title(entry.title) - if not base_title: - return None - - key = create_slug( - { - "title": base_title, - "platform": entry.platform, - "regions": list(entry.regions), - } - ) - label = re.sub(r"\s+", " ", match.group(0).strip().strip("()").strip()) - - return Membership( - key=key, - index=index, - label=label, - title=base_title, - ) - - -STRATEGY = DiscStrategy() diff --git a/db/make.py b/db/make.py deleted file mode 100644 index 944c83c7..00000000 --- a/db/make.py +++ /dev/null @@ -1,231 +0,0 @@ -#!/usr/bin/env python -""" -Build driver. Loads the source registry, walks db/platforms.yml, and runs -scrape → parse → insert for every (platform, source) pair. -""" -from __future__ import annotations - -import os -import sys -from pathlib import Path -from typing import Any, Iterable - -import yaml - -from core import ( - BuildContext, - PlatformConfig, - Source, - load_registry, -) -from database import db_manager -from grouping import EntryRef, build_groups, load_strategies -from parsers import gametdb, libretro, mame, no_intro, retroachievements, wii_rom_set_by_ghostware - - -PARSERS = { - 'no_intro': no_intro, - 'libretro': libretro, - 'gametdb': gametdb, - 'mame': mame, - 'wii_rom_set_by_ghostware': wii_rom_set_by_ghostware, -} - - -def get_parser(name: str) -> Any: - """Retrieve a parser by name.""" - return PARSERS.get(name) - - -def load_platforms(file_path: str | Path = 'platforms.yml') -> dict[str, list[dict[str, Any]]]: - """Load the per-platform routing config.""" - with open(file_path, 'r', encoding='utf-8') as f: - data = yaml.safe_load(f) or {} - if not isinstance(data, dict): - raise ValueError(f"{file_path} must be a mapping of platform -> list") - return data - - -def _build_platform_config(entry: dict[str, Any]) -> tuple[str, PlatformConfig]: - """Pull the source id and the typed config out of one platforms.yml entry.""" - extras = { - k: v for k, v in entry.items() - if k not in {'source', 'format', 'regions', 'urls', 'filter', 'type', 'parsers'} - } - config = PlatformConfig( - format=entry['format'], - regions=list(entry.get('regions') or []), - urls=list(entry.get('urls') or []), - type=entry.get('type', ''), - parsers=dict(entry.get('parsers') or {}), - filter=entry.get('filter'), - extras=extras, - ) - return entry['source'], config - - -def _print_source_header(index: int, source_id: str, config: PlatformConfig) -> None: - print(f" {index}) ", end='') - print(f"[{config.format}] ", end='') - if config.regions: - print(f"[{', '.join(config.regions)}] ", end='') - print(f"[{source_id}] ", end='') - print(f"[{config.type}]") - - -def _tag_links(entries_out: list[dict[str, Any]], source: Source) -> tuple[int, int]: - """Default source_id + requires_auth on every link. - - Scrapers may override either by setting them explicitly on the link - dict (the IA scraper does this for restricted entries). Returns - (entries_count, links_count) for source_health bookkeeping. - """ - n_entries = 0 - n_links = 0 - for entry in entries_out: - n_entries += 1 - for link in entry.get('links', []): - n_links += 1 - link.setdefault('source_id', source.manifest.id) - link.setdefault('requires_auth', int(bool(source.manifest.auth_required))) - return n_entries, n_links - - -def process_platforms( - platforms: dict[str, list[dict[str, Any]]], - registry: Any, - ctx: BuildContext, - source_filter: Iterable[str] | None = None, -) -> dict[str, dict[str, int]]: - """Iterate platforms.yml, run each source/platform pair through the pipeline. - - Returns per-source aggregate counts so source_health can be recorded - once the build is complete. - """ - filter_set = set(source_filter) if source_filter else None - source_stats: dict[str, dict[str, int]] = {} - - for platform, entries in platforms.items(): - if filter_set is not None: - entries = [e for e in entries if e.get('source') in filter_set] - if not entries: - continue - - print(f"\n{platform}:") - for i, entry in enumerate(entries, start=1): - source_id, config = _build_platform_config(entry) - _print_source_header(i, source_id, config) - - source = registry.get(source_id) - if source is None: - print(f"Source '{source_id}' not found in registry.") - sys.exit(1) - - entries_out = source.scrape(platform, config, ctx) - - for parser_name, parser_flags in config.parsers.items(): - parser = get_parser(parser_name) - if not parser: - print(f"Parser '{parser_name}' not found.") - sys.exit(1) - entries_out = parser.parse(entries_out, parser_flags) - - # RetroAchievements enrichment runs globally (not per-platform in - # platforms.yml): it must see the cleaned title, so it runs after - # the configured parsers, and it no-ops for platforms RA doesn't - # support. RA_CONSOLES is the single source of per-platform opt-in. - entries_out = retroachievements.parse(entries_out, {}) - - entries_out = list(entries_out) - ne, nl = _tag_links(entries_out, source) - stats = source_stats.setdefault(source_id, {'entries': 0, 'links': 0}) - stats['entries'] += ne - stats['links'] += nl - - for entry_out in entries_out: - db_manager.insert_entry(entry_out) - - return source_stats - - -def build_entry_groups() -> None: - """Run every grouping strategy over the inserted entries and persist - the resulting groups. Runs after all sources are scraped so it sees the - full catalog; strategies are discovered as plugins under db/grouping/.""" - strategies = load_strategies() - if not strategies: - return - entries = [EntryRef(*row) for row in db_manager.fetch_entries_for_grouping()] - groups = build_groups(entries, strategies) - db_manager.store_entry_groups(groups) - grouped = sum(len(g.members) for g in groups) - print(f"\nGrouping: {len(groups)} groups covering {grouped} entries " - f"({', '.join(s.kind for s in strategies)}).") - - -def make( - use_cached: bool = False, - platforms_file: str | Path = 'platforms.yml', - source_filter: Iterable[str] | None = None, -) -> None: - """Initialize the database, run the pipeline, finalize.""" - db_root = Path(__file__).resolve().parent - registry = load_registry(db_root) - platforms = load_platforms(platforms_file) - db_manager.init_database() - - # Up-front so source_health always has a referent even for skipped sources. - for manifest in registry.manifests.values(): - db_manager.register_source(manifest) - - if source_filter: - print(f"Filtering to sources: {', '.join(source_filter)}") - - ctx = BuildContext(use_cached=use_cached) - source_stats = process_platforms(platforms, registry, ctx, source_filter) - - build_entry_groups() - - for source_id in registry.ids(): - stats = source_stats.get(source_id) - if stats is None: - db_manager.record_source_health( - source_id, - status='unknown', - reason='not run in this build', - ) - else: - db_manager.record_source_health( - source_id, - status='ok', - entry_count=stats['entries'], - link_count=stats['links'], - ) - - db_manager.close_database() - print("Database created successfully.") - - -def _parse_args(argv: list[str]) -> dict[str, Any]: - args = argv[1:] if len(argv) > 1 else [] - out = { - 'use_cached': '--use-cached' in args, - 'platforms_file': 'platforms.yml', - 'source_filter': None, - } - for i, arg in enumerate(args): - if arg == '--platforms' and i + 1 < len(args): - out['platforms_file'] = args[i + 1] - elif arg in ('--sources', '--scrapers') and i + 1 < len(args): - out['source_filter'] = [s.strip() for s in args[i + 1].split(',')] - return out - - -if __name__ == '__main__': - os.chdir(os.path.dirname(os.path.realpath(__file__))) - opts = _parse_args(sys.argv) - make( - use_cached=opts['use_cached'], - platforms_file=opts['platforms_file'], - source_filter=opts['source_filter'], - ) diff --git a/db/parsers/gametdb.py b/db/parsers/gametdb.py deleted file mode 100644 index ac29b052..00000000 --- a/db/parsers/gametdb.py +++ /dev/null @@ -1,393 +0,0 @@ -""" -This module provides functionality for parsing game data from GameTDB XML files, -retrieving box art URLs, and enriching game entries with additional metadata. -""" -import re -import xml.etree.ElementTree as ET -from typing import Any -from utils.parse_utils import create_search_key - -# List of XML filenames containing game data -XML_FILENAMES = [ - 'dstdb.xml', - 'wiitdb.xml', - '3dstdb.xml', - 'wiiutdb.xml', - 'ps3tdb.xml' -] - -# Mapping of platforms to their respective XML files -PLATFORM_XML_MAP = { - 'nds': 'dstdb.xml', - 'dsi': 'dstdb.xml', - 'wii': 'wiitdb.xml', - 'gc': 'wiitdb.xml', - '3ds': '3dstdb.xml', - 'n3ds': '3dstdb.xml', - 'wiiu': 'wiiutdb.xml', - 'ps3': 'ps3tdb.xml' -} - -# Mapping of game types to platforms for each XML file -TYPE_PLATFORM_MAP = { - 'dstdb.xml': { - 'DS': 'nds', - 'DSi': 'dsi', - 'DSiWare': 'dsi', - 'CUSTOM': 'nds' - }, - 'wiitdb.xml': { - 'WiiWare': 'wii', - 'VC-NES': 'wii', - 'VC-SNES': 'wii', - 'VC-N64': 'wii', - 'VC-SMS': 'wii', - 'VC-MD': 'wii', - 'VC-PCE': 'wii', - 'VC-NEOGEO': 'wii', - 'VC-Arcade': 'wii', - 'VC-C64': 'wii', - 'VC-MSX': 'wii', - 'Channel': 'wii', - 'GameCube': 'gc', - 'Homebrew': 'wii', - 'CUSTOM': 'wii' - }, - '3dstdb.xml': { - '3DS': '3ds', - 'None': '3ds', - '3DSWare': '3ds', - 'New3DS': 'n3ds', - 'New3DSWare': 'n3ds', - 'VC-NES': '3ds', - 'VC-GB': '3ds', - 'VC-GBC': '3ds', - 'VC-GBA': '3ds', - 'VC-GG': '3ds', - 'CUSTOM': '3ds', - 'Homebrew': '3ds' - }, - 'wiiutdb.xml': { - 'WiiU': 'wiiu', - 'eShop': 'wiiu', - 'VC-NES': 'wiiu', - 'VC-SNES': 'wiiu', - 'VC-N64': 'wiiu', - 'VC-GBA': 'wiiu', - 'VC-DS': 'wiiu', - 'VC-PCE': 'wiiu', - 'VC-MSX': 'wiiu', - 'Channel': 'wiiu', - 'CUSTOM': 'wiiu' - }, - 'ps3tdb.xml': { - 'PS3': 'ps3', - 'CUSTOM': 'ps3', - 'SEN': 'ps3', - 'Homebrew': 'ps3' - } -} - -# Mapping of regions to database region codes -REGION_REGION_MAP = { - 'NTSC-U': 'us', - 'NTSC-J': 'jp', - 'PAL': 'eu', - 'NTSC-K': 'other', - 'NTSC-T': 'other', - 'PAL-R': 'other', - 'NTSC-A': 'other' -} - -# Patterns for capturing region codes in game IDs -ID_REGION_CODE_PATTERN_MAP = { - 'dstdb.xml': '.{3}(.)', - 'wiitdb.xml': '.{3}(.)', - '3dstdb.xml': '.{3}(.)', - 'wiiutdb.xml': '.{3}(.)', - 'ps3tdb.xml': '([A-Z]{4})' -} - -# List of supported countries for GameTDB artwork -GAMETDB_COUNTRIES = [ - 'US', 'EN', 'JA', 'FR', 'DE', 'ES', 'IT', 'NL', 'PT', 'NO', 'FI', 'SE', - 'ZH', 'KO', 'RU', 'AU', 'DK', 'other' -] - -# Mapping of region codes to countries for each XML file -REGION_CODE_COUNTRY_MAP = { - 'dstdb.xml': { - r"E": 'US', - r"J": 'JA', - r"K": 'KO', - r"D": 'DE', - r"F": 'FR', - r"H": 'NL', - r"I": 'IT', - r"S": 'ES', - r"Z": 'SE', - r"N": 'NO', - r"Q": 'DK', - r"M": 'SE', - r"G": 'GR', - r"T": 'US', - r"": 'EN' - }, - 'wiitdb.xml': { - r"E": 'US', - r"J": 'JA', - r"D": 'DE', - r"F": 'FR', - r"S": 'ES', - r"M": 'SE', - r"Y": 'DE', - r"K": 'KO', - r"H": 'NL', - r"I": 'IT', - r"Z": 'ES', - r"": 'EN' - }, - '3dstdb.xml': { - r"J": 'JA', - r"E": 'US', - r"K": 'KO', - r"D": 'DE', - r"W": 'ZH', - r"I": 'IT', - r"H": 'NL', - r"V": 'IT', - r"": 'EN' - }, - 'wiiutdb.xml': { - r"E": 'US', - r"J": 'JA', - r"R": 'RU', - r"A": 'JA', - r"": 'EN' - }, - 'ps3tdb.xml': { - r"BCAS": 'ZH', - r"BCAX": 'JA', - r"BCJB": 'JA', - r"BCJN": 'JA', - r"BCJS": 'JA', - r"BCJX": 'JA', - r"BCKS": 'KO', - r"BCUS": 'US', - r"BLAS": 'ZH', - r"BLJB": 'JA', - r"BLJM": 'JA', - r"BLJS": 'JA', - r"BLKS": 'KO', - r"BLMJ": 'JA', - r"BLUS": 'US', - r"CPCS": 'JA', - r"HOP3": 'JA', - r"KTGS": 'JA', - r"XCUS": 'US', - r"..J.": 'JA', - r"..U.": 'US', - r"..H.": 'US', - r"": 'EN' - } -} - -# Patterns for capturing GameTDB IDs in game serials -SERIAL_GAMETDB_ID_PATTERN_MAP = { - 'nds': r"(\w{4})", - 'dsi': r"(\w{4})", - 'wii': r"(\w{4})", - 'gc': r"(\w{4})", - '3ds': r"(\w{4})", - 'n3ds': r"(\w{4})", - 'wiiu': r"(\w{6}|\w{4})", - 'ps3': r"(\w{4}).*(\w{5})" -} - -# Mapping of platform paths for building box art URLs -BOXART_URL_PLATFORM_PATHS_MAP = { - 'nds': 'ds/coverS', - 'dsi': 'ds/coverS', - 'wii': 'wii/cover', - 'gc': 'wii/cover', - '3ds': '3ds/coverM', - 'n3ds': '3ds/coverM', - 'wiiu': 'wiiu/coverM', - 'ps3': 'ps3/cover' -} - -# Base URL for GameTDB artwork -GAMETDB_ARTWORK_BASE_URL = 'https://art.gametdb.com' - -# Global variable to store parsed TDB data -tdbs: dict[str, list[dict[str, str]]] | None = None - -def load_tdbs() -> None: - """Load TDB data from XML files into memory.""" - global tdbs - tdbs = {} - - for xml_filename in XML_FILENAMES: - try: - tree = ET.parse(f'data/gametdb/{xml_filename}') - root = tree.getroot() - - tdbs[xml_filename] = [] - - for game in root.findall('game'): - id_elem = game.find('id') - type_elem = game.find('type') - region_elem = game.find('region') - if id_elem is None or type_elem is None or region_elem is None: - continue - tdbs[xml_filename].append( - { - 'name': game.get('name') or '', - 'id': id_elem.text or '', - 'type': type_elem.text or '', - 'region': region_elem.text or '' - } - ) - except FileNotFoundError: - print(f"Warning: {xml_filename} not found, skipping GameTDB enrichment for related platforms...") - tdbs[xml_filename] = [] - - -def build_boxart_url(platform: str, country: str, id: str) -> str: - """Build a boxart URL for a specific platform, country, and game ID.""" - file_extension = 'jpg' if platform in ( - '3ds', 'n3ds', 'wiiu', 'ps3') else 'png' - - base_path = BOXART_URL_PLATFORM_PATHS_MAP[platform] - - return f'{GAMETDB_ARTWORK_BASE_URL}/{base_path}/{country}/{id}.{file_extension}' - - -def find_full_id(id: str, platform: str) -> str | None: - """Retrieve the first game ID that contains a the given ID as a substring""" - if tdbs is None: - return None - xml_filename = PLATFORM_XML_MAP.get(platform) - if not xml_filename: - return None - for game in tdbs[xml_filename]: - if game['id'].startswith(id): - return game['id'] - return None - - -def get_boxart_url_by_id(id: str, platform: str) -> str | None: - """Retrieve the boxart URL for a game by its ID and platform.""" - xml_filename = PLATFORM_XML_MAP.get(platform) - if not xml_filename: - return None - region_code_pattern = ID_REGION_CODE_PATTERN_MAP[xml_filename] - valid_id_pattern = SERIAL_GAMETDB_ID_PATTERN_MAP[platform] - - match = re.search(valid_id_pattern, id) - if not match: - return None - valid_id = ''.join(match.groups()) - full_valid_id = find_full_id(valid_id, platform) - if not full_valid_id: - return None - - match = re.match(region_code_pattern, full_valid_id) - if not match: - return None - region_code = match.group(1) - - boxart_url = None - for pattern, country in REGION_CODE_COUNTRY_MAP[xml_filename].items(): - if not re.match(pattern, region_code): - continue - - boxart_url = build_boxart_url(platform, country, full_valid_id) - break - return boxart_url - - -def parse(entries: list[dict[str, Any]], flags: dict[str, Any]) -> list[dict[str, Any]]: - """Parse game entries and enrich them with additional data.""" - if not tdbs: - load_tdbs() - - if tdbs is None: - return entries - - parse_boxart = flags.get('parse_boxart', True) - parse_name = flags.get('parse_name', False) - - total = len(entries) - progress_interval = max(1, total // 10) # Report every 10% - - for i, entry in enumerate(entries): - if i > 0 and i % progress_interval == 0: - percent = (i * 100) // total - print(f" Enriching entries... {percent}% ({i}/{total})") - - xml_filename = PLATFORM_XML_MAP.get(entry['platform']) - if not xml_filename: - continue - - # If a rom ID is set already, parse the box art URL or name directly - if entry.get('rom_id'): - if parse_boxart: - entry['boxart_url'] = get_boxart_url_by_id( - entry['rom_id'], entry['platform']) - if parse_name: - for game in tdbs[xml_filename]: - if game['id'] != entry['rom_id']: - continue - - entry['title'] = game['name'] - break - - continue - - # We do not have a rom ID, use the logic to find the best matching game in TDB - - # Get a simple to compare value from the entry title - title_compare_value = create_search_key( - re.sub(r"\(.*", '', entry['title'])) - - regions = entry['regions'] - platform = entry['platform'] - - best_match = None - best_match_name = None - - for game in tdbs[xml_filename]: - # Skip if platform does not match - if platform != TYPE_PLATFORM_MAP[xml_filename].get(game['type'], platform): - continue - - # Skip if game region does not match any of the entry regions - game_region = REGION_REGION_MAP.get(game['region']) - if regions and game_region not in regions: - continue - - # Get a simple to compare value from the game name - name_compare_value = create_search_key( - re.sub(r"\(.*", '', game['name'])) - - # Skip if entry title is not a substring of game name - if title_compare_value not in name_compare_value: - continue - - # Update best match - if not best_match_name or len(name_compare_value) < len(best_match_name): - best_match = game - best_match_name = game['name'] - - if best_match: - if parse_boxart: - entry['boxart_url'] = get_boxart_url_by_id( - best_match['id'], platform) - if parse_name: - entry['title'] = best_match['name'] - - if total > 0: - print(f" Enriching entries... done ({total} entries)") - - return entries diff --git a/db/parsers/libretro.py b/db/parsers/libretro.py deleted file mode 100644 index dbdd0828..00000000 --- a/db/parsers/libretro.py +++ /dev/null @@ -1,407 +0,0 @@ -""" -This module provides functionality for parsing and enriching game metadata -from libretro DAT files. It includes platform-specific configurations, -functions to load and parse DAT files, and methods to enhance game entries -with ROM IDs and box art URLs. -""" -import requests -import re -from typing import Any -from urllib.parse import quote, unquote -from utils.parse_utils import remove_ext - -# Platform-specific metadata definitions -PLATFORMS = { - 'nes': { - 'system': 'Nintendo - Nintendo Entertainment System', - 'dats': [ - 'metadat/no-intro/Nintendo - Nintendo Entertainment System.dat', - 'dat/Nintendo - Nintendo Entertainment System.dat' - ] - }, - 'fds': { - 'system': 'Nintendo - Family Computer Disk System', - 'dats': [ - 'metadat/no-intro/Nintendo - Family Computer Disk System.dat' - ] - }, - 'snes': { - 'system': 'Nintendo - Super Nintendo Entertainment System', - 'dats': [ - 'metadat/no-intro/Nintendo - Super Nintendo Entertainment System.dat', - 'dat/Nintendo - Super Nintendo Entertainment System.dat' - ] - }, - 'gb': { - 'system': 'Nintendo - Game Boy', - 'dats': [ - 'metadat/no-intro/Nintendo - Game Boy.dat' - ] - }, - 'gbc': { - 'system': 'Nintendo - Game Boy Color', - 'dats': [ - 'metadat/no-intro/Nintendo - Game Boy Color.dat' - ] - }, - 'gba': { - 'system': 'Nintendo - Game Boy Advance', - 'dats': [ - 'metadat/no-intro/Nintendo - Game Boy Advance.dat' - ] - }, - 'min': { - 'system': 'Nintendo - Pokemon Mini', - 'dats': [ - 'metadat/no-intro/Nintendo - Pokemon Mini.dat' - ] - }, - 'vb': { - 'system': 'Nintendo - Virtual Boy', - 'dats': [ - 'metadat/no-intro/Nintendo - Virtual Boy.dat' - ] - }, - 'n64': { - 'system': 'Nintendo - Nintendo 64', - 'dats': [ - 'metadat/no-intro/Nintendo - Nintendo 64.dat' - ] - }, - 'ndd': { - 'system': 'Nintendo - Nintendo 64DD', - 'dats': [ - 'metadat/no-intro/Nintendo - Nintendo 64DD.dat' - ] - }, - 'gc': { - 'system': 'Nintendo - GameCube', - 'dats': [ - 'metadat/redump/Nintendo - GameCube.dat', - 'dat/Nintendo - GameCube.dat' - ] - }, - 'nds': { - 'system': 'Nintendo - Nintendo DS', - 'dats': [ - 'metadat/no-intro/Nintendo - Nintendo DS.dat', - 'metadat/no-intro/Nintendo - Nintendo DS (Download Play).dat' - ] - }, - 'dsi': { - 'system': 'Nintendo - Nintendo DSi', - 'dats': [ - 'metadat/no-intro/Nintendo - Nintendo DSi.dat' - ] - }, - 'wii': { - 'system': 'Nintendo - Wii', - 'dats': [ - 'metadat/redump/Nintendo - Wii.dat', - 'dat/Nintendo - Wii.dat', - ] - }, - '3ds': { - 'system': 'Nintendo - Nintendo 3DS', - 'dats': [ - 'metadat/no-intro/Nintendo - Nintendo 3DS.dat', - 'metadat/no-intro/Nintendo - Nintendo 3DS (Digital).dat' - ] - }, - 'n3ds': { - 'system': 'Nintendo - Nintendo 3DS', - 'dats': [ - 'metadat/no-intro/Nintendo - New Nintendo 3DS.dat', - 'metadat/no-intro/Nintendo - New Nintendo 3DS (Digital).dat' - ] - }, - 'wiiu': { - 'system': 'Nintendo - Wii U', - 'dats': [ - 'dat/Nintendo - Wii U.dat' - ] - }, - 'ps1': { - 'system': 'Sony - PlayStation', - 'dats': [ - 'metadat/redump/Sony - PlayStation.dat' - ] - }, - 'ps2': { - 'system': 'Sony - PlayStation 2', - 'dats': [ - 'metadat/redump/Sony - PlayStation 2.dat' - ] - }, - 'psp': { - 'system': 'Sony - PlayStation Portable', - 'dats': [ - 'metadat/redump/Sony - PlayStation Portable.dat', - 'metadat/no-intro/Sony - PlayStation Portable.dat', - 'metadat/no-intro/Sony - PlayStation Portable (PSN).dat', - 'metadat/no-intro/Sony - PlayStation Portable (PSX2PSP).dat', - 'metadat/no-intro/Sony - PlayStation Portable (UMD Music).dat', - 'metadat/no-intro/Sony - PlayStation Portable (UMD Video).dat', - 'dat/Sony - PlayStation Minis.dat' - ] - }, - 'ps3': { - 'system': 'Sony - PlayStation 3', - 'dats': [ - 'metadat/no-intro/Sony - PlayStation 3 (PSN).dat', - 'dat/Sony - PlayStation 3.dat' - ] - }, - 'psv': { - 'system': 'Sony - PlayStation Vita', - 'dats': [ - 'metadat/no-intro/Sony - PlayStation Vita.dat', - 'metadat/no-intro/Sony - PlayStation Vita (PSN).dat' - ] - }, - 'xbox': { - 'system': 'Microsoft - Xbox', - 'dats': [ - 'metadat/redump/Microsoft - Xbox.dat' - ] - }, - 'x360': { - 'system': 'Microsoft - Xbox 360', - 'dats': [ - 'metadat/redump/Microsoft - Xbox 360.dat', - 'metadat/no-intro/Microsoft - Xbox 360.dat', - 'metadat/no-intro/Microsoft - Xbox 360 (Digital).dat' - ] - }, - 'sms': { - 'system': 'Sega - Master System - Mark III', - 'dats': [ - 'metadat/no-intro/Sega - Master System - Mark III.dat' - ] - }, - 'gg': { - 'system': 'Sega - Game Gear', - 'dats': [ - 'metadat/no-intro/Sega - Game Gear.dat' - ] - }, - 'smd': { - 'system': 'Sega - Mega Drive - Genesis', - 'dats': [ - 'metadat/no-intro/Sega - Mega Drive - Genesis.dat' - ] - }, - 'scd': { - 'system': 'Sega - Mega-CD - Sega CD', - 'dats': [ - 'metadat/redump/Sega - Mega-CD - Sega CD.dat' - ] - }, - '32x': { - 'system': 'Sega - 32X', - 'dats': [ - 'metadat/no-intro/Sega - 32X.dat' - ] - }, - 'sat': { - 'system': 'Sega - Saturn', - 'dats': [ - 'metadat/redump/Sega - Saturn.dat', - 'dat/Sega - Saturn.dat' - ] - }, - 'dc': { - 'system': 'Sega - Dreamcast', - 'dats': [ - 'metadat/redump/Sega - Dreamcast.dat' - ] - }, - 'mame': { - 'system': 'MAME', - 'dats': [] - }, - 'fbneo': { - 'system': 'FBNeo - Arcade Games', - 'dats': [] - }, - 'a26': { - 'system': 'Atari - 2600', - 'dats': [ - 'metadat/no-intro/Atari - 2600.dat' - ] - }, - 'a52': { - 'system': 'Atari - 5200', - 'dats': [ - 'metadat/no-intro/Atari - 5200.dat' - ] - }, - 'a78': { - 'system': 'Atari - 7800', - 'dats': [ - 'metadat/no-intro/Atari - 7800.dat' - ] - }, - 'lynx': { - 'system': 'Atari - Lynx', - 'dats': [ - 'metadat/no-intro/Atari - Lynx.dat' - ] - }, - 'jag': { - 'system': 'Atari - Jaguar', - 'dats': [ - 'metadat/no-intro/Atari - Jaguar.dat' - ] - }, - 'jcd': { - 'system': 'Atari - Jaguar CD', - 'dats': [ - 'metadat/redump/Atari - Jaguar CD.dat' - ] - }, - 'tg16': { - 'system': 'NEC - PC Engine - TurboGrafx 16', - 'dats': [ - 'metadat/no-intro/NEC - PC Engine - TurboGrafx 16.dat' - ] - }, - 'tgcd': { - 'system': 'NEC - PC Engine CD - TurboGrafx-CD', - 'dats': [ - 'metadat/redump/NEC - PC Engine CD - TurboGrafx-CD.dat' - ] - }, - 'pcfx': { - 'system': 'NEC - PC-FX', - 'dats': [ - 'metadat/redump/NEC - PC-FX.dat' - ] - }, - 'pc98': { - 'system': 'NEC - PC-98', - 'dats': [ - 'metadat/redump/NEC - PC-98.dat', - 'dat/NEC - PC-98.dat' - ] - }, - 'intv': { - 'system': 'Mattel - Intellivision', - 'dats': [ - 'metadat/no-intro/Mattel - Intellivision.dat' - ] - }, - 'cv': { - 'system': 'Coleco - ColecoVision', - 'dats': [ - 'metadat/no-intro/Coleco - ColecoVision.dat' - ] - }, - '3do': { - 'system': 'The 3DO Company - 3DO', - 'dats': [ - 'metadat/redump/The 3DO Company - 3DO.dat' - ] - }, - 'cdi': { - 'system': 'Philips - CD-i', - 'dats': [ - 'metadat/redump/Philips - CD-i.dat' - ] - }, - 'ngcd': { - 'system': 'SNK - Neo Geo CD', - 'dats': [ - 'metadat/redump/SNK - Neo Geo CD.dat' - ] - } -} - -# Global variable to store parsed DATs -dbs: dict[str, dict[str, str]] | None = None - - -def load_dbs() -> None: - """Load and parse the libretro DAT files for each platform.""" - global dbs - dbs = {} - - for platform, data in PLATFORMS.items(): - dbs[platform] = {} - - for dat_filename in data['dats']: - # Open and read the .dat file - with open(f'data/libretro/{dat_filename}', encoding='utf-8') as f: - lines = f.readlines() - - game = None - in_rom_section = False - for line in lines: - line = line.strip() - if line.startswith('game ('): - # Start of a new game entry - game = {} - in_rom_section = False - elif line.startswith('rom ('): - # Start of a ROM section - in_rom_section = True - if line.endswith(')'): - # End of ROM section - in_rom_section = False - elif line == ')': - # End of a game entry - if in_rom_section: - in_rom_section = False - elif game is not None: - # Save game data if both name and serial are present - if 'name' in game and 'serial' in game: - # Do not overwrite if present - if game['name'] not in dbs[platform]: - dbs[platform][game['name']] = game['serial'] - game = None - elif not in_rom_section: - # Parse game name and serial - if line.startswith('name') and game is not None: - game['name'] = line.split('"', 1)[1].rsplit('"', 1)[0] - elif line.startswith('serial') and game is not None: - game['serial'] = line.split( - '"', 1)[1].rsplit('"', 1)[0] - - -def parse(entries: list[dict[str, Any]], flags: dict[str, Any]) -> list[dict[str, Any]]: - """Parse a list of entries and enrich them with ROM IDs and box art URLs.""" - if not dbs: - load_dbs() - - if dbs is None: - return entries - - for entry in entries: - platform_id = entry['platform'] - platform_info = PLATFORMS.get(platform_id) - - if not platform_info: - continue - - db = dbs.get(platform_id) - entry['rom_id'] = db.get(entry['title']) if db else None - - try: - index_url = f"https://thumbnails.libretro.com/{quote(platform_info['system'])}/Named_Boxarts/" - - if 'available_boxarts' not in platform_info: - platform_info['available_boxarts'] = [] - r = requests.get(index_url, timeout=30) - - results = re.findall( - r".*alt=\"\[IMG\]\".*?href=\"(.*?)\".*?>.*?", r.text) - for result in results: - platform_info['available_boxarts'].append( - remove_ext(unquote(result))) - - if entry['title'] in platform_info['available_boxarts']: - entry['boxart_url'] = f"{index_url}{quote(entry['title'])}.png" - except Exception as e: - print(f"Warning: libretro enrichment failed for {platform_id}/{entry.get('title', '?')}: {e}") - - return entries diff --git a/db/parsers/mame.py b/db/parsers/mame.py deleted file mode 100644 index 7f7e40c0..00000000 --- a/db/parsers/mame.py +++ /dev/null @@ -1,53 +0,0 @@ -""" -This module provides functionality to parse and update entries based on ROM data -extracted from XML files in the MAME software directory. -""" -import os -import xml.etree.ElementTree as ET -from typing import Any - -# Directory containing XML files with MAME software data -XMLS_DIR = 'data/mame/hash' - -# Global dictionary to store ROMs data -roms: dict[str, str] | None = None - - -def load_roms() -> None: - """Load ROM data from XML files in the specified directory.""" - global roms - roms = {} - - for filename in os.listdir(XMLS_DIR): - if not filename.endswith('.xml'): - continue - - filepath = os.path.join(XMLS_DIR, filename) - - tree = ET.parse(filepath) - root = tree.getroot() - - for software in root.findall('software'): - name = software.get('name') - desc_elem = software.find('description') - description = desc_elem.text if desc_elem is not None else None - if name is not None and description is not None: - roms[name] = description - - -def parse(entries: list[dict[str, Any]], flags: dict[str, Any]) -> list[dict[str, Any]]: - """Parse a list of entries and update their titles based on ROM data.""" - if roms is None: - load_roms() - - if roms is None: - return entries - - for entry in entries: - # Check if the entry's title matches a ROM name - if entry['title'] in roms: - entry['rom_id'] = entry['title'] - # Update the title with the ROM description - entry['title'] = roms[entry['title']] - - return entries diff --git a/db/parsers/no_intro.py b/db/parsers/no_intro.py deleted file mode 100644 index 05da0324..00000000 --- a/db/parsers/no_intro.py +++ /dev/null @@ -1,188 +0,0 @@ -""" -This module provides utilities for parsing and processing game titles, -primarily following the No-Intro naming convention. It includes functions -to extract regions, clean up titles, and normalize their structure. -""" -import re -from typing import Any -from utils.parse_utils import normalize_repeated_chars - -# Mapping of regions to their respective database region -REGIONS_MAP = { - 'USA': 'us', - 'Canada': 'us', - 'Mexico': 'us', - 'Europe': 'eu', - 'Australia': 'eu', - 'Italy': 'eu', - 'Germany': 'eu', - 'France': 'eu', - 'Spain': 'eu', - 'United Kingdom': 'eu', - 'UK': 'eu', - 'Netherlands': 'eu', - 'Austria': 'eu', - 'Belgium': 'eu', - 'Croatia': 'eu', - 'Denmark': 'eu', - 'Finland': 'eu', - 'Greece': 'eu', - 'Ireland': 'eu', - 'Poland': 'eu', - 'Portugal': 'eu', - 'Sweden': 'eu', - 'Turkey': 'eu', - 'Japan': 'jp', - 'Argentina': 'other', - 'Brazil': 'other', - 'China': 'other', - 'Hong Kong': 'other', - 'India': 'other', - 'Israel': 'other', - 'Korea': 'other', - 'Latin America': 'other', - 'New Zealand': 'other', - 'Norway': 'other', - 'Russia': 'other', - 'Scandinavia': 'other', - 'South Africa': 'other', - 'Switzerland': 'other', - 'Taiwan': 'other', - 'United Arab Emirates': 'other', - 'Asia': 'other', - 'Unknown': 'other' -} - -# List of possible languages described in a title -LANGUAGES = [ - 'En', 'Ja', 'Fr', 'De', 'Es', 'It', 'Nl', 'Pt', 'Sv', 'No', 'Da', 'Fi', - 'Zh', 'Ko', 'Pl', 'Ru', 'Cs', 'Hu', 'Zh-Hant', 'Zh-Hans', 'El', 'Es-XL', - 'Pt-BR', 'Tr', 'En-GB', 'Ar', 'En+En', 'It+En', 'Ro', 'Af' -] - -# List of contents that are in parentheses to remove from titles -TITLE_REMOVE_LIST = [ - 'Europe', 'USA', 'Japan', 'World' -] + LANGUAGES - -# List of articles to handle in titles -WORD_ARTICLES = [ - 'the', 'die', 'la', 'des', 'das', 'le', 'l\'', 'ein', 'der', 'het', 'el', - 'il', 'i', 'los', 'os' -] - - -def parse_regions(title: str) -> list[str]: - """Parse the regions from a title.""" - # Extract all groups of parentheses from the title - matches = re.findall(r"\((.*?)\)", title) - - # Split the contents of each parentheses group into subgroups - groups = [group.split(',') for group in matches] - - regions = [] - for group in groups: - for content in group: - content = content.strip() - region = REGIONS_MAP.get(content) - if region and region not in regions: - regions.append(region) - # Stop processing further groups if regions are found - if regions: - break - - return regions - - -def remove_groups_with_contents(title: str, contents_to_remove: list[str]) -> str: - """Remove parentheses groups containing specific contents.""" - - # Construct a regex pattern to match parentheses groups containing any of the specified contents - contents_pattern = '|'.join(contents_to_remove) - pattern = rf"\((?:{contents_pattern})(?:,(?:{contents_pattern}))*\)" - - return re.sub(pattern, '', title) - - -def move_article(title: str) -> str: - """Move the article in a title to the beginning.""" - # Match the title structure: main name, article, and optional extra info - match = re.match(r"^(.*?),\s*(\S+)(?:\s+(.*))?$", title) - - if match: - name = match.group(1) - - # Avoid changes if the main name contains parentheses (to prevent false positives) - if '(' in name: - return title - - article = match.group(2) - other = match.group(3) - - # Construct the final title - if other: - if article.endswith("'"): - return f'{article}{name} {other}' - return f'{article} {name} {other}' - else: - if article.endswith("'"): - return f'{article}{name}' - return f'{article} {name}' - - return title - - -def get_clean_title(title: str) -> str: - """Clean the title by removing unnecessary groups and normalizing it.""" - clean_title = title - - # Extract all groups of parentheses from the title - matches = re.findall(r"\((.*?)\)", title) - - # Split the contents of each parentheses group into subgroups - groups = [group.split(',') for group in matches] - - # Remove parentheses groups if they match specific criteria - for group in groups: - remove_group = True - for content in group: - content = content.strip() - if content not in REGIONS_MAP and content not in LANGUAGES and content not in TITLE_REMOVE_LIST: - remove_group = False - break - if content in REGIONS_MAP and content not in TITLE_REMOVE_LIST: - remove_group = False - break - if remove_group: - clean_title = remove_groups_with_contents(clean_title, group) - - # Normalize repeated spaces and move articles - clean_title = normalize_repeated_chars(clean_title, ' ') - - return clean_title - - -def process_entry(entry: dict[str, Any], parse_title_regions: bool, clean_title_contents: bool, move_title_article: bool) -> None: - """Process a single entry by applying various transformations.""" - if parse_title_regions: - if not entry.get('regions'): - entry['regions'] = parse_regions(entry['title']) - - if clean_title_contents: - entry['title'] = get_clean_title(entry['title']) - - if move_title_article: - entry['title'] = move_article(entry['title']) - - -def parse(entries: list[dict[str, Any]], flags: dict[str, Any]) -> list[dict[str, Any]]: - """Parse a list of entries and process each one.""" - parse_title_regions = flags.get('parse_title_regions', True) - clean_title_contents = flags.get('clean_title_contents', True) - move_title_article = flags.get('move_title_article', True) - - for entry in entries: - process_entry(entry, parse_title_regions, - clean_title_contents, move_title_article) - - return entries diff --git a/db/parsers/retroachievements.py b/db/parsers/retroachievements.py deleted file mode 100644 index dd912100..00000000 --- a/db/parsers/retroachievements.py +++ /dev/null @@ -1,153 +0,0 @@ -""" -Enrich entries with RetroAchievements (RA) support info. - -RA identifies a game by an exact ROM hash, which this catalog does not -compute (we only have download links, not ROM files). So we match at the -*title* level per console: for each RA-supported platform we load the list -of games that have achievements -- fetched ahead of time by -scripts/download_retroachievements.py into data/retroachievements/.json --- and match each entry by a normalized title. - -A match sets two fields on the entry, which db_manager persists: - entry['ra_game_id'] -> RA game id (badge + deep link) - entry['ra_num_achievements'] -> number of achievements in the set - -This runs after no_intro (see make.process_platforms) so it sees the cleaned -title rather than the raw No-Intro filename. Platforms RA does not support are -absent from RA_CONSOLES and are silently skipped. -""" -import json -import os -from typing import Any - -from utils.parse_utils import create_search_key - -# Our platform id -> RetroAchievements console id. Only platforms RA supports -# appear here; anything not listed is a no-op. IDs are stable RA console ids. -RA_CONSOLES = { - 'nes': 7, - 'snes': 3, - 'gb': 4, - 'gbc': 6, - 'gba': 5, - 'vb': 28, - 'n64': 2, - 'gc': 16, - 'nds': 18, - 'dsi': 78, - 'ps1': 12, - 'ps2': 21, - 'psp': 41, - 'sms': 11, - 'gg': 15, - 'smd': 1, - 'scd': 9, - '32x': 10, - 'sat': 39, - 'dc': 40, - 'a26': 25, - 'a52': 50, - 'a78': 51, - 'lynx': 13, - 'jag': 17, - 'jcd': 77, - 'tg16': 8, - 'tgcd': 76, - 'pcfx': 49, - 'intv': 45, - 'cv': 44, - '3do': 43, - 'ngcd': 56, - 'min': 24, - 'mame': 27, -} - -# Where download_retroachievements.py writes the per-console JSON lists. -DATA_DIR = 'data/retroachievements' - -# platform id -> { normalized_title: (ra_game_id, num_achievements) }. -# Lazily populated by load_dbs() and cached for the rest of the build. -dbs: dict[str, dict[str, tuple[int, int]]] | None = None - - -def ra_normalize(title: str) -> str: - """Normalize a title for matching, reusing the app's own search key logic.""" - return create_search_key(title) - - -def load_dbs() -> None: - """Load each RA-supported console's game list into the title->info index. - - Missing or malformed files are skipped (offline / partial-download builds - just produce fewer matches rather than failing). - """ - global dbs - dbs = {} - - for platform_id, console_id in RA_CONSOLES.items(): - path = os.path.join(DATA_DIR, f'{console_id}.json') - try: - with open(path, encoding='utf-8') as f: - games = json.load(f) - except (FileNotFoundError, json.JSONDecodeError): - continue - - if not isinstance(games, list): - continue - - index: dict[str, tuple[int, int]] = {} - for game in games: - if not isinstance(game, dict): - continue - num = game.get('NumAchievements') or 0 - if num <= 0: - continue - title = game.get('Title') - game_id = game.get('ID') - if not title or game_id is None: - continue - key = ra_normalize(title) - if not key: - continue - # On normalized-title collision keep the richer set (deterministic). - existing = index.get(key) - if existing is None or num > existing[1]: - index[key] = (int(game_id), int(num)) - - if index: - dbs[platform_id] = index - - -def parse(entries: list[dict[str, Any]], flags: dict[str, Any]) -> list[dict[str, Any]]: - """Enrich entries on RA-supported platforms with RA game id + achievement count.""" - if dbs is None: - load_dbs() - - if not dbs: - return entries - - min_achievements = (flags or {}).get('min_achievements', 1) - stats: dict[str, list[int]] = {} - - for entry in entries: - platform = entry.get('platform', '') - if platform not in RA_CONSOLES: - continue - - counts = stats.setdefault(platform, [0, 0]) - counts[1] += 1 - - index = dbs.get(platform) - if not index: - continue - - match = index.get(ra_normalize(entry.get('title', ''))) - if match and match[1] >= min_achievements: - entry['ra_game_id'] = match[0] - entry['ra_num_achievements'] = match[1] - counts[0] += 1 - - for platform, (matched, total) in stats.items(): - print(f" RetroAchievements: matched {matched}/{total} {platform} entries") - - return entries diff --git a/db/parsers/wii_rom_set_by_ghostware.py b/db/parsers/wii_rom_set_by_ghostware.py deleted file mode 100644 index 2115bcf1..00000000 --- a/db/parsers/wii_rom_set_by_ghostware.py +++ /dev/null @@ -1,38 +0,0 @@ -""" -This module provides utilities for parsing and processing entries specifically -from the "WiiRomSetByGhostware" source. -It includes functions to extract ROM IDs from titles, clean up title strings, -and process a list of entries. -""" -import re -from typing import Any - -TITLE_ID_PATTERN = r"[_[({ ]{1,2}([A-Z0-9]{6}).*" - - -def parse_id(name: str) -> str | None: - """Extract the ROM ID from the given title string.""" - match = re.search(TITLE_ID_PATTERN, name) - if not match: - return None - - return match.group(1) - - -def get_clean_title(name: str) -> str: - """Clean the title string by removing the ROM ID and extra characters.""" - return re.sub(TITLE_ID_PATTERN, '', name).strip() - - -def process_entry(entry: dict[str, Any]) -> None: - """Process a single entry by extracting the ROM ID and cleaning the title.""" - entry['rom_id'] = parse_id(entry['title']) - entry['title'] = get_clean_title(entry['title']) - - -def parse(entries: list[dict[str, Any]], flags: dict[str, Any]) -> list[dict[str, Any]]: - """Process a list of entries by extracting ROM IDs and cleaning titles.""" - for entry in entries: - process_entry(entry) - - return entries diff --git a/db/platforms.yml b/db/platforms.yml deleted file mode 100644 index 1cb13c28..00000000 --- a/db/platforms.yml +++ /dev/null @@ -1,1660 +0,0 @@ -# Per-platform source routing. -# `source` must match a folder name under db/sources/. - -nes: -- source: minerva - format: nes - urls: - - ./No-Intro/Nintendo - Nintendo Entertainment System (Headered)/ - - ./No-Intro/Nintendo - Nintendo Entertainment System (Headerless) (Private)/ - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} -fds: -- source: minerva - format: fds - urls: - - ./No-Intro/Nintendo - Family Computer Disk System (FDS)/ - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} -- source: minerva - format: qd - urls: - - ./No-Intro/Nintendo - Family Computer Disk System (QD)/ - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} -snes: -- source: minerva - format: sfc - urls: - - ./No-Intro/Nintendo - Super Nintendo Entertainment System (Private)/ - - ./No-Intro/Nintendo - Super Nintendo Entertainment System/ - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} -gb: -- source: minerva - format: gb - urls: - - ./No-Intro/Nintendo - Game Boy (Private)/ - - ./No-Intro/Nintendo - Game Boy/ - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} -gbc: -- source: minerva - format: gbc - urls: - - ./No-Intro/Nintendo - Game Boy Color (Private)/ - - ./No-Intro/Nintendo - Game Boy Color/ - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} -gba: -- source: minerva - format: gba - urls: - - ./No-Intro/Nintendo - Game Boy Advance (Multiboot)/ - - ./No-Intro/Nintendo - Game Boy Advance (Private)/ - - ./No-Intro/Nintendo - Game Boy Advance (Video)/ - - ./No-Intro/Nintendo - Game Boy Advance/ - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} -- source: minerva - format: asf - urls: - - ./No-Intro/Nintendo - Game Boy Advance (Play-Yan)/ - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} -min: -- source: minerva - format: min - urls: - - ./No-Intro/Nintendo - Pokemon Mini/ - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} -vb: -- source: minerva - format: vb - urls: - - ./No-Intro/Nintendo - Virtual Boy (Private)/ - - ./No-Intro/Nintendo - Virtual Boy/ - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} -n64: -- source: minerva - format: z64 - urls: - - ./No-Intro/Nintendo - Nintendo 64 (BigEndian) (Private)/ - - ./No-Intro/Nintendo - Nintendo 64 (BigEndian)/ - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} -ndd: -- source: minerva - format: ndd - urls: - - ./No-Intro/Nintendo - Nintendo 64DD/ - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} -gc: -- source: minerva - format: rvz - urls: - - ./Redump/Nintendo - GameCube - NKit RVZ [zstd-19-128k]/ - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} - gametdb: {} -- source: internet_archive - format: iso - urls: - - https://archive.org/download/AsiaGamecubeCollectionByGhostware - filter: (.*)\.iso - type: Game - parsers: - no_intro: - parse_title_regions: false - clean_title_contents: false - move_title_article: true - gametdb: {} -- source: internet_archive - format: iso - urls: - - https://archive.org/download/NCubeJ - filter: (.*)\.7z - type: Game - parsers: - libretro: {} - no_intro: {} - gametdb: {} -nds: -- source: minerva - format: nds - urls: - - ./No-Intro/Nintendo - Nintendo DS (Decrypted) (Private)/ - - ./No-Intro/Nintendo - Nintendo DS (Decrypted)/ - - ./No-Intro/Nintendo - Nintendo DS (Download Play)/ - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} - gametdb: {} -- source: minerva - format: img - urls: - - ./No-Intro/Nintendo - Nintendo DS (DSvision SD cards)/ - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} - gametdb: {} -dsi: -- source: minerva - format: dsi - urls: - - ./No-Intro/Nintendo - Nintendo DSi (Decrypted)/ - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} - gametdb: {} -- source: mariocube - format: nds - urls: - - https://repo.mariocube.com/DSiWare/NDS/%23/ - - https://repo.mariocube.com/DSiWare/NDS/A/ - - https://repo.mariocube.com/DSiWare/NDS/B/ - - https://repo.mariocube.com/DSiWare/NDS/C/ - - https://repo.mariocube.com/DSiWare/NDS/D/ - - https://repo.mariocube.com/DSiWare/NDS/E/ - - https://repo.mariocube.com/DSiWare/NDS/F/ - - https://repo.mariocube.com/DSiWare/NDS/G/ - - https://repo.mariocube.com/DSiWare/NDS/H/ - - https://repo.mariocube.com/DSiWare/NDS/I/ - - https://repo.mariocube.com/DSiWare/NDS/J/ - - https://repo.mariocube.com/DSiWare/NDS/K/ - - https://repo.mariocube.com/DSiWare/NDS/L/ - - https://repo.mariocube.com/DSiWare/NDS/M/ - - https://repo.mariocube.com/DSiWare/NDS/N/ - - https://repo.mariocube.com/DSiWare/NDS/O/ - - https://repo.mariocube.com/DSiWare/NDS/P/ - - https://repo.mariocube.com/DSiWare/NDS/Q/ - - https://repo.mariocube.com/DSiWare/NDS/R/ - - https://repo.mariocube.com/DSiWare/NDS/S/ - - https://repo.mariocube.com/DSiWare/NDS/T/ - - https://repo.mariocube.com/DSiWare/NDS/U/ - - https://repo.mariocube.com/DSiWare/NDS/V/ - - https://repo.mariocube.com/DSiWare/NDS/W/ - - https://repo.mariocube.com/DSiWare/NDS/X/ - - https://repo.mariocube.com/DSiWare/NDS/Y/ - - https://repo.mariocube.com/DSiWare/NDS/Z/ - filter: (.*)\.nds - type: Game - parsers: - libretro: {} - no_intro: {} - gametdb: {} -- source: internet_archive - format: nds - urls: - - https://archive.org/download/MarioCubeLite/DSiWare/NDS/%23/ - - https://archive.org/download/MarioCubeLite/DSiWare/NDS/A/ - - https://archive.org/download/MarioCubeLite/DSiWare/NDS/B/ - - https://archive.org/download/MarioCubeLite/DSiWare/NDS/C/ - - https://archive.org/download/MarioCubeLite/DSiWare/NDS/D/ - - https://archive.org/download/MarioCubeLite/DSiWare/NDS/E/ - - https://archive.org/download/MarioCubeLite/DSiWare/NDS/F/ - - https://archive.org/download/MarioCubeLite/DSiWare/NDS/G/ - - https://archive.org/download/MarioCubeLite/DSiWare/NDS/H/ - - https://archive.org/download/MarioCubeLite/DSiWare/NDS/I/ - - https://archive.org/download/MarioCubeLite/DSiWare/NDS/J/ - - https://archive.org/download/MarioCubeLite/DSiWare/NDS/K/ - - https://archive.org/download/MarioCubeLite/DSiWare/NDS/L/ - - https://archive.org/download/MarioCubeLite/DSiWare/NDS/M/ - - https://archive.org/download/MarioCubeLite/DSiWare/NDS/N/ - - https://archive.org/download/MarioCubeLite/DSiWare/NDS/O/ - - https://archive.org/download/MarioCubeLite/DSiWare/NDS/P/ - - https://archive.org/download/MarioCubeLite/DSiWare/NDS/Q/ - - https://archive.org/download/MarioCubeLite/DSiWare/NDS/R/ - - https://archive.org/download/MarioCubeLite/DSiWare/NDS/S/ - - https://archive.org/download/MarioCubeLite/DSiWare/NDS/T/ - - https://archive.org/download/MarioCubeLite/DSiWare/NDS/U/ - - https://archive.org/download/MarioCubeLite/DSiWare/NDS/V/ - - https://archive.org/download/MarioCubeLite/DSiWare/NDS/W/ - - https://archive.org/download/MarioCubeLite/DSiWare/NDS/X/ - - https://archive.org/download/MarioCubeLite/DSiWare/NDS/Y/ - - https://archive.org/download/MarioCubeLite/DSiWare/NDS/Z/ - filter: (.*)\.nds - type: Game - parsers: - libretro: {} - no_intro: {} - gametdb: {} -- source: minerva - format: DUMP - urls: - - ./No-Intro/Nintendo - Nintendo DSi (Digital)/ - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} - gametdb: {} -- source: minerva - format: CDN - urls: - - ./No-Intro/Nintendo - Nintendo DSi (Digital) (CDN) (Decrypted)/ - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} - gametdb: {} -wii: -- source: minerva - format: rvz - urls: - - ./Redump/Nintendo - Wii - NKit RVZ [zstd-19-128k]/ - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} - gametdb: {} -- source: minerva - format: iso - urls: - - ./No-Intro/Non-Redump - Nintendo - Wii/ - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} - gametdb: {} -- source: minerva - format: wad - urls: - - ./No-Intro/Unofficial - Nintendo - Wii (Digital) (Deprecated) (WAD)/ - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} - gametdb: {} -- source: mariocube - format: wad - urls: - - https://repo.mariocube.com/WADs/_WiiWare,%20VC,%20DLC,%20Channels%20&%20IOS/0-9/ - - https://repo.mariocube.com/WADs/_WiiWare,%20VC,%20DLC,%20Channels%20&%20IOS/A/ - - https://repo.mariocube.com/WADs/_WiiWare,%20VC,%20DLC,%20Channels%20&%20IOS/B/ - - https://repo.mariocube.com/WADs/_WiiWare,%20VC,%20DLC,%20Channels%20&%20IOS/C/ - - https://repo.mariocube.com/WADs/_WiiWare,%20VC,%20DLC,%20Channels%20&%20IOS/D/ - - https://repo.mariocube.com/WADs/_WiiWare,%20VC,%20DLC,%20Channels%20&%20IOS/E/ - - https://repo.mariocube.com/WADs/_WiiWare,%20VC,%20DLC,%20Channels%20&%20IOS/F/ - - https://repo.mariocube.com/WADs/_WiiWare,%20VC,%20DLC,%20Channels%20&%20IOS/G/ - - https://repo.mariocube.com/WADs/_WiiWare,%20VC,%20DLC,%20Channels%20&%20IOS/H/ - - https://repo.mariocube.com/WADs/_WiiWare,%20VC,%20DLC,%20Channels%20&%20IOS/I/ - - https://repo.mariocube.com/WADs/_WiiWare,%20VC,%20DLC,%20Channels%20&%20IOS/J/ - - https://repo.mariocube.com/WADs/_WiiWare,%20VC,%20DLC,%20Channels%20&%20IOS/K/ - - https://repo.mariocube.com/WADs/_WiiWare,%20VC,%20DLC,%20Channels%20&%20IOS/L/ - - https://repo.mariocube.com/WADs/_WiiWare,%20VC,%20DLC,%20Channels%20&%20IOS/M/ - - https://repo.mariocube.com/WADs/_WiiWare,%20VC,%20DLC,%20Channels%20&%20IOS/N/ - - https://repo.mariocube.com/WADs/_WiiWare,%20VC,%20DLC,%20Channels%20&%20IOS/O/ - - https://repo.mariocube.com/WADs/_WiiWare,%20VC,%20DLC,%20Channels%20&%20IOS/P/ - - https://repo.mariocube.com/WADs/_WiiWare,%20VC,%20DLC,%20Channels%20&%20IOS/Q/ - - https://repo.mariocube.com/WADs/_WiiWare,%20VC,%20DLC,%20Channels%20&%20IOS/R/ - - https://repo.mariocube.com/WADs/_WiiWare,%20VC,%20DLC,%20Channels%20&%20IOS/S/ - - https://repo.mariocube.com/WADs/_WiiWare,%20VC,%20DLC,%20Channels%20&%20IOS/T/ - - https://repo.mariocube.com/WADs/_WiiWare,%20VC,%20DLC,%20Channels%20&%20IOS/U/ - - https://repo.mariocube.com/WADs/_WiiWare,%20VC,%20DLC,%20Channels%20&%20IOS/V/ - - https://repo.mariocube.com/WADs/_WiiWare,%20VC,%20DLC,%20Channels%20&%20IOS/W/ - - https://repo.mariocube.com/WADs/_WiiWare,%20VC,%20DLC,%20Channels%20&%20IOS/X/ - - https://repo.mariocube.com/WADs/_WiiWare,%20VC,%20DLC,%20Channels%20&%20IOS/Y/ - - https://repo.mariocube.com/WADs/_WiiWare,%20VC,%20DLC,%20Channels%20&%20IOS/Z/ - filter: (.*)\.wad - type: Game - parsers: - libretro: {} - no_intro: {} - gametdb: {} -- source: internet_archive - format: wad - urls: - - https://archive.org/download/MarioCubeLite/WADs/_WiiWare%2C%20VC%2C%20DLC%2C%20Channels%20%26%20IOS/%23/ - - https://archive.org/download/MarioCubeLite/WADs/_WiiWare%2C%20VC%2C%20DLC%2C%20Channels%20%26%20IOS/A/ - - https://archive.org/download/MarioCubeLite/WADs/_WiiWare%2C%20VC%2C%20DLC%2C%20Channels%20%26%20IOS/B/ - - https://archive.org/download/MarioCubeLite/WADs/_WiiWare%2C%20VC%2C%20DLC%2C%20Channels%20%26%20IOS/C/ - - https://archive.org/download/MarioCubeLite/WADs/_WiiWare%2C%20VC%2C%20DLC%2C%20Channels%20%26%20IOS/D/ - - https://archive.org/download/MarioCubeLite/WADs/_WiiWare%2C%20VC%2C%20DLC%2C%20Channels%20%26%20IOS/E/ - - https://archive.org/download/MarioCubeLite/WADs/_WiiWare%2C%20VC%2C%20DLC%2C%20Channels%20%26%20IOS/F/ - - https://archive.org/download/MarioCubeLite/WADs/_WiiWare%2C%20VC%2C%20DLC%2C%20Channels%20%26%20IOS/G/ - - https://archive.org/download/MarioCubeLite/WADs/_WiiWare%2C%20VC%2C%20DLC%2C%20Channels%20%26%20IOS/H/ - - https://archive.org/download/MarioCubeLite/WADs/_WiiWare%2C%20VC%2C%20DLC%2C%20Channels%20%26%20IOS/I/ - - https://archive.org/download/MarioCubeLite/WADs/_WiiWare%2C%20VC%2C%20DLC%2C%20Channels%20%26%20IOS/J/ - - https://archive.org/download/MarioCubeLite/WADs/_WiiWare%2C%20VC%2C%20DLC%2C%20Channels%20%26%20IOS/K/ - - https://archive.org/download/MarioCubeLite/WADs/_WiiWare%2C%20VC%2C%20DLC%2C%20Channels%20%26%20IOS/L/ - - https://archive.org/download/MarioCubeLite/WADs/_WiiWare%2C%20VC%2C%20DLC%2C%20Channels%20%26%20IOS/M/ - - https://archive.org/download/MarioCubeLite/WADs/_WiiWare%2C%20VC%2C%20DLC%2C%20Channels%20%26%20IOS/N/ - - https://archive.org/download/MarioCubeLite/WADs/_WiiWare%2C%20VC%2C%20DLC%2C%20Channels%20%26%20IOS/O/ - - https://archive.org/download/MarioCubeLite/WADs/_WiiWare%2C%20VC%2C%20DLC%2C%20Channels%20%26%20IOS/P/ - - https://archive.org/download/MarioCubeLite/WADs/_WiiWare%2C%20VC%2C%20DLC%2C%20Channels%20%26%20IOS/Q/ - - https://archive.org/download/MarioCubeLite/WADs/_WiiWare%2C%20VC%2C%20DLC%2C%20Channels%20%26%20IOS/R/ - - https://archive.org/download/MarioCubeLite/WADs/_WiiWare%2C%20VC%2C%20DLC%2C%20Channels%20%26%20IOS/S/ - - https://archive.org/download/MarioCubeLite/WADs/_WiiWare%2C%20VC%2C%20DLC%2C%20Channels%20%26%20IOS/T/ - - https://archive.org/download/MarioCubeLite/WADs/_WiiWare%2C%20VC%2C%20DLC%2C%20Channels%20%26%20IOS/U/ - - https://archive.org/download/MarioCubeLite/WADs/_WiiWare%2C%20VC%2C%20DLC%2C%20Channels%20%26%20IOS/V/ - - https://archive.org/download/MarioCubeLite/WADs/_WiiWare%2C%20VC%2C%20DLC%2C%20Channels%20%26%20IOS/W/ - - https://archive.org/download/MarioCubeLite/WADs/_WiiWare%2C%20VC%2C%20DLC%2C%20Channels%20%26%20IOS/X/ - - https://archive.org/download/MarioCubeLite/WADs/_WiiWare%2C%20VC%2C%20DLC%2C%20Channels%20%26%20IOS/Y/ - - https://archive.org/download/MarioCubeLite/WADs/_WiiWare%2C%20VC%2C%20DLC%2C%20Channels%20%26%20IOS/Z/ - filter: (.*)\.wad - type: Game - parsers: - libretro: {} - no_intro: {} - gametdb: {} -- source: minerva - format: DUMP - urls: - - ./No-Intro/Nintendo - Wii (Digital) (CDN)/ - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} - gametdb: {} -3ds: -- source: minerva - format: 3ds - urls: - - ./No-Intro/Nintendo - Nintendo 3DS (Decrypted)/ - - ./No-Intro/Nintendo - Nintendo 3DS (Digital) (Deprecated)/ - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} - gametdb: {} -- source: internet_archive - format: 3ds - urls: - - https://archive.org/download/3ds-main-encrypted - - https://archive.org/download/3ds-main-encrypted-p2 - filter: (.*)\.7z - type: Game (Encrypted) - parsers: - libretro: {} - no_intro: {} - gametdb: {} -- source: minerva - format: CDN - urls: - - ./No-Intro/Nintendo - Nintendo 3DS (Digital) (CDN)/ - - ./No-Intro/Nintendo - Nintendo 3DS (Digital) (Pre-Install)/ - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} - gametdb: {} -- source: minerva - format: cia - urls: - - ./No-Intro/Nintendo - Nintendo 3DS (Digital) (Dev ROMs)/ - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} - gametdb: {} -n3ds: -- source: minerva - format: 3ds - urls: - - ./No-Intro/Nintendo - New Nintendo 3DS (Decrypted)/ - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} - gametdb: {} -- source: minerva - format: CDN - urls: - - ./No-Intro/Nintendo - New Nintendo 3DS (Digital) (Deprecated)/ - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} - gametdb: {} -wiiu: -- source: minerva - format: CDN - urls: - - ./No-Intro/Nintendo - Wii U (Digital) (CDN) (Dev)/ - - ./No-Intro/Nintendo - Wii U (Digital) (CDN) (Lotcheck)/ - - ./No-Intro/Nintendo - Wii U (Digital) (CDN)/ - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} - gametdb: {} -- source: minerva - format: GAME FILES - urls: - - ./No-Intro/Unofficial - Nintendo - Wii U (Digital) (Deprecated)/ - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} - gametdb: {} -- source: minerva - format: wux - urls: - - ./Redump/Nintendo - Wii U - WUX/ - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} - gametdb: {} -- source: minerva - format: key - urls: - - ./Redump/Nintendo - Wii U - Disc Keys/ - filter: (.*)\.zip - type: Disc Key - parsers: - no_intro: {} -- source: minerva - format: wud - urls: - - ./No-Intro/Non-Redump - Nintendo - Wii U/ - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} - gametdb: {} -ps1: -- source: minerva - format: bin/cue - urls: - - ./Redump/Sony - PlayStation/ - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} -- source: internet_archive - format: bin/cue - urls: - - https://archive.org/download/sony_playstation_part1 - - https://archive.org/download/sony_playstation_part2 - - https://archive.org/download/sony_playstation_part3 - - https://archive.org/download/sony_playstation_part4 - - https://archive.org/download/sony_playstation_part5 - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} -- source: internet_archive - format: chd - regions: - - us - urls: - - https://archive.org/download/chd_psx/CHD-PSX-USA/ - filter: (.*)\.chd - type: Game - parsers: - libretro: {} - no_intro: {} -- source: internet_archive - format: chd - regions: - - eu - urls: - - https://archive.org/download/chd_psx_eur/CHD-PSX-EUR/ - filter: (.*)\.chd - type: Game - parsers: - libretro: {} - no_intro: {} -- source: internet_archive - format: chd - regions: - - jp - urls: - - https://archive.org/download/chd_psx_jap/CHD-PSX-JAP/ - - https://archive.org/download/chd_psx_jap_p2/CHD-PSX-JAP/ - filter: (.*)\.chd - type: Game - parsers: - libretro: {} - no_intro: {} -- source: internet_archive - format: chd - urls: - - https://archive.org/download/chd_psx_misc/CHD-PSX-Misc/ - filter: (.*)\.chd - type: Game - parsers: - libretro: {} - no_intro: {} -- source: minerva - format: iso - urls: - - ./No-Intro/Non-Redump - Sony - PlayStation/ - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} -ps2: -- source: minerva - format: iso - urls: - - ./Redump/Sony - PlayStation 2/ - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} -- source: internet_archive - format: iso - urls: - - https://archive.org/download/sony_playstation2_numberssymbols - - https://archive.org/download/sony_playstation2_a - - https://archive.org/download/sony_playstation2_b - - https://archive.org/download/sony_playstation2_c - - https://archive.org/download/sony_playstation2_d_part1 - - https://archive.org/download/sony_playstation2_d_part2 - - https://archive.org/download/sony_playstation2_e - - https://archive.org/download/sony_playstation2_f - - https://archive.org/download/sony_playstation2_g - - https://archive.org/download/sony_playstation2_h - - https://archive.org/download/sony_playstation2_i - - https://archive.org/download/sony_playstation2_j - - https://archive.org/download/sony_playstation2_k - - https://archive.org/download/sony_playstation2_l - - https://archive.org/download/sony_playstation2_m_part1 - - https://archive.org/download/sony_playstation2_m_part2 - - https://archive.org/download/sony_playstation2_n - - https://archive.org/download/sony_playstation2_o_part1 - - https://archive.org/download/sony_playstation2_o_part2 - - https://archive.org/download/sony_playstation2_p - - https://archive.org/download/sony_playstation2_q - - https://archive.org/download/sony_playstation2_r - - https://archive.org/download/sony_playstation2_s_part1 - - https://archive.org/download/sony_playstation2_s_part2 - - https://archive.org/download/sony_playstation2_s_part3 - - https://archive.org/download/sony_playstation2_s_part4 - - https://archive.org/download/sony_playstation2_t - - https://archive.org/download/sony_playstation2_u - - https://archive.org/download/sony_playstation2_v - - https://archive.org/download/sony_playstation2_w - - https://archive.org/download/sony_playstation2_x - - https://archive.org/download/sony_playstation2_z - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} -- source: minerva - format: bin - urls: - - ./No-Intro/Non-Redump - Sony - PlayStation 2/ - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} -psp: -- source: minerva - format: iso - urls: - - ./No-Intro/Non-Redump - Sony - PlayStation Portable/ - - ./Redump/Sony - PlayStation Portable/ - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} -- source: internet_archive - format: iso - urls: - - https://archive.org/download/sony_playstation_portable_part1 - - https://archive.org/download/sony_playstation_portable_part2 - - https://archive.org/download/sony_playstation_portable_part3 - - https://archive.org/download/sony_playstation_portable_part4 - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} -- source: internet_archive - format: chd - urls: - - https://archive.org/download/psp-chd-zstd-redump-part1/psp-chd-zstd/ - - https://archive.org/download/psp-chd-zstd-redump-part2/psp-chd-zstd/ - filter: (.*)\.chd - type: Game - parsers: - libretro: {} - no_intro: {} -- source: internet_archive - format: iso - urls: - - https://archive.org/download/PSNCollectionByGhostware - filter: (.*)\.iso - type: Game - parsers: - libretro: {} - no_intro: {} -ps3: -- source: nopaystation - format: pkg - urls: - - https://nopaystation.com/tsv/PS3_GAMES.tsv - - https://nopaystation.com/tsv/PS3_DEMOS.tsv - filter: null - type: Game - parsers: - gametdb: {} -- source: nopaystation - format: pkg - urls: - - https://nopaystation.com/tsv/PS3_DLCS.tsv - filter: null - type: DLC - parsers: - gametdb: {} -- source: minerva - format: pkg/rap - urls: - - ./No-Intro/Sony - PlayStation 3 (PSN) (Content)/ - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} - gametdb: {} -- source: internet_archive - format: pkg/rap - regions: - - us - urls: - - https://archive.org/download/PS3_NOINTRO_USA_1 - - https://archive.org/download/PS3_NOINTRO_USA_2 - - https://archive.org/download/PS3_NOINTRO_USA_3 - - https://archive.org/download/PS3_NOINTRO_USA_4 - - https://archive.org/download/PS3_NOINTRO_USA__5 - - https://archive.org/download/PS3_NOINTRO_USA_6 - - https://archive.org/download/PS3_NOINTRO_USA_7 - - https://archive.org/download/PS3_NOINTRO_USA_8 - - https://archive.org/download/PS3_NOINTRO_USA_9 - - https://archive.org/download/PS3_NOINTRO_USA_10 - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} - gametdb: {} -- source: internet_archive - format: pkg/rap - regions: - - eu - urls: - - https://archive.org/download/PS3_NOINTRO_EUR_1 - - https://archive.org/download/PS3_NOINTRO_EUR_2 - - https://archive.org/download/PS3_NOINTRO_EUR_3 - - https://archive.org/download/PS3_NOINTRO_EUR_4 - - https://archive.org/download/PS3_NOINTRO_EUR_5 - - https://archive.org/download/PS3_NOINTRO_EUR_6 - - https://archive.org/download/PS3_NOINTRO_EUR_7 - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} - gametdb: {} -- source: internet_archive - format: pkg/rap - regions: - - jp - urls: - - https://archive.org/download/PS3_NOINTRO_JAP_1 - - https://archive.org/download/PS3_NOINTRO_JAP_2 - - https://archive.org/download/PS3_NOINTRO_JAP_3 - - https://archive.org/download/PS3_NOINTRO_JAP_4 - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} - gametdb: {} -- source: minerva - format: iso - urls: - - ./Redump/Sony - PlayStation 3/ - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} - gametdb: {} -- source: internet_archive - format: iso - urls: - - https://archive.org/download/sony_playstation3_numberssymbols - - https://archive.org/download/sony_playstation3_a_part1 - - https://archive.org/download/sony_playstation3_a_part2 - - https://archive.org/download/sony_playstation3_a_part3 - - https://archive.org/download/sony_playstation3_b_part1 - - https://archive.org/download/sony_playstation3_b_part2 - - https://archive.org/download/sony_playstation3_b_part3 - - https://archive.org/download/sony_playstation3_c_part1 - - https://archive.org/download/sony_playstation3_c_part2 - - https://archive.org/download/sony_playstation3_c_part3 - - https://archive.org/download/sony_playstation3_d_part1 - - https://archive.org/download/sony_playstation3_d_part2 - - https://archive.org/download/sony_playstation3_d_part3 - - https://archive.org/download/sony_playstation3_d_part4 - - https://archive.org/download/sony_playstation3_d_part5 - - https://archive.org/download/sony_playstation3_e - - https://archive.org/download/sony_playstation3_f_part1 - - https://archive.org/download/sony_playstation3_f_part2 - - https://archive.org/download/sony_playstation3_f_part3 - - https://archive.org/download/sony_playstation3_g_part1 - - https://archive.org/download/sony_playstation3_g_part2 - - https://archive.org/download/sony_playstation3_g_part3 - - https://archive.org/download/sony_playstation3_h_part1 - - https://archive.org/download/sony_playstation3_h_part2 - - https://archive.org/download/sony_playstation3_i - - https://archive.org/download/sony_playstation3_j - - https://archive.org/download/sony_playstation3_k - - https://archive.org/download/sony_playstation3_l_part1 - - https://archive.org/download/sony_playstation3_l_part2 - - https://archive.org/download/sony_playstation3_l_part3 - - https://archive.org/download/sony_playstation3_m_part1 - - https://archive.org/download/sony_playstation3_m_part2 - - https://archive.org/download/sony_playstation3_m_part3 - - https://archive.org/download/sony_playstation3_m_part4 - - https://archive.org/download/sony_playstation3_m_part5 - - https://archive.org/download/sony_playstation3_n_part1 - - https://archive.org/download/sony_playstation3_n_part2 - - https://archive.org/download/sony_playstation3_n_part3 - - https://archive.org/download/sony_playstation3_o_part1 - - https://archive.org/download/sony_playstation3_o_part2 - - https://archive.org/download/sony_playstation3_o_part3 - - https://archive.org/download/sony_playstation3_p_part1 - - https://archive.org/download/sony_playstation3_p_part2 - - https://archive.org/download/sony_playstation3_q - - https://archive.org/download/sony_playstation3_r_part1 - - https://archive.org/download/sony_playstation3_r_part2 - - https://archive.org/download/sony_playstation3_r_part3 - - https://archive.org/download/sony_playstation3_r_part4 - - https://archive.org/download/sony_playstation3_s_part1 - - https://archive.org/download/sony_playstation3_s_part2 - - https://archive.org/download/sony_playstation3_s_part3 - - https://archive.org/download/sony_playstation3_s_part4 - - https://archive.org/download/sony_playstation3_s_part5 - - https://archive.org/download/sony_playstation3_s_part6 - - https://archive.org/download/sony_playstation3_t_part1 - - https://archive.org/download/sony_playstation3_t_part2 - - https://archive.org/download/sony_playstation3_t_part3 - - https://archive.org/download/sony_playstation3_t_part4 - - https://archive.org/download/sony_playstation3_u_part1 - - https://archive.org/download/sony_playstation3_u_part2 - - https://archive.org/download/sony_playstation3_v - - https://archive.org/download/sony_playstation3_w_part1 - - https://archive.org/download/sony_playstation3_w_part2 - - https://archive.org/download/sony_playstation3_x - - https://archive.org/download/sony_playstation3_y - - https://archive.org/download/sony_playstation3_z - filter: (.*)\.iso - type: Game - parsers: - libretro: {} - no_intro: {} - gametdb: {} -- source: minerva - format: key - urls: - - ./Redump/Sony - PlayStation 3 - Disc Keys/ - filter: (.*)\.zip - type: Disc Key - parsers: - no_intro: {} -- source: minerva - format: dkey - urls: - - ./Redump/Sony - PlayStation 3 - Disc Keys TXT/ - filter: (.*)\.zip - type: Disc Key TXT - parsers: - no_intro: {} -psv: -- source: nopaystation - format: pkg - urls: - - https://nopaystation.com/tsv/PSV_GAMES.tsv - - https://nopaystation.com/tsv/PSV_DEMOS.tsv - filter: null - type: Game - parsers: {} -- source: nopaystation - format: pkg - urls: - - https://nopaystation.com/tsv/PSV_DLCS.tsv - filter: null - type: DLC - parsers: {} -- source: minerva - format: pkg - urls: - - ./No-Intro/Sony - PlayStation Vita (PSN) (Content)/ - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} -- source: minerva - format: psv - urls: - - ./No-Intro/Unofficial - Sony - PlayStation Vita (BlackFinPSV)/ - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} -- source: minerva - format: NoNpDrm - urls: - - ./No-Intro/Unofficial - Sony - PlayStation Vita (NoNpDrm)/ - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} -- source: minerva - format: vpk - urls: - - ./No-Intro/Unofficial - Sony - PlayStation Vita (PSN) (Decrypted) (VPK)/ - - ./No-Intro/Unofficial - Sony - PlayStation Vita (VPK)/ - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} -- source: minerva - format: PSVgameSD - urls: - - ./No-Intro/Unofficial - Sony - PlayStation Vita (PSVgameSD)/ - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} -xbox: -- source: minerva - format: iso - urls: - - ./Redump/Microsoft - Xbox/ - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} -- source: internet_archive - format: iso - urls: - - https://archive.org/download/microsoft_xbox_numberssymbols - - https://archive.org/download/microsoft_xbox_a - - https://archive.org/download/microsoft_xbox_b - - https://archive.org/download/microsoft_xbox_c_part1 - - https://archive.org/download/microsoft_xbox_c_part2 - - https://archive.org/download/microsoft_xbox_d_part1 - - https://archive.org/download/microsoft_xbox_d_part2 - - https://archive.org/download/microsoft_xbox_e - - https://archive.org/download/microsoft_xbox_f - - https://archive.org/download/microsoft_xbox_g - - https://archive.org/download/microsoft_xbox_h - - https://archive.org/download/microsoft_xbox_i - - https://archive.org/download/microsoft_xbox_j - - https://archive.org/download/microsoft_xbox_k - - https://archive.org/download/microsoft_xbox_l - - https://archive.org/download/microsoft_xbox_m_part1 - - https://archive.org/download/microsoft_xbox_m_part2 - - https://archive.org/download/microsoft_xbox_n_part1 - - https://archive.org/download/microsoft_xbox_n_part2 - - https://archive.org/download/microsoft_xbox_o_part1 - - https://archive.org/download/microsoft_xbox_o_part2 - - https://archive.org/download/microsoft_xbox_p - - https://archive.org/download/microsoft_xbox_q - - https://archive.org/download/microsoft_xbox_r - - https://archive.org/download/microsoft_xbox_s_part1 - - https://archive.org/download/microsoft_xbox_s_part2 - - https://archive.org/download/microsoft_xbox_t_part1 - - https://archive.org/download/microsoft_xbox_t_part2 - - https://archive.org/download/microsoft_xbox_u - - https://archive.org/download/microsoft_xbox_v - - https://archive.org/download/microsoft_xbox_w - - https://archive.org/download/microsoft_xbox_x - - https://archive.org/download/microsoft_xbox_y - - https://archive.org/download/microsoft_xbox_z - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} -x360: -- source: minerva - format: iso - urls: - - ./Redump/Microsoft - Xbox 360/ - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} -- source: internet_archive - format: iso - urls: - - https://archive.org/download/microsoft_xbox360_numberssymbols - - https://archive.org/download/microsoft_xbox360_a_part1 - - https://archive.org/download/microsoft_xbox360_a_part2 - - https://archive.org/download/microsoft_xbox360_b_part1 - - https://archive.org/download/microsoft_xbox360_b_part2 - - https://archive.org/download/microsoft_xbox360_c_part1 - - https://archive.org/download/microsoft_xbox360_c_part2 - - https://archive.org/download/microsoft_xbox360_d_part1 - - https://archive.org/download/microsoft_xbox360_d_part2 - - https://archive.org/download/microsoft_xbox360_d_part3 - - https://archive.org/download/microsoft_xbox360_e - - https://archive.org/download/microsoft_xbox360_f_part1 - - https://archive.org/download/microsoft_xbox360_f_part2 - - https://archive.org/download/microsoft_xbox360_g - - https://archive.org/download/microsoft_xbox360_h - - https://archive.org/download/microsoft_xbox360_i - - https://archive.org/download/microsoft_xbox360_j - - https://archive.org/download/microsoft_xbox360_k - - https://archive.org/download/microsoft_xbox360_l - - https://archive.org/download/microsoft_xbox360_m_part1 - - https://archive.org/download/microsoft_xbox360_m_part2 - - https://archive.org/download/microsoft_xbox360_n_part1 - - https://archive.org/download/microsoft_xbox360_n_part2 - - https://archive.org/download/microsoft_xbox360_o - - https://archive.org/download/microsoft_xbox360_p - - https://archive.org/download/microsoft_xbox360_q - - https://archive.org/download/microsoft_xbox360_r - - https://archive.org/download/microsoft_xbox360_s_part1 - - https://archive.org/download/microsoft_xbox360_s_part2 - - https://archive.org/download/microsoft_xbox360_t_part1 - - https://archive.org/download/microsoft_xbox360_t_part2 - - https://archive.org/download/microsoft_xbox360_u - - https://archive.org/download/microsoft_xbox360_v - - https://archive.org/download/microsoft_xbox360_w - - https://archive.org/download/microsoft_xbox360_x_part1 - - https://archive.org/download/microsoft_xbox360_x_part2 - - https://archive.org/download/microsoft_xbox360_y - - https://archive.org/download/microsoft_xbox360_z - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} -- source: minerva - format: DIGITAL - urls: - - ./No-Intro/Microsoft - Xbox 360 (Digital)/ - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} -- source: internet_archive - format: DIGITAL - urls: - - https://archive.org/download/microsoft_xbox360_digital_part1 - - https://archive.org/download/microsoft_xbox360_digital_part2 - - https://archive.org/download/microsoft_xbox360_digital_part3 - - https://archive.org/download/microsoft_xbox360_digital_part4 - - https://archive.org/download/microsoft_xbox360_digital_part5 - - https://archive.org/download/microsoft_xbox360_digital_part6 - - https://archive.org/download/microsoft_xbox360_digital_part7 - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} -sms: -- source: minerva - format: sms - urls: - - ./No-Intro/Sega - Master System - Mark III/ - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} -- source: internet_archive - format: sms - urls: - - https://archive.org/download/nointro.ms-mkiii - filter: (.*)\.7z - type: Game - parsers: - libretro: {} - no_intro: {} -gg: -- source: minerva - format: gg - urls: - - ./No-Intro/Sega - Game Gear/ - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} -- source: internet_archive - format: gg - urls: - - https://archive.org/download/nointro.gg - filter: (.*)\.7z - type: Game - parsers: - libretro: {} - no_intro: {} -smd: -- source: minerva - format: md - urls: - - ./No-Intro/Sega - Mega Drive - Genesis/ - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} -- source: internet_archive - format: md - urls: - - https://archive.org/download/nointro.md - filter: (.*)\.7z - type: Game - parsers: - libretro: {} - no_intro: {} -scd: -- source: minerva - format: bin/cue - urls: - - ./Redump/Sega - Mega CD & Sega CD/ - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} -- source: internet_archive - format: bin/cue - urls: - - https://archive.org/download/sega_mega-cd_sega-cd - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} -32x: -- source: minerva - format: 32x - urls: - - ./No-Intro/Sega - 32X/ - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} -- source: internet_archive - format: 32x - urls: - - https://archive.org/download/nointro.32x - filter: (.*)\.7z - type: Game - parsers: - libretro: {} - no_intro: {} -sat: -- source: minerva - format: bin/cue - urls: - - ./Redump/Sega - Saturn/ - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} -- source: internet_archive - format: bin/cue - urls: - - https://archive.org/download/sega_saturn - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} -- source: internet_archive - format: chd - regions: - - us - urls: - - https://archive.org/download/chd_saturn/CHD-Saturn/USA/ - filter: (.*)\.chd - type: Game - parsers: - libretro: {} - no_intro: {} -- source: internet_archive - format: chd - regions: - - eu - urls: - - https://archive.org/download/chd_saturn/CHD-Saturn/Europe/ - filter: (.*)\.chd - type: Game - parsers: - libretro: {} - no_intro: {} -- source: internet_archive - format: chd - regions: - - jp - urls: - - https://archive.org/download/chd_saturn/CHD-Saturn/Japan/ - filter: (.*)\.chd - type: Game - parsers: - libretro: {} - no_intro: {} -- source: internet_archive - format: chd - urls: - - https://archive.org/download/chd_saturn/CHD-Saturn/Other-Regions/ - filter: (.*)\.chd - type: Game - parsers: - libretro: {} - no_intro: {} -dc: -- source: minerva - format: bin/cue - urls: - - ./Redump/Sega - Dreamcast/ - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} -- source: internet_archive - format: bin/cue - urls: - - https://archive.org/download/sega_dreamcast - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} -- source: internet_archive - format: chd - urls: - - https://archive.org/download/dc-chd-zstd-redump/dc-chd-zstd/ - filter: (.*)\.chd - type: Game - parsers: - libretro: {} - no_intro: {} -mame: -- source: internet_archive - format: rom - urls: - - https://archive.org/download/mame-chds-roms-extras-complete - filter: (.*)\.zip - type: Game - parsers: - mame: {} - libretro: {} - no_intro: {} -fbneo: -- source: minerva - format: rom - urls: - - ./FinalBurn Neo/arcade/ - filter: (.*)\.zip - type: Game - parsers: - mame: {} - libretro: {} -- source: internet_archive - format: rom - urls: - - https://archive.org/download/fbnarcade-fullnonmerged/arcade/ - filter: (.*)\.zip - type: Game - parsers: - mame: {} - libretro: {} -a26: -- source: minerva - format: a26 - urls: - - ./No-Intro/Atari - Atari 2600/ - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} -- source: internet_archive - format: a26 - urls: - - https://archive.org/download/nointro.atari-2600 - filter: (.*)\.7z - type: Game - parsers: - libretro: {} - no_intro: {} -a52: -- source: minerva - format: a52 - urls: - - ./No-Intro/Atari - Atari 5200/ - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} -- source: internet_archive - format: a52 - urls: - - https://archive.org/download/nointro.atari-5200 - filter: (.*)\.7z - type: Game - parsers: - libretro: {} - no_intro: {} -a78: -- source: minerva - format: bin - urls: - - ./No-Intro/Atari - Atari 7800 (BIN) (Aftermarket)/ - - ./No-Intro/Atari - Atari 7800 (BIN) (Private)/ - - ./No-Intro/Atari - Atari 7800 (BIN)/ - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} -- source: minerva - format: a78 - urls: - - ./No-Intro/Atari - Atari 7800 (A78) (Aftermarket)/ - - ./No-Intro/Atari - Atari 7800 (A78) (Private)/ - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} -- source: internet_archive - format: a78 - urls: - - https://archive.org/download/nointro.atari-7800 - filter: (.*)\.7z - type: Game - parsers: - libretro: {} - no_intro: {} -lynx: -- source: minerva - format: lnx - urls: - - ./No-Intro/Atari - Atari Lynx (LNX) (Aftermarket)/ - - ./No-Intro/Atari - Atari Lynx (LNX) (Private)/ - - ./No-Intro/Atari - Atari Lynx (LNX)/ - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} -- source: internet_archive - format: lnx - urls: - - https://archive.org/download/AtariLynxRomCollectionByGhostware - filter: (.*)\.lnx - type: Game - parsers: - libretro: {} - no_intro: {} -- source: minerva - format: lyx - urls: - - ./No-Intro/Atari - Atari Lynx (LYX) (Aftermarket)/ - - ./No-Intro/Atari - Atari Lynx (LYX) (Private)/ - - ./No-Intro/Atari - Atari Lynx (LYX)/ - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} -- source: minerva - format: bll - urls: - - ./No-Intro/Atari - Atari Lynx (BLL) (Aftermarket)/ - - ./No-Intro/Atari - Atari Lynx (BLL)/ - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} -jag: -- source: minerva - format: j64 - urls: - - ./No-Intro/Atari - Atari Jaguar (J64) (Aftermarket)/ - - ./No-Intro/Atari - Atari Jaguar (J64)/ - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} -- source: minerva - format: abs - urls: - - ./No-Intro/Atari - Atari Jaguar (ABS) (Aftermarket)/ - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} -- source: minerva - format: cof - urls: - - ./No-Intro/Atari - Atari Jaguar (COF) (Aftermarket)/ - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} -- source: minerva - format: rom - urls: - - ./No-Intro/Atari - Atari Jaguar (ROM) (Aftermarket)/ - - ./No-Intro/Atari - Atari Jaguar (ROM)/ - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} -- source: minerva - format: jag - urls: - - ./No-Intro/Atari - Atari Jaguar (JAG) (Aftermarket)/ - - ./No-Intro/Atari - Atari Jaguar (JAG)/ - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} -- source: internet_archive - format: chd - urls: - - https://archive.org/download/jagcd-chd-zstd/jagcd-chd-zstd/ - filter: (.*)\.chd - type: Game - parsers: - libretro: {} - no_intro: {} -jcd: -- source: minerva - format: bin/cue - urls: - - ./Redump/Atari - Jaguar CD Interactive Multimedia System/ - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} -- source: internet_archive - format: bin/cue - urls: - - https://archive.org/download/atari_jaguar-cd - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} -tg16: -- source: minerva - format: pce - urls: - - ./No-Intro/NEC - PC Engine - TurboGrafx-16/ - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} -- source: internet_archive - format: pce - urls: - - https://archive.org/download/nointro.tg-16 - filter: (.*)\.7z - type: Game - parsers: - libretro: {} - no_intro: {} -tgcd: -- source: minerva - format: bin/cue - urls: - - ./Redump/NEC - PC Engine CD & TurboGrafx CD/ - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} -- source: internet_archive - format: bin/cue - urls: - - https://archive.org/download/nec_pc-engine-cd_turbografx-cd - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} -pcfx: -- source: minerva - format: bin/cue - urls: - - ./Redump/NEC - PC-FX & PC-FXGA/ - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} -- source: internet_archive - format: bin/cue - urls: - - https://archive.org/download/nec_pc-fxpc_fxga - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} -pc98: -- source: minerva - format: bin/cue - urls: - - ./Redump/NEC - PC-98 series/ - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} -- source: internet_archive - format: bin/cue - urls: - - https://archive.org/download/nec_pc-98_series - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} -intv: -- source: minerva - format: int - urls: - - ./No-Intro/Mattel - Intellivision/ - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} -cv: -- source: minerva - format: col - urls: - - ./No-Intro/Coleco - ColecoVision/ - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} -3do: -- source: minerva - format: bin/cue - urls: - - ./Redump/Panasonic - 3DO Interactive Multiplayer/ - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} -- source: internet_archive - format: bin/cue - urls: - - https://archive.org/download/panasonic_3do_interactive_multiplayer - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} -cdi: -- source: minerva - format: bin/cue - urls: - - ./Redump/Philips - CD-i/ - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} -- source: internet_archive - format: bin/cue - urls: - - https://archive.org/download/philips_cd-i - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} -fmt: -- source: minerva - format: bin/cue - urls: - - ./Redump/Fujitsu - FM-Towns/ - filter: (.*)\.zip - type: Game - parsers: - no_intro: {} -- source: internet_archive - format: bin/cue - urls: - - https://archive.org/download/fujitsu_fm_towns_series - filter: (.*)\.zip - type: Game - parsers: - no_intro: {} -ngcd: -- source: minerva - format: bin/cue - urls: - - ./Redump/SNK - Neo Geo CD/ - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} -- source: internet_archive - format: bin/cue - urls: - - https://archive.org/download/snk_neo_geo - filter: (.*)\.zip - type: Game - parsers: - libretro: {} - no_intro: {} -pip: -- source: minerva - format: bin/cue - urls: - - ./Redump/Bandai - Pippin/ - filter: (.*)\.zip - type: Game - parsers: - no_intro: {} -- source: internet_archive - format: bin/cue - urls: - - https://archive.org/download/bandai_pippin - filter: (.*)\.zip - type: Game - parsers: - no_intro: {} diff --git a/db/requirements.txt b/db/requirements.txt deleted file mode 100644 index 76557424..00000000 --- a/db/requirements.txt +++ /dev/null @@ -1,7 +0,0 @@ -requests -cloudscraper -unidecode -playwright -pyyaml -pytest -pytest-cov diff --git a/db/scripts/download_gametdb_xmls.py b/db/scripts/download_gametdb_xmls.py deleted file mode 100644 index 8847b90e..00000000 --- a/db/scripts/download_gametdb_xmls.py +++ /dev/null @@ -1,89 +0,0 @@ -#!/usr/bin/env python -""" -This script downloads and extracts GameTDB XML files. -Uses cloudscraper to bypass Cloudflare protection. -""" -import os -import sys -import zipfile - -# Add parent directory to path for imports -sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) - -import cloudscraper - -DOWNLOADS = [ - {'url': 'https://www.gametdb.com/dstdb.zip?LANG=EN', 'xml': 'dstdb.xml', 'referer': 'https://www.gametdb.com/DS/Downloads'}, - {'url': 'https://www.gametdb.com/wiitdb.zip?LANG=EN&WIIWARE=1&GAMECUBE=1', 'xml': 'wiitdb.xml', 'referer': 'https://www.gametdb.com/Wii/Downloads'}, - {'url': 'https://www.gametdb.com/3dstdb.zip?LANG=EN', 'xml': '3dstdb.xml', 'referer': 'https://www.gametdb.com/3DS/Downloads'}, - {'url': 'https://www.gametdb.com/wiiutdb.zip?LANG=EN', 'xml': 'wiiutdb.xml', 'referer': 'https://www.gametdb.com/WiiU/Downloads'}, - {'url': 'https://www.gametdb.com/ps3tdb.zip?LANG=EN', 'xml': 'ps3tdb.xml', 'referer': 'https://www.gametdb.com/PS3/Downloads'}, -] - - -def download_gametdb_xmls(): - """Download and extract GameTDB XML files.""" - print("Downloading GameTDB XML files...") - - destination = 'data/gametdb' - os.makedirs(destination, exist_ok=True) - - # Create cloudscraper session (bypasses Cloudflare) - session = cloudscraper.create_scraper( - browser={ - 'browser': 'chrome', - 'platform': 'windows', - 'mobile': False - } - ) - - success_count = 0 - - for item in DOWNLOADS: - xml_file = item['xml'] - xml_file_path = os.path.join(destination, xml_file) - zip_file_name = item['url'].split('/')[-1].split('?')[0] - zip_file_path = os.path.join(destination, zip_file_name) - - # Skip if already exists - if os.path.exists(xml_file_path): - print(f" {xml_file}: cached") - success_count += 1 - continue - - print(f" {xml_file}: ", end='', flush=True) - - try: - # Set referer header for this request - response = session.get( - item['url'], - headers={'Referer': item['referer']}, - timeout=120 - ) - - if response.ok and len(response.content) > 1000: - # Save zip - with open(zip_file_path, 'wb') as f: - f.write(response.content) - - # Extract - with zipfile.ZipFile(zip_file_path, 'r') as zip_ref: - zip_ref.extractall(destination) - os.remove(zip_file_path) - print("OK") - success_count += 1 - else: - print(f"failed ({response.status_code})") - - except Exception as e: - print(f"failed ({e})") - if os.path.exists(zip_file_path): - os.remove(zip_file_path) - - print(f"GameTDB: {success_count}/{len(DOWNLOADS)} files") - - -if __name__ == '__main__': - os.chdir(os.path.dirname(os.path.realpath(__file__))) - os.chdir('../') - download_gametdb_xmls() diff --git a/db/scripts/download_libretro_dats.py b/db/scripts/download_libretro_dats.py deleted file mode 100644 index db8bc512..00000000 --- a/db/scripts/download_libretro_dats.py +++ /dev/null @@ -1,81 +0,0 @@ -#!/usr/bin/env python -""" -This script downloads Libretro DAT files from the official Libretro GitHub repository. -It uses a sparse Git checkout to efficiently clone only the required directories and -copies the files to a specified destination directory. -""" -import os -import sys -import subprocess -import shutil -import tempfile - - -def download_libretro_dats(): - """Download Libretro DAT files from the official GitHub repository.""" - print("Downloading Libretro DAT files...") - - # Check if Git is installed and available in PATH - if not shutil.which('git'): - print("Git is not installed or not found in PATH. Please install Git to proceed.") - sys.exit(1) - - # Base destination directory for the downloaded files - base_destination = 'data/libretro' - - # URL of the Libretro database repository - repo_url = 'https://github.com/libretro/libretro-database.git' - - # Directories to clone from the repository - directories_to_clone = ['dat', 'metadat'] - - # Ensure the destination directories exist - for dir_name in directories_to_clone: - destination = os.path.join(base_destination, dir_name) - os.makedirs(destination, exist_ok=True) - - # Use a temporary directory for cloning the repository - with tempfile.TemporaryDirectory() as temp_dir: - # Clone the repository with sparse checkout enabled - subprocess.run( - ['git', 'clone', '--depth', '1', '--filter=blob:none', - '--sparse', repo_url, temp_dir], - check=True, - stdout=subprocess.DEVNULL, - stderr=subprocess.DEVNULL - ) - - # Sparse checkout for each required directory - for dir_name in directories_to_clone: - subprocess.run( - ['git', '-C', temp_dir, 'sparse-checkout', 'set', dir_name], - check=True, - stdout=subprocess.DEVNULL, - stderr=subprocess.DEVNULL - ) - - # Source and destination paths for copying files - source_dir = os.path.join(temp_dir, dir_name) - destination = os.path.join(base_destination, dir_name) - - # Copy files from the sparse-checked-out directory to the destination - if os.path.exists(source_dir): - for item in os.listdir(source_dir): - s = os.path.join(source_dir, item) - d = os.path.join(destination, item) - if os.path.isdir(s): - # Copy directories recursively - shutil.copytree(s, d, dirs_exist_ok=True) - else: - # Copy individual files - shutil.copy2(s, d) - - print(f"Successfully downloaded Libretro DAT files to {base_destination}") - - -if __name__ == '__main__': - # Change the working directory to main db repository location - os.chdir(os.path.dirname(os.path.realpath(__file__))) - os.chdir('../') - - download_libretro_dats() diff --git a/db/scripts/download_mame_hashes.py b/db/scripts/download_mame_hashes.py deleted file mode 100644 index 5943ca15..00000000 --- a/db/scripts/download_mame_hashes.py +++ /dev/null @@ -1,74 +0,0 @@ -#!/usr/bin/env python -""" -This script downloads the MAME hash files from the official MAME GitHub repository -using a sparse Git checkout. It ensures that only the necessary `hash` directory -is downloaded, minimizing the data transfer. The downloaded files are saved in -the `data/mame/hash` directory. -""" -import os -import sys -import subprocess -import shutil -import tempfile - - -def download_mame_hashes(): - """Download the MAME hash files from the official MAME GitHub repository.""" - print("Downloading MAME hash files...") - - # Check if Git is installed and available in PATH - if not shutil.which('git'): - print("Git is not installed or not found in PATH. Please install Git to proceed.") - sys.exit(1) - - # Destination directory for the hash files - destination = 'data/mame/hash' - os.makedirs(destination, exist_ok=True) - - # URL of the MAME GitHub repository - repo_url = 'https://github.com/mamedev/mame.git' - - # Directory to be sparsely checked out from the repository - dir_path = 'hash' - - # Use a temporary directory for cloning the repository - with tempfile.TemporaryDirectory() as temp_dir: - # Clone the repository with sparse checkout enabled - subprocess.run( - ['git', 'clone', '--depth', '1', '--filter=blob:none', - '--sparse', repo_url, temp_dir], - check=True, - stdout=subprocess.DEVNULL, - stderr=subprocess.DEVNULL - ) - # Set the sparse-checkout to only include the `hash` directory - subprocess.run( - ['git', '-C', temp_dir, 'sparse-checkout', 'set', dir_path], - check=True, - stdout=subprocess.DEVNULL, - stderr=subprocess.DEVNULL - ) - - # Path to the `hash` directory in the cloned repository - hash_dir = os.path.join(temp_dir, dir_path) - if os.path.exists(hash_dir): - # Copy all files and subdirectories from the `hash` directory to the destination - for item in os.listdir(hash_dir): - s = os.path.join(hash_dir, item) - d = os.path.join(destination, item) - if os.path.isdir(s): - # Copy directories recursively - shutil.copytree(s, d, dirs_exist_ok=True) - else: - # Copy individual files - shutil.copy2(s, d) - - print(f"Successfully downloaded MAME hash files to {destination}") - - -if __name__ == '__main__': - # Change the working directory to main db repository location - os.chdir(os.path.dirname(os.path.realpath(__file__))) - os.chdir('../') - - download_mame_hashes() diff --git a/db/scripts/download_minerva_artefacts.py b/db/scripts/download_minerva_artefacts.py deleted file mode 100644 index 00908b95..00000000 --- a/db/scripts/download_minerva_artefacts.py +++ /dev/null @@ -1,172 +0,0 @@ -""" -Mirror MiNERVA's hashes.db into data/minerva/. The old index.txt.gz was -removed upstream; the scraper derives the path index from hashes.db. - -ETag-cached: a HEAD request decides whether the local copy is fresh; if -so we skip the download. Supports resume via Range requests so a -partial 1.7 GB hashes.db can be picked up without restarting. -""" -from __future__ import annotations - -import http.client -import os -import sys -import time -import urllib.error -import urllib.request -from pathlib import Path -from typing import Iterable - - -_DEFAULT_DEST = Path('data/minerva') -_USER_AGENT = 'romgi/1.x (workflow.py)' -_TIMEOUT = 60 - -# (url, filename) pairs. The build pipeline references these by name. -_ARTEFACTS: list[tuple[str, str]] = [ - ('https://minerva-archive.org/assets/hashes.db', 'hashes.db'), -] - - -class MinervaDownloadError(RuntimeError): - pass - - -def download_minerva_artefacts( - *, - dest: Path | None = None, - artefacts: Iterable[tuple[str, str]] | None = None, -) -> None: - """Download both artefacts. Network failures leave existing copies alone.""" - target = dest or _DEFAULT_DEST - target.mkdir(parents=True, exist_ok=True) - - print('Downloading MiNERVA artefacts...') - for url, name in (artefacts or _ARTEFACTS): - _fetch_if_changed(url, target / name) - - -# -- internals --------------------------------------------------------------- - -def _fetch_if_changed(url: str, dest: Path) -> None: - """Download with atomic completion. - - The canonical file at `dest` is only created via `os.replace` after - a complete download. In-progress bytes live in `dest.with_suffix(.part)` - so a truncated download never gets handed to the scraper. - """ - etag_path = dest.with_name(dest.name + '.etag') - part_path = dest.with_name(dest.name + '.part') - cached_etag = etag_path.read_text(encoding='utf-8').strip() if etag_path.is_file() else None - - try: - head = _request(url, method='HEAD') - remote_etag = (head.headers.get('ETag') or '').strip('"').strip() - remote_size = int(head.headers.get('Content-Length') or 0) - except (urllib.error.URLError, urllib.error.HTTPError, TimeoutError) as e: - print(f' [minerva] HEAD {url} failed ({e}); leaving any existing copy in place.') - return - - if ( - cached_etag - and remote_etag - and cached_etag == remote_etag - and dest.is_file() - and (remote_size == 0 or dest.stat().st_size == remote_size) - ): - size_str = _humanize(remote_size) if remote_size else f'{dest.stat().st_size:,} B' - print(f' [minerva] {dest.name} unchanged ({size_str}, etag matches), skipping.') - return - - # Resume from the part file when available. If the canonical file - # exists but the etag changed, restart from scratch. - start = part_path.stat().st_size if part_path.is_file() else 0 - if remote_size and start >= remote_size: - # Server's file shrank or the partial got bigger somehow; restart. - start = 0 - try: - part_path.unlink() - except FileNotFoundError: - pass - - if start: - print( - f' [minerva] fetching {dest.name}' - f' ({_humanize(remote_size)} total, resuming from {_humanize(start)})' - ) - else: - print(f' [minerva] fetching {dest.name} ({_humanize(remote_size)})') - - try: - _stream_download(url, part_path, start=start, total=remote_size) - except (urllib.error.URLError, urllib.error.HTTPError, TimeoutError, OSError) as e: - print(f' [minerva] download interrupted: {e}') - print(f' [minerva] partial copy retained at {part_path.name}; re-run to resume.') - return - - final_size = part_path.stat().st_size - if remote_size and final_size != remote_size: - print( - f' [minerva] download finished short ' - f'({_humanize(final_size)} of {_humanize(remote_size)}); ' - f'leaving {part_path.name} in place to resume.' - ) - return - - os.replace(part_path, dest) - if remote_etag: - etag_path.write_text(remote_etag, encoding='utf-8') - print(f' [minerva] {dest.name} done.') - - -def _stream_download(url: str, part_path: Path, *, start: int, total: int) -> None: - headers = {'User-Agent': _USER_AGENT} - mode = 'wb' - if start > 0: - headers['Range'] = f'bytes={start}-' - mode = 'ab' - - req = urllib.request.Request(url, headers=headers, method='GET') - with urllib.request.urlopen(req, timeout=_TIMEOUT) as resp, part_path.open(mode) as f: - downloaded = start - chunk_size = 1 << 20 # 1 MiB - last_print = 0.0 - while True: - chunk = resp.read(chunk_size) - if not chunk: - break - f.write(chunk) - downloaded += len(chunk) - now = time.monotonic() - if total and now - last_print >= 1.0: - pct = downloaded / total * 100 - sys.stdout.write( - f'\r {_humanize(downloaded)} / {_humanize(total)} ({pct:5.1f}%)' - ) - sys.stdout.flush() - last_print = now - if total: - sys.stdout.write('\r' + ' ' * 60 + '\r') - sys.stdout.flush() - - -def _request(url: str, *, method: str = 'GET') -> http.client.HTTPResponse: - req = urllib.request.Request( - url, headers={'User-Agent': _USER_AGENT}, method=method, - ) - return urllib.request.urlopen(req, timeout=_TIMEOUT) - - -def _humanize(n: int) -> str: - if n <= 0: - return '0 B' - size = float(n) - for unit in ('B', 'KB', 'MB', 'GB', 'TB'): - if size < 1024: - return f'{size:.1f} {unit}' if unit != 'B' else f'{size:.0f} {unit}' - size /= 1024 - return f'{size:.1f} PB' - - -if __name__ == '__main__': - download_minerva_artefacts() diff --git a/db/scripts/download_retroachievements.py b/db/scripts/download_retroachievements.py deleted file mode 100644 index ae1b749d..00000000 --- a/db/scripts/download_retroachievements.py +++ /dev/null @@ -1,167 +0,0 @@ -#!/usr/bin/env python -""" -Download RetroAchievements (RA) game lists into data/retroachievements/. - -For each RA-supported console we fetch the list of games that have an -achievement set (API_GetGameList with f=1) and store it as -data/retroachievements/.json. The retroachievements parser then -matches catalog entries against these lists by title. - -RA game data is very static and the API is rate limited, so we fetch -sequentially with polite throttling and exponential backoff (honouring the -Retry-After header). Credentials come from the RA_API_USER and RA_API_KEY -environment variables; if either is missing this no-ops and the build simply -leaves the RA columns empty. -""" -import json -import os -import sys -import time - -# Add parent directory to path for imports -sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) - -import requests - -from parsers.retroachievements import DATA_DIR, RA_CONSOLES - -API_URL = 'https://retroachievements.org/API/API_GetGameList.php' -USER_AGENT = 'romgi-db-builder/1.0 (https://github.com/christianprado/romgi)' - -# Politeness knobs. Kept as module constants so tests can drive them small. -MIN_INTERVAL = 1.0 # minimum seconds between requests -MAX_RETRIES = 4 # attempts per console before giving up -BASE_BACKOFF = 2.0 # first backoff delay; doubles each retry - -# Monotonic timestamp of the last request, used to space out calls. -_last_request_time = 0.0 - - -def _throttle(min_interval: float) -> None: - """Sleep just long enough that requests are spaced >= min_interval apart.""" - global _last_request_time - elapsed = time.monotonic() - _last_request_time - wait = min_interval - elapsed - if wait > 0: - time.sleep(wait) - _last_request_time = time.monotonic() - - -def _retry_delay(response: requests.Response, fallback: float) -> float: - """Prefer the server's Retry-After header, else use the backoff fallback.""" - retry_after = response.headers.get('Retry-After') - if retry_after: - try: - return float(retry_after) - except ValueError: - pass - return fallback - - -def _get_with_backoff( - session: requests.Session, - url: str, - params: dict, - *, - min_interval: float = MIN_INTERVAL, - max_retries: int = MAX_RETRIES, -) -> requests.Response | None: - """GET with throttling + retry/backoff on 429 and 5xx. - - Returns the 200 Response, or None if it could not be fetched (non-retryable - status, or retries exhausted). - """ - backoff = BASE_BACKOFF - for attempt in range(1, max_retries + 1): - _throttle(min_interval) - try: - response = session.get( - url, params=params, timeout=60, - headers={'User-Agent': USER_AGENT}, - ) - except requests.RequestException as e: - print(f" request error ({e}); attempt {attempt}/{max_retries}") - time.sleep(backoff) - backoff *= 2 - continue - - if response.status_code == 200: - return response - - if response.status_code == 429 or response.status_code >= 500: - delay = _retry_delay(response, backoff) - print(f" HTTP {response.status_code}, backing off {delay}s " - f"(attempt {attempt}/{max_retries})") - time.sleep(delay) - backoff *= 2 - continue - - # Other 4xx are not retryable. - print(f" HTTP {response.status_code}; giving up") - return None - - return None - - -def _write_atomic(dest: str, games: list) -> None: - """Write JSON to a temp file then atomically replace dest (never corrupts cache).""" - tmp = dest + '.tmp' - with open(tmp, 'w', encoding='utf-8') as f: - json.dump(games, f) - os.replace(tmp, dest) - - -def download_retroachievements(*, use_cached: bool = False) -> None: - """Download RA game lists for every supported console into data/retroachievements/.""" - user = os.environ.get('RA_API_USER') - key = os.environ.get('RA_API_KEY') - - if not user or not key: - print("RetroAchievements: RA_API_USER/RA_API_KEY not set, skipping " - "(RA columns will be empty).") - return - - os.makedirs(DATA_DIR, exist_ok=True) - session = requests.Session() - - total = len(RA_CONSOLES) - ok = 0 - print(f"Downloading RetroAchievements game lists ({total} consoles)...") - - for platform_id, console_id in RA_CONSOLES.items(): - dest = os.path.join(DATA_DIR, f'{console_id}.json') - - if use_cached and os.path.exists(dest): - print(f" {platform_id} (console {console_id}): cached") - ok += 1 - continue - - response = _get_with_backoff( - session, API_URL, - {'i': console_id, 'f': 1, 'z': user, 'y': key}, - ) - if response is None: - print(f" {platform_id} (console {console_id}): failed") - continue - - try: - games = response.json() - except ValueError: - print(f" {platform_id} (console {console_id}): invalid JSON") - continue - - if not isinstance(games, list) or not games: - print(f" {platform_id} (console {console_id}): empty/invalid response") - continue - - _write_atomic(dest, games) - print(f" {platform_id} (console {console_id}): {len(games)} games") - ok += 1 - - print(f"RetroAchievements: {ok}/{total} consoles") - - -if __name__ == '__main__': - os.chdir(os.path.dirname(os.path.realpath(__file__))) - os.chdir('../') - download_retroachievements(use_cached='--use-cached' in sys.argv[1:]) diff --git a/db/sources/__init__.py b/db/sources/__init__.py deleted file mode 100644 index e547ed45..00000000 --- a/db/sources/__init__.py +++ /dev/null @@ -1 +0,0 @@ -"""Source plugin packages. Auto-discovered by db/core/registry.py.""" diff --git a/db/sources/internet_archive/__init__.py b/db/sources/internet_archive/__init__.py deleted file mode 100644 index 66b92483..00000000 --- a/db/sources/internet_archive/__init__.py +++ /dev/null @@ -1 +0,0 @@ -from .scraper import SOURCE # noqa: F401 diff --git a/db/sources/internet_archive/scraper.py b/db/sources/internet_archive/scraper.py deleted file mode 100644 index 642d2239..00000000 --- a/db/sources/internet_archive/scraper.py +++ /dev/null @@ -1,226 +0,0 @@ -""" -Internet Archive source plugin. - -Scrapes publicly-linked and login-required ("restricted") files from -archive.org item pages. CI uses credentials at -secrets/internet_archive_creds.json to log in for restricted listings; -the app gates restricted downloads on the user's own session. -""" -import re -import urllib.parse -import html -import json - -import cloudscraper - -from utils import cache_manager -from utils.scrape_utils import fetch_url -from utils.parse_utils import size_bytes_to_str, size_str_to_bytes, join_urls - -from typing import Any -from core.contract import BuildContext, PlatformConfig, SourceManifest - - -HOST_NAME = 'Internet Archive' -LOGIN_URL = 'https://archive.org/account/login' - -session: Any = None - - -def get_login_session(creds_path: str = 'secrets/internet_archive_creds.json') -> Any: - """Create and return a session logged into the Internet Archive.""" - try: - with open(creds_path, 'r') as f: - creds = json.load(f) - - session = cloudscraper.create_scraper() - session.get(LOGIN_URL) - - r = session.post(LOGIN_URL, data={ - 'username': creds['username'], - 'password': creds['password'] - }) - - if not r.ok: - raise Exception("Wrong or invalid credentials") - - return session - except (FileNotFoundError, json.JSONDecodeError) as e: - print(f"Warning: Internet Archive credentials not found: {e}") - return None - except Exception as e: - print(f"Warning: Failed to log into Internet Archive: {e}") - return None - - -def extract_entries(response: str, source: dict[str, Any], platform: str, base_url: str, debug: bool = False) -> list[dict[str, Any]]: - """Extract entries from the HTML response using regex.""" - entries = [] - matches = [] - # Common ROM file extensions - file_ext = r'(zip|chd|iso|7z|rar|nsp|xci|wbfs|rvz|cso|pbp|pkg|bin|nds|3ds|cia|gba|gbc|gb|n64|z64|v64|nes|sfc|smc|gen|md|sms|gg|pce|vpk|app|cue|wad|dol|gcm|wux|wua|lnx|lyx|a26|a78|col|int|jag|ngp|ngc|psx|ws|wsc|vb|vec)' - - # Strategy 1: Find linked files (public downloads) - # Pattern: filename.ext - link_pattern = rf']*>' - for match in re.finditer(link_pattern, response, re.IGNORECASE): - href = match.group(1) - start_pos = match.end() - chunk = response[start_pos:start_pos + 500] - size_match = re.search(r'(\d+\.?\d*)\s*([KMGT])i?B?', chunk, re.IGNORECASE) - size_str = f"{size_match.group(1)}{size_match.group(2)}" if size_match else '' - filename = html.unescape(urllib.parse.unquote(href)) - matches.append((href, filename, size_str)) - - if debug: - print(f" Found {len(matches)} linked files") - - # Strategy 2: Find restricted files (no links, just text in ) - # These rows have class "__restricted-file" and plain text filenames - # Pattern: filename.extdatesize - restricted_pattern = rf']*restricted-file[^>]*>\s*([^<]+\.{file_ext})\s*[^<]*\s*([^<]*)' - for match in re.finditer(restricted_pattern, response, re.IGNORECASE | re.DOTALL): - filename = html.unescape(match.group(1).strip()) - size_str = match.group(3).strip() - size_match = re.search(r'(\d+\.?\d*)\s*([KMGT])', size_str, re.IGNORECASE) - size_str = f"{size_match.group(1)}{size_match.group(2)}" if size_match else '' - href = urllib.parse.quote(filename) - matches.append((href, filename, size_str)) - - if debug: - print(f" Found {len(matches)} total files (including restricted)") - - for link, filename, size_str in matches: - filename = filename.strip() - # Strip HTML tags from size (e.g., 2.5G -> 2.5G) - size_str = re.sub(r'<[^>]+>', '', size_str).strip() - - # Skip non-file entries (parent directory link, etc.) - if not filename or 'parent directory' in filename.lower() or '.' not in filename: - if debug: - print(f" Skipped (no file): link={link[:30]}, filename={filename[:30] if filename else 'empty'}") - continue - - # Apply the filter from the source configuration - match = re.match(source['filter'], filename) - if not match: - if debug: - print(f" Skipped (filter): {filename[:50]} didn't match {source['filter'][:50]}") - continue - - title = match.group(1) # Extract the filtered title - - # Create an entry and add it to the list - entries.append(create_entry( - link, filename, title, size_str, source, platform, base_url)) - - return entries - - -def create_entry(link: str, filename: str, title: str, size_str: str, source: dict[str, Any], platform: str, base_url: str) -> dict[str, Any]: - """Create a dictionary representing a single entry.""" - name = html.unescape(title) - size = size_str_to_bytes(size_str) - size_str = size_bytes_to_str(size) - url = join_urls(base_url, link) - - return { - 'title': name, - 'platform': platform, - 'regions': source['regions'], - 'links': [ - { - 'name': name, - 'type': source['type'], - 'format': source['format'], - 'url': url, - 'filename': filename, - 'host': HOST_NAME, - 'size': size, - 'size_str': size_str, - 'source_url': base_url - } - ] - } - - -def fetch_response(url: str, session: Any, use_cached: bool) -> str | None: - """Fetch the response from a URL, optionally using a cached version.""" - url_stripped = url.rstrip('/') - short_url = url_stripped.split('/')[-1][:50] if '/' in url_stripped else url_stripped[:50] - - if use_cached: - response = cache_manager.get_cached_response(url) - if response: - print(f" {short_url}... cached") - return response - - # Fetch the URL using the provided session - return fetch_url(url, session) - - -def scrape(source: dict[str, Any], platform: str, use_cached: bool = False) -> list[dict[str, Any]]: - """Scrapes entries from the Internet Archive based on the source configuration.""" - global session - - entries = [] - - # First attempt: scrape without login session - for url in source['urls']: - response = fetch_response(url, session, use_cached) - if not response: - print(f"Warning: Failed to get response from {url}, skipping...") - continue - - parsed_entries = extract_entries(response, source, platform, url) - if parsed_entries: - entries.extend(parsed_entries) - else: - # Initialize the session if not already done - if not session: - session = get_login_session() - if not session: - print("Warning: Unable to create Internet Archive session, skipping login-required content...") - # Try debug mode to see what HTML we got - extract_entries(response, source, platform, url, debug=True) - continue - - # Retry with login session (bypass cache to get authenticated response) - response = fetch_response(url, session, use_cached=False) - if not response: - print(f"Warning: Failed to get response from {url} with login, skipping...") - continue - - parsed_entries = extract_entries(response, source, platform, url) - if parsed_entries: - for entry in parsed_entries: - for link in entry['links']: - # Structured flag the app's resolver/adapter read. - link['requires_auth'] = 1 - # Human-readable suffix for UI label fallback. - link['type'] += " (Requires Internet Archive Log in)" - entries.extend(parsed_entries) - else: - # Show debug info when parsing fails - print(f"Warning: No entries parsed from {url}, skipping...") - extract_entries(response, source, platform, url, debug=True) - - return entries - - -class InternetArchiveSource: - """Adapter from the plugin contract to the legacy scrape().""" - - def __init__(self, manifest: SourceManifest): - self.manifest = manifest - - def scrape( - self, - platform: str, - config: PlatformConfig, - ctx: BuildContext, - ) -> list[dict[str, Any]]: - return scrape(config.to_legacy_dict(), platform, ctx.use_cached) - - -SOURCE = InternetArchiveSource diff --git a/db/sources/internet_archive/source.yml b/db/sources/internet_archive/source.yml deleted file mode 100644 index 9c5c2ac0..00000000 --- a/db/sources/internet_archive/source.yml +++ /dev/null @@ -1,49 +0,0 @@ -id: internet_archive -name: Internet Archive -homepage: https://archive.org -kind: catalog -priority: 50 -auth: - required: false - optional: true -capabilities: - - http_range_resume - - filename_listing - - restricted_content -platforms: - - 32x - - 3do - - 3ds - - a26 - - a52 - - a78 - - cdi - - dc - - dsi - - fmt - - gc - - gg - - jag - - jcd - - lynx - - mame - - ngcd - - pc98 - - pcfx - - pip - - ps1 - - ps2 - - ps3 - - psp - - sat - - scd - - smd - - sms - - tg16 - - tgcd - - wii - - x360 - - xbox -notes: | - Mixed public + login-required content. Login-required links carry - requires_auth=1; the app gates downloads via IAAuthManager. diff --git a/db/sources/mariocube/__init__.py b/db/sources/mariocube/__init__.py deleted file mode 100644 index 66b92483..00000000 --- a/db/sources/mariocube/__init__.py +++ /dev/null @@ -1 +0,0 @@ -from .scraper import SOURCE # noqa: F401 diff --git a/db/sources/mariocube/scraper.py b/db/sources/mariocube/scraper.py deleted file mode 100644 index 260a5ca0..00000000 --- a/db/sources/mariocube/scraper.py +++ /dev/null @@ -1,139 +0,0 @@ -""" -MarioCube source plugin. - -archive.mariocube.com serves a plain-text directory listing to curl-style -User-Agents. Strip ANSI colour, split into (size, filename), emit entries. -""" -import html -import re -import urllib.parse - -from utils import cache_manager -from utils.scrape_utils import fetch_url, create_scraper_session -from utils.parse_utils import size_str_to_bytes, join_urls - -from typing import Any, Generator -from core.contract import BuildContext, PlatformConfig, SourceManifest - - -HOST_NAME = 'MarioCube' - -# Curl-like headers to get plain-text directory listing instead of HTML -CURL_HEADERS = { - 'User-Agent': 'curl/8.0', - 'Accept': '*/*' -} - - -def extract_entries(response: str, source: dict[str, Any], platform: str, base_url: str) -> list[dict[str, Any]]: - """Extract entries from the ANSI-colored directory listing response.""" - entries = [] - - for filename, size_str in parse_listing_lines(response): - match = re.match(source['filter'], filename) - if not match: - continue - - title = match.group(1) - encoded_link = urllib.parse.quote(filename) - entries.append(create_entry( - encoded_link, filename, title, size_str, source, platform, base_url)) - - return entries - - -def create_entry(link: str, filename: str, title: str, size_str: str, source: dict[str, Any], platform: str, base_url: str) -> dict[str, Any]: - """Create a dictionary representing a single entry.""" - name = html.unescape(title) - size = size_str_to_bytes(size_str) - url = join_urls(base_url, link) - - return { - 'title': name, - 'platform': platform, - 'regions': source['regions'], - 'links': [ - { - 'name': name, - 'type': source['type'], - 'format': source['format'], - 'url': url, - 'filename': filename, - 'host': HOST_NAME, - 'size': size, - 'size_str': size_str, - 'source_url': base_url - } - ] - } - - -def parse_listing_lines(response: str) -> Generator[tuple[str, str], None, None]: - """Yield filename and size pairs from the raw listing response.""" - for raw_line in response.splitlines(): - line = re.compile(r'\x1B\[[0-?]*[ -/]*[@-~]').sub('', raw_line).strip() - if not line or line.startswith('#'): - continue - - parts = line.split(maxsplit=2) - if len(parts) < 3: - continue - - _, size_str, filename = parts - yield filename, size_str - - -def fetch_response(url: str, use_cached: bool, session: Any = None) -> str | None: - """Fetch the response from a URL, optionally using a cached version.""" - url_stripped = url.rstrip('/') - short_url = url_stripped.split('/')[-1][:50] if '/' in url_stripped else url_stripped[:50] - - if use_cached: - response = cache_manager.get_cached_response(url) - if response: - print(f" {short_url}... cached") - return response - - # Fetch the URL directly if no cached response is available - return fetch_url(url, session=session) - - -def scrape(source: dict[str, Any], platform: str, use_cached: bool = False) -> list[dict[str, Any]]: - """Scrape entries from MarioCube based on the source configuration.""" - entries = [] - session = create_scraper_session(CURL_HEADERS) - - for url in source['urls']: - # Fetch the response for each URL - response = fetch_response(url, use_cached, session=session) - if not response: - print(f"Warning: Failed to get response from {url}, skipping...") - continue - - # Extract entries from the response - parsed_entries = extract_entries(response, source, platform, url) - if not parsed_entries: - print(f"Warning: No entries parsed from {url}, skipping...") - continue - - entries.extend(parsed_entries) - - return entries - - -class MarioCubeSource: - """Adapter from the plugin contract to the legacy scrape().""" - - def __init__(self, manifest: SourceManifest): - self.manifest = manifest - - def scrape( - self, - platform: str, - config: PlatformConfig, - ctx: BuildContext, - ) -> list[dict[str, Any]]: - return scrape(config.to_legacy_dict(), platform, ctx.use_cached) - - -SOURCE = MarioCubeSource diff --git a/db/sources/mariocube/source.yml b/db/sources/mariocube/source.yml deleted file mode 100644 index c21726c4..00000000 --- a/db/sources/mariocube/source.yml +++ /dev/null @@ -1,16 +0,0 @@ -id: mariocube -name: MarioCube -homepage: https://archive.mariocube.com -kind: catalog -priority: 70 -auth: - required: false -capabilities: - - http_range_resume - - filename_listing -platforms: - - dsi - - wii -notes: | - Wii / DSi mirror. Serves a plain-text ANSI-colored listing to - User-Agent: curl/8.0; the scraper strips colour codes before parsing. diff --git a/db/sources/minerva/__init__.py b/db/sources/minerva/__init__.py deleted file mode 100644 index 66b92483..00000000 --- a/db/sources/minerva/__init__.py +++ /dev/null @@ -1 +0,0 @@ -from .scraper import SOURCE # noqa: F401 diff --git a/db/sources/minerva/scraper.py b/db/sources/minerva/scraper.py deleted file mode 100644 index 35f839de..00000000 --- a/db/sources/minerva/scraper.py +++ /dev/null @@ -1,292 +0,0 @@ -""" -MiNERVA source plugin. - -Reads two upstream artefacts the build pipeline mirrors locally: - - data/minerva/hashes.db SQLite per-file metadata (magnet, so_id, - torrents, sha1, ...) - data/minerva/index.txt optional flat list of every path; derived - from hashes.db when absent (upstream - removed the file in Aug 2026) - -For each platform's path prefix(es), join the index with hashes.db and -emit one entry per ROM file. Each link carries the torrent infohash and -file index so the TorrentAdapter can do selective-file downloads. -""" -from __future__ import annotations - -import gzip -import os -import re -import sqlite3 -import urllib.parse -from pathlib import Path - -from typing import Any -from core.contract import BuildContext, PlatformConfig, SourceManifest - - -HOST_NAME = 'MiNERVA Archive' -ENV_HASHES_DB = 'MINERVA_HASHES_DB' -ENV_INDEX_TXT = 'MINERVA_INDEX_TXT' -DEFAULT_DATA_DIR = 'data/minerva' - -_INFOHASH_RE = re.compile(r'urn:btih:([0-9a-fA-F]{40}|[A-Z2-7]{32})', re.IGNORECASE) -_TRACKER_RE = re.compile(r'[?&]tr=([^&]+)', re.IGNORECASE) - - -def _resolve_artefact(env_key: str, default_relpath: str) -> Path | None: - """Find a MiNERVA artefact via env var or default-cache location.""" - env = os.environ.get(env_key) - if env: - p = Path(env) - return p if p.is_file() else None - p = Path(default_relpath) - return p if p.is_file() else None - - -def _load_index(index_path: Path) -> list[str]: - """Load index.txt(.gz) as a list of paths.""" - if str(index_path).endswith('.gz'): - with gzip.open(index_path, 'rb') as f: - data = f.read() - else: - data = index_path.read_bytes() - return [ - line.decode('utf-8', errors='replace') - for line in data.splitlines() - if line - ] - - -def _extract_infohash(magnet: str) -> str | None: - """Pull the BTIH infohash out of a magnet URI. Lowercase hex.""" - if not magnet: - return None - m = _INFOHASH_RE.search(magnet) - if not m: - return None - raw = m.group(1) - if len(raw) == 40: # already hex - return raw.lower() - # Base32 encoding (legacy v1 magnets); decode to hex. - import base64 - try: - return base64.b32decode(raw.upper(), casefold=True).hex() - except Exception: - return None - - -def _extract_trackers(magnet: str) -> list[str]: - if not magnet: - return [] - return [urllib.parse.unquote(m.group(1)) for m in _TRACKER_RE.finditer(magnet)] - - -def _strip_extension(filename: str) -> str: - """ROM-aware: take everything before the *last* dot, keep dotted titles.""" - return filename.rsplit('.', 1)[0] if '.' in filename else filename - - -def _path_to_filename(full_path: str) -> str: - return full_path.rsplit('/', 1)[-1] - - -def _file_extension(filename: str) -> str: - return filename.rsplit('.', 1)[-1].lower() if '.' in filename else '' - - -def _query_metadata( - db: sqlite3.Connection, - paths: list[str], -) -> dict[str, sqlite3.Row]: - """Pull magnet/torrent/size info for the given full_paths. - - SQLite IN-clauses are limited (~999 params). Chunk to stay under the limit. - """ - if not paths: - return {} - db.row_factory = sqlite3.Row - out: dict[str, sqlite3.Row] = {} - chunk = 800 - for i in range(0, len(paths), chunk): - batch = paths[i:i + chunk] - placeholders = ','.join('?' * len(batch)) - rows = db.execute( - f'SELECT * FROM files WHERE full_path IN ({placeholders})', - batch, - ).fetchall() - for r in rows: - out[r['full_path']] = r - return out - - -def _select_paths(index: list[str], prefixes: list[str]) -> list[str]: - """Filter the path index by prefix match against any of the provided paths.""" - norm = [p if p.endswith('/') else p + '/' for p in prefixes] - return [line for line in index if any(line.startswith(p) for p in norm)] - - -def scrape_with_artefacts( - config: PlatformConfig, - platform: str, - *, - index: list[str], - db: sqlite3.Connection, -) -> list[dict[str, Any]]: - """Pure function over pre-loaded artefacts. Tests use this directly.""" - if not config.urls: - return [] - - paths = _select_paths(index, config.urls) - if not paths: - return [] - - if config.filter: - compiled = re.compile(config.filter) - paths = [p for p in paths if compiled.match(_path_to_filename(p))] - if not paths: - return [] - - metadata = _query_metadata(db, paths) - entries: list[dict[str, Any]] = [] - - for full_path in paths: - meta = metadata.get(full_path) - if meta is None: - continue - - magnet = meta['magnet'] if 'magnet' in meta.keys() else None - infohash = _extract_infohash(magnet) if magnet else None - if not infohash: - continue - - filename = _path_to_filename(full_path) - title = _strip_extension(filename) - ext = _file_extension(filename) - size = int(meta['size']) if meta['size'] is not None else 0 - torrent_filename = meta['torrents'] if 'torrents' in meta.keys() else None - so_id = meta['so_id'] if 'so_id' in meta.keys() else None - try: - file_index = int(so_id) if so_id is not None else None - except (TypeError, ValueError): - file_index = None - - full_magnet = magnet - torrent_url = ( - f'https://minerva-archive.org/assets/{urllib.parse.quote(torrent_filename)}' - if torrent_filename else None - ) - - entries.append({ - 'title': title, - 'platform': platform, - 'regions': list(config.regions), - 'links': [ - { - 'name': title, - 'type': config.type or 'Game', - 'format': config.format or ext, - 'url': torrent_url or '', - 'filename': filename, - 'host': HOST_NAME, - 'size': size, - 'size_str': '', # parsers will fill this from `size` - 'source_url': torrent_url or '', - 'torrent_infohash': infohash, - 'torrent_file_index': file_index, - 'torrent_file_path': full_path.lstrip('./'), - '_torrent_meta': { - 'infohash': infohash, - 'source_id': 'minerva', - 'name': torrent_filename, - 'magnet': full_magnet, - 'trackers': _extract_trackers(full_magnet or ''), - }, - } - ], - }) - - return entries - - -class MinervaSource: - """Plugin entry point. Skips gracefully if artefacts aren't mirrored - or the local hashes.db is corrupt (e.g. truncated mid-download). - """ - - def __init__(self, manifest: SourceManifest): - self.manifest = manifest - self._index: list[str] | None = None - self._db: sqlite3.Connection | None = None - self._artefacts_unusable = False - - def _ensure_artefacts(self) -> bool: - if self._artefacts_unusable: - return False - if self._index is not None and self._db is not None: - return True - index_path = _resolve_artefact(ENV_INDEX_TXT, f'{DEFAULT_DATA_DIR}/index.txt.gz') - if index_path is None: - index_path = _resolve_artefact(ENV_INDEX_TXT, f'{DEFAULT_DATA_DIR}/index.txt') - db_path = _resolve_artefact(ENV_HASHES_DB, f'{DEFAULT_DATA_DIR}/hashes.db') - - if db_path is None: - print( - f" [minerva] hashes.db missing; skipping. Set " - f"{ENV_HASHES_DB} or place it under {DEFAULT_DATA_DIR}/." - ) - self._artefacts_unusable = True - return False - - try: - db = sqlite3.connect(f'file:{db_path}?mode=ro', uri=True) - # Trip a fast read to confirm the file is actually a SQLite - # database. A truncated download will fail here rather than - # later when we try to scan the files table. - db.execute('SELECT name FROM sqlite_master LIMIT 1').fetchone() - db.execute('SELECT 1 FROM files LIMIT 1').fetchone() - self._db = db - except sqlite3.DatabaseError as e: - print( - f" [minerva] hashes.db at {db_path} is unusable ({e}); " - f"skipping. Re-run `python workflow.py` to resume the " - f"download (a .part file is preserved on disk)." - ) - self._artefacts_unusable = True - return False - - if index_path is not None: - try: - self._index = _load_index(index_path) - except Exception as e: - print(f" [minerva] index unreadable ({index_path}): {e}; deriving from hashes.db.") - - if self._index is None: - self._index = [ - row[0] for row in self._db.execute('SELECT full_path FROM files') - ] - print(f" [minerva] index derived from hashes.db ({len(self._index):,} paths).") - - return True - - def scrape( - self, - platform: str, - config: PlatformConfig, - ctx: BuildContext, - ) -> list[dict[str, Any]]: - if not self._ensure_artefacts(): - return [] - assert self._index is not None and self._db is not None - try: - return scrape_with_artefacts( - config, platform, index=self._index, db=self._db, - ) - except sqlite3.DatabaseError as e: - print(f" [minerva] query failed ({e}); skipping rest of build.") - self._artefacts_unusable = True - return [] - - -SOURCE = MinervaSource diff --git a/db/sources/minerva/source.yml b/db/sources/minerva/source.yml deleted file mode 100644 index c665b681..00000000 --- a/db/sources/minerva/source.yml +++ /dev/null @@ -1,69 +0,0 @@ -id: minerva -name: MiNERVA Archive -homepage: https://minerva-archive.org -kind: catalog -priority: 200 -auth: - required: false -delivery: torrent -capabilities: - - torrent_distribution - - selective_file_download # multi-ROM packs; per-file libtorrent priorities - - filename_listing -platforms: - - 32x - - 3do - - 3ds - - a26 - - a52 - - a78 - - cdi - - cv - - dc - - dsi - - fds - - fmt - - gb - - gba - - gbc - - gc - - gg - - intv - - jag - - jcd - - lynx - - min - - n3ds - - n64 - - ndd - - nds - - nes - - ngcd - - pc98 - - pcfx - - pip - - ps1 - - ps2 - - ps3 - - psp - - psv - - sat - - scd - - smd - - sms - - snes - - tg16 - - tgcd - - vb - - wii - - wiiu - - x360 - - xbox -notes: | - ROMs are distributed as torrents — typically multi-ROM packs. The - scraper joins assets/index.txt.gz (paths) with assets/hashes.db - (per-file magnets/so_id) and emits one entry per ROM file with - (torrent_infohash, torrent_file_index) for selective download. - - platforms.yml entries put MiNERVA path prefixes in `urls`, e.g. - './No-Intro/Nintendo - Nintendo Entertainment System (Headered)/'. diff --git a/db/sources/nopaystation/__init__.py b/db/sources/nopaystation/__init__.py deleted file mode 100644 index 66b92483..00000000 --- a/db/sources/nopaystation/__init__.py +++ /dev/null @@ -1 +0,0 @@ -from .scraper import SOURCE # noqa: F401 diff --git a/db/sources/nopaystation/scraper.py b/db/sources/nopaystation/scraper.py deleted file mode 100644 index d6ba0c38..00000000 --- a/db/sources/nopaystation/scraper.py +++ /dev/null @@ -1,248 +0,0 @@ -""" -NoPayStation source plugin. - -Scrapes the NoPayStation TSV database for PS3/PSV titles. Generates RAP -(PS3) and ZRIF (PSV) key files into static/content/ alongside the -catalog DB on the GitHub raw mirror. -""" -import os -import csv -import io -import re -import xml.etree.ElementTree as ET -from typing import Any - -import requests - -from utils import cache_manager -from utils.scrape_utils import fetch_url -from utils.parse_utils import size_bytes_to_str, join_urls - -from core.contract import BuildContext, PlatformConfig, SourceManifest - - -HOST_NAME = 'NoPayStation' - -REGIONS_MAP = { - 'US': 'us', - 'EU': 'eu', - 'JP': 'jp' -} - -# Base URL for static content hosted in the repository -MAIN_SITE = 'https://raw.githubusercontent.com/caprado/romgi/main/db' - -# Directories and base URLs for PS3 RAP files and PSV ZRIF files -PS3_RAPS_DIR = 'static/content/ps3/raps' -PS3_RAPS_BASE_URL = f'{MAIN_SITE}/static/content/ps3/raps' - -PSV_ZRIFS_DIR = 'static/content/psv/zrifs' -PSV_ZRIFS_BASE_URL = f'{MAIN_SITE}/static/content/psv/zrifs' - - -def _title_filename(name: str, url: str) -> str: - """CDN URLs end in an opaque token; name the file after the title.""" - token = url.rstrip('/').split('/')[-1] - ext = os.path.splitext(token)[1] or '.pkg' - safe = re.sub(r'[<>:"/\\|?*]', '', name).strip().rstrip('.') - return f'{safe}{ext}' if safe else token - - -def create_rap_file(rap: str, filepath: str) -> None: - """Create a RAP file from a hex string.""" - with open(filepath, 'wb') as f: - f.write(bytes.fromhex(rap)) - - -def create_zrif_file(zrif: str, filepath: str) -> None: - """Create a ZRIF file from a string.""" - with open(filepath, 'w') as f: - f.write(zrif) - - -def add_ps3_links(result: dict[str, Any], links: list[dict[str, Any]], base_url: str) -> None: - """Add PS3-specific links (e.g., RAP files) to the links list.""" - name = result['Name'] - rap = result['RAP'] - content_id = result['Content ID'] - - if len(rap) == 32 and content_id: - filename = f'{content_id}.rap' - filepath = os.path.join(PS3_RAPS_DIR, filename) - create_rap_file(rap, filepath) - - links.append({ - 'name': name, - 'type': 'RAP file', - 'format': 'rap', - 'url': join_urls(PS3_RAPS_BASE_URL, filename), - 'filename': filename, - 'host': HOST_NAME, - 'size': 16, - 'size_str': size_bytes_to_str(16), - 'source_url': base_url - }) - - -def add_psv_links(result: dict[str, Any], links: list[dict[str, Any]], base_url: str) -> None: - """Add PSV-specific links (e.g., ZRIF strings) to the links list.""" - name = result['Name'] - zrif = result['zRIF'] - content_id = result['Content ID'] - - if zrif and content_id: - filename = content_id - filepath = os.path.join(PSV_ZRIFS_DIR, filename) - create_zrif_file(zrif, filepath) - - links.append({ - 'name': name, - 'type': 'ZRIF string', - 'format': 'string', - 'url': join_urls(PSV_ZRIFS_BASE_URL, filename), - 'filename': filename, - 'host': HOST_NAME, - 'size': len(zrif), - 'size_str': size_bytes_to_str(len(zrif)), - 'source_url': base_url - }) - - -def parse_links(result: dict[str, Any], source: dict[str, Any], platform: str, base_url: str) -> list[dict[str, Any]]: - """Parse links from the result and generate metadata for each link.""" - links = [] - url = result['PKG direct link'] - if not url.startswith('http'): - return links - - name = result['Name'] - filename = _title_filename(name, url) - file_size_val = result.get('File Size', '') - size = round(float(file_size_val)) if file_size_val and file_size_val.isdigit() else 0 - size_str = size_bytes_to_str(size) if size else '0B' - - if url.endswith('.xml'): - # Handle XML files containing multiple URLs - r = requests.get(url) - if r.ok: - root = ET.fromstring(r.text) - urls = [piece.attrib['url'] for piece in root.findall('pieces')] - for i, url in enumerate(urls): - filename = _title_filename(f'{name} (part {i})', url) - - links.append({ - 'name': name, - 'type': f"{source['type']} #{i}", - 'format': source['format'], - 'url': url, - 'filename': filename, - 'host': HOST_NAME, - 'size': size, - 'size_str': size_str, - 'source_url': base_url - }) - else: - # Handle direct links - links.append({ - 'name': name, - 'type': source['type'], - 'format': source['format'], - 'url': url, - 'filename': filename, - 'host': HOST_NAME, - 'size': size, - 'size_str': size_str, - 'source_url': base_url - }) - - # Add platform-specific links - if platform == 'ps3': - add_ps3_links(result, links, base_url) - elif platform == 'psv': - add_psv_links(result, links, base_url) - - return links - - -def create_entry(result: dict[str, Any], source: dict[str, Any], platform: str, base_url: str) -> dict[str, Any]: - """Create an entry for a ROM based on the result data.""" - rom_id = result['Title ID'] - name = result['Name'] - region = REGIONS_MAP.get(result['Region'], 'other') - links = parse_links(result, source, platform, base_url) - - return { - 'rom_id': rom_id, - 'title': name, - 'platform': platform, - 'regions': [region], - 'links': links - } - - -def parse_response(response: str, source: dict[str, Any], platform: str, base_url: str) -> list[dict[str, Any]]: - """Parse the response and extract entries.""" - entries = [] - results = csv.DictReader(io.StringIO(response), delimiter='\t') - - for result in results: - entry = create_entry(result, source, platform, base_url) - if entry and entry['links']: - entries.append(entry) - - return entries - - -def fetch_response(url: str, use_cached: bool) -> str | None: - """Fetch the response from a URL, optionally using a cached version.""" - short_url = url.split('/')[-1][:50] if '/' in url else url[:50] - - if use_cached: - response = cache_manager.get_cached_response(url) - if response: - print(f" {short_url}... cached") - return response - - return fetch_url(url) - - -def scrape(source: dict[str, Any], platform: str, use_cached: bool = False) -> list[dict[str, Any]]: - """Scrape data from the source and extract entries.""" - # Ensure directories exist - for path in (PS3_RAPS_DIR, PSV_ZRIFS_DIR): - os.makedirs(path, exist_ok=True) - - entries = [] - - for url in source['urls']: - response = fetch_response(url, use_cached) - if not response: - print(f"Warning: Failed to get response from {url}, skipping...") - continue - - parsed_entries = parse_response(response, source, platform, url) - if not parsed_entries: - print(f"Warning: No entries parsed from {url}, skipping...") - continue - - entries.extend(parsed_entries) - - return entries - - -class NoPayStationSource: - """Adapter from the plugin contract to the legacy scrape().""" - - def __init__(self, manifest: SourceManifest): - self.manifest = manifest - - def scrape( - self, - platform: str, - config: PlatformConfig, - ctx: BuildContext, - ) -> list[dict[str, Any]]: - return scrape(config.to_legacy_dict(), platform, ctx.use_cached) - - -SOURCE = NoPayStationSource diff --git a/db/sources/nopaystation/source.yml b/db/sources/nopaystation/source.yml deleted file mode 100644 index 89f15150..00000000 --- a/db/sources/nopaystation/source.yml +++ /dev/null @@ -1,17 +0,0 @@ -id: nopaystation -name: NoPayStation -homepage: https://nopaystation.com -kind: catalog -priority: 80 -auth: - required: false -capabilities: - - http_range_resume - - filename_listing - - generates_keys # produces RAP/ZRIF key files alongside ROM links -platforms: - - ps3 - - psv -notes: | - PS3/PSV catalog. Generates RAP/ZRIF key files into static/content/ - alongside the catalog DB on the GitHub raw mirror. diff --git a/db/tests/__init__.py b/db/tests/__init__.py deleted file mode 100644 index e69de29b..00000000 diff --git a/db/tests/conftest.py b/db/tests/conftest.py deleted file mode 100644 index 8e101983..00000000 --- a/db/tests/conftest.py +++ /dev/null @@ -1,30 +0,0 @@ -"""Shared fixtures and sys.path setup for all db/ tests.""" -from __future__ import annotations - -import sys -from pathlib import Path -from typing import Any - -import pytest - -DB_ROOT = Path(__file__).resolve().parent.parent -if str(DB_ROOT) not in sys.path: - sys.path.insert(0, str(DB_ROOT)) - - -@pytest.fixture() -def sample_entry() -> dict[str, Any]: - return { - 'title': 'Super Mario Bros.', - 'platform': 'nes', - 'regions': ['us'], - 'links': [], - } - - -@pytest.fixture() -def tmp_cache_dir(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> Path: - cache_dir = tmp_path / 'cache' - cache_dir.mkdir() - monkeypatch.setattr('utils.cache_manager.CACHE_DIRNAME', str(cache_dir)) - return cache_dir diff --git a/db/tests/fixtures/__init__.py b/db/tests/fixtures/__init__.py deleted file mode 100644 index e69de29b..00000000 diff --git a/db/tests/fixtures/build_minerva_fixture.py b/db/tests/fixtures/build_minerva_fixture.py deleted file mode 100644 index be63e379..00000000 --- a/db/tests/fixtures/build_minerva_fixture.py +++ /dev/null @@ -1,127 +0,0 @@ -""" -Build a synthetic MiNERVA-shaped SQLite fixture for tests. - -Mirrors enough of `assets/hashes.db`'s public schema for the scraper to -run end-to-end without the real 1.76 GB upstream file. The resulting -SQLite is committed; rerun this script to refresh it. - - python tests/fixtures/build_minerva_fixture.py -""" -from __future__ import annotations - -import gzip -import sqlite3 -import sys -from pathlib import Path - - -HERE = Path(__file__).resolve().parent -INDEX_PATH = HERE / "minerva_index.txt.gz" -DB_PATH = HERE / "minerva_hashes.db" - -# Two torrents (different infohashes, different platforms). -NES_HASH = "aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa" -SNES_HASH = "bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb" - -NES_PREFIX = "./No-Intro/Nintendo - Nintendo Entertainment System (Headered)/" -SNES_PREFIX = "./No-Intro/Nintendo - Super Nintendo Entertainment System/" -RANDOM_PREFIX = "./Miscellaneous/random/" - -ROWS = [ - # (full_path, file_name, size, magnet, so_id, torrents) - ( - f"{NES_PREFIX}Super Mario Bros. (USA).zip", - "Super Mario Bros. (USA).zip", - 40960, - f"magnet:?xt=urn:btih:{NES_HASH}&dn=NES%20Pack", - "0", - "nes_pack.torrent", - ), - ( - f"{NES_PREFIX}Legend of Zelda, The (USA).zip", - "Legend of Zelda, The (USA).zip", - 131072, - f"magnet:?xt=urn:btih:{NES_HASH}&dn=NES%20Pack", - "1", - "nes_pack.torrent", - ), - ( - f"{NES_PREFIX}Megaman 2 (USA).zip", - "Megaman 2 (USA).zip", - 262144, - f"magnet:?xt=urn:btih:{NES_HASH}&dn=NES%20Pack", - "2", - "nes_pack.torrent", - ), - ( - f"{SNES_PREFIX}Super Metroid (USA).zip", - "Super Metroid (USA).zip", - 3145728, - f"magnet:?xt=urn:btih:{SNES_HASH}&dn=SNES%20Pack", - "0", - "snes_pack.torrent", - ), - ( - f"{SNES_PREFIX}Final Fantasy III (USA).zip", - "Final Fantasy III (USA).zip", - 2097152, - f"magnet:?xt=urn:btih:{SNES_HASH}&dn=SNES%20Pack", - "1", - "snes_pack.torrent", - ), - # Out-of-scope — present in index but neither prefix matches. - ( - f"{RANDOM_PREFIX}readme.txt", - "readme.txt", - 100, - "magnet:?xt=urn:btih:cccccccccccccccccccccccccccccccccccccccc", - "0", - "misc.torrent", - ), -] - - -def write_index(rows: list[tuple]) -> None: - paths = [r[0] for r in rows] - with gzip.open(INDEX_PATH, "wb") as f: - f.write(("\n".join(paths) + "\n").encode("utf-8")) - - -def write_db(rows: list[tuple]) -> None: - if DB_PATH.exists(): - DB_PATH.unlink() - con = sqlite3.connect(DB_PATH) - cur = con.cursor() - cur.execute(""" - CREATE TABLE files ( - full_path TEXT PRIMARY KEY, - file_name TEXT, - size INTEGER, - magnet TEXT, - so_id TEXT, - torrents TEXT, - crc32 TEXT, - md5 TEXT, - sha1 TEXT - ) - """) - for full_path, file_name, size, magnet, so_id, torrents in rows: - cur.execute( - "INSERT INTO files (full_path, file_name, size, magnet, so_id, torrents) " - "VALUES (?, ?, ?, ?, ?, ?)", - (full_path, file_name, size, magnet, so_id, torrents), - ) - con.commit() - con.close() - - -def main() -> None: - write_index(ROWS) - write_db(ROWS) - print(f"wrote {INDEX_PATH} ({INDEX_PATH.stat().st_size} B)") - print(f"wrote {DB_PATH} ({DB_PATH.stat().st_size} B)") - - -if __name__ == "__main__": - main() - sys.exit(0) diff --git a/db/tests/fixtures/minerva_index.txt.gz b/db/tests/fixtures/minerva_index.txt.gz deleted file mode 100644 index d38d38b6..00000000 Binary files a/db/tests/fixtures/minerva_index.txt.gz and /dev/null differ diff --git a/db/tests/test_cache_manager.py b/db/tests/test_cache_manager.py deleted file mode 100644 index e439fc27..00000000 --- a/db/tests/test_cache_manager.py +++ /dev/null @@ -1,86 +0,0 @@ -"""Tests for utils/cache_manager.py.""" -from __future__ import annotations - -import os -import time -from pathlib import Path - -import pytest -from utils import cache_manager - - -# -- get_cached_response_filename -------------------------------------------- - -def test_filename_replaces_special_chars(): - result = cache_manager.get_cached_response_filename('https://example.com/page?q=1&x=2') - assert ':' not in result - assert '?' not in result - assert '/' not in result or '\\' not in result - - -def test_filename_simple(): - result = cache_manager.get_cached_response_filename('simple_url') - assert result == 'simple_url' - - -# -- cache_response + get_cached_response ------------------------------------ - -def test_round_trip(tmp_cache_dir: Path): - url = 'https://example.com/test' - cache_manager.cache_response(url, 'content') - result = cache_manager.get_cached_response(url) - assert result == 'content' - - -def test_cache_miss(tmp_cache_dir: Path): - assert cache_manager.get_cached_response('https://never-cached.com') is None - - -def test_cache_never_expire(tmp_cache_dir: Path): - url = 'https://example.com/forever' - cache_manager.cache_response(url, 'data') - result = cache_manager.get_cached_response(url, max_age_days=0) - assert result == 'data' - - -def test_cache_expired(tmp_cache_dir: Path, monkeypatch: pytest.MonkeyPatch): - url = 'https://example.com/old' - cache_manager.cache_response(url, 'old data') - filename = cache_manager.get_cached_response_filename(url) - filepath = str(tmp_cache_dir / filename) - old_time = time.time() - (10 * 86400) - os.utime(filepath, (old_time, old_time)) - assert cache_manager.get_cached_response(url, max_age_days=7) is None - - -def test_cache_utf8(tmp_cache_dir: Path): - url = 'https://example.com/unicode' - content = 'Pokémon ポケモン 宝可梦' - cache_manager.cache_response(url, content) - assert cache_manager.get_cached_response(url) == content - - -# -- get_cache_age_days ------------------------------------------------------ - -def test_age_not_cached(tmp_cache_dir: Path): - assert cache_manager.get_cache_age_days('https://not-here.com') is None - - -def test_age_just_cached(tmp_cache_dir: Path): - url = 'https://example.com/fresh' - cache_manager.cache_response(url, 'data') - age = cache_manager.get_cache_age_days(url) - assert age is not None - assert age < 0.01 - - -def test_age_old_file(tmp_cache_dir: Path): - url = 'https://example.com/aged' - cache_manager.cache_response(url, 'data') - filename = cache_manager.get_cached_response_filename(url) - filepath = str(tmp_cache_dir / filename) - old_time = time.time() - (2 * 86400) - os.utime(filepath, (old_time, old_time)) - age = cache_manager.get_cache_age_days(url) - assert age is not None - assert 1.9 < age < 2.1 diff --git a/db/tests/test_download_retroachievements.py b/db/tests/test_download_retroachievements.py deleted file mode 100644 index 1e4233fc..00000000 --- a/db/tests/test_download_retroachievements.py +++ /dev/null @@ -1,284 +0,0 @@ -"""Tests for the RetroAchievements download script. - -Focus is the rate limiter and resilience: throttling spaces requests, 429/5xx -back off (honouring Retry-After), give-up after the retry cap, atomic writes, -and graceful no-op without credentials. No real network is used. -""" -from __future__ import annotations - -import json -import sys -from pathlib import Path - -import pytest -import requests - -DB_ROOT = Path(__file__).resolve().parent.parent -if str(DB_ROOT) not in sys.path: - sys.path.insert(0, str(DB_ROOT)) - -import scripts.download_retroachievements as dl # noqa: E402 - - -# --- test doubles ---------------------------------------------------------- - -class FakeResponse: - def __init__(self, status_code, *, json_data=None, headers=None, bad_json=False): - self.status_code = status_code - self._json = json_data - self.headers = headers or {} - self._bad_json = bad_json - - def json(self): - if self._bad_json: - raise ValueError('no json') - return self._json - - -class FakeSession: - """Yields a scripted sequence of responses (or raises) per .get() call.""" - - def __init__(self, outcomes): - self._outcomes = list(outcomes) - self.calls = 0 - - def get(self, url, params=None, timeout=None, headers=None): - outcome = self._outcomes[self.calls] - self.calls += 1 - if isinstance(outcome, Exception): - raise outcome - return outcome - - -def _seq(values): - it = iter(values) - return lambda: next(it) - - -@pytest.fixture(autouse=True) -def reset_state(monkeypatch): - dl._last_request_time = 0.0 - # Never actually sleep in tests. - monkeypatch.setattr(dl.time, 'sleep', lambda s: None) - yield - - -# --- throttle -------------------------------------------------------------- - -def test_throttle_sleeps_to_maintain_min_interval(monkeypatch): - sleeps = [] - monkeypatch.setattr(dl.time, 'sleep', lambda s: sleeps.append(s)) - monkeypatch.setattr(dl.time, 'monotonic', _seq([100.1, 101.0])) - dl._last_request_time = 100.0 # only 0.1s since last call - - dl._throttle(1.0) - - assert sleeps == [pytest.approx(0.9)] - - -def test_throttle_no_sleep_when_interval_elapsed(monkeypatch): - sleeps = [] - monkeypatch.setattr(dl.time, 'sleep', lambda s: sleeps.append(s)) - monkeypatch.setattr(dl.time, 'monotonic', _seq([105.0, 105.0])) - dl._last_request_time = 100.0 # 5s since last call - - dl._throttle(1.0) - - assert sleeps == [] - - -# --- retry_delay ----------------------------------------------------------- - -def test_retry_delay_prefers_retry_after_header(): - resp = FakeResponse(429, headers={'Retry-After': '7'}) - assert dl._retry_delay(resp, fallback=2.0) == 7.0 # type: ignore[arg-type] - - -def test_retry_delay_falls_back_on_missing_or_bad_header(): - assert dl._retry_delay(FakeResponse(429), fallback=2.0) == 2.0 # type: ignore[arg-type] - bad = FakeResponse(429, headers={'Retry-After': 'soon'}) - assert dl._retry_delay(bad, fallback=3.0) == 3.0 # type: ignore[arg-type] - - -# --- _get_with_backoff ----------------------------------------------------- - -def test_get_returns_200(monkeypatch): - monkeypatch.setattr(dl, '_throttle', lambda mi: None) - resp = FakeResponse(200, json_data=[{'ok': True}]) - session = FakeSession([resp]) - - out = dl._get_with_backoff(session, 'url', {}, max_retries=3) # type: ignore[arg-type] - - assert out is resp - assert session.calls == 1 - - -def test_get_retries_on_429_then_succeeds(monkeypatch): - monkeypatch.setattr(dl, '_throttle', lambda mi: None) - sleeps = [] - monkeypatch.setattr(dl.time, 'sleep', lambda s: sleeps.append(s)) - session = FakeSession([ - FakeResponse(429, headers={'Retry-After': '7'}), - FakeResponse(200, json_data=[]), - ]) - - out = dl._get_with_backoff(session, 'url', {}, max_retries=3) # type: ignore[arg-type] - - assert out is not None - assert out.status_code == 200 - assert session.calls == 2 - assert sleeps == [7.0] # honoured Retry-After - - -def test_get_uses_exponential_backoff_without_header(monkeypatch): - monkeypatch.setattr(dl, '_throttle', lambda mi: None) - sleeps = [] - monkeypatch.setattr(dl.time, 'sleep', lambda s: sleeps.append(s)) - session = FakeSession([ - FakeResponse(500), - FakeResponse(503), - FakeResponse(200, json_data=[]), - ]) - - out = dl._get_with_backoff(session, 'url', {}, max_retries=4) # type: ignore[arg-type] - - assert out is not None - assert out.status_code == 200 - # BASE_BACKOFF then doubled. - assert sleeps == [dl.BASE_BACKOFF, dl.BASE_BACKOFF * 2] - - -def test_get_gives_up_after_max_retries(monkeypatch): - monkeypatch.setattr(dl, '_throttle', lambda mi: None) - session = FakeSession([FakeResponse(500), FakeResponse(500)]) - - out = dl._get_with_backoff(session, 'url', {}, max_retries=2) # type: ignore[arg-type] - - assert out is None - assert session.calls == 2 - - -def test_get_non_retryable_4xx_returns_none(monkeypatch): - monkeypatch.setattr(dl, '_throttle', lambda mi: None) - session = FakeSession([FakeResponse(404)]) - - out = dl._get_with_backoff(session, 'url', {}, max_retries=3) # type: ignore[arg-type] - - assert out is None - assert session.calls == 1 # not retried - - -def test_get_retries_on_request_exception(monkeypatch): - monkeypatch.setattr(dl, '_throttle', lambda mi: None) - session = FakeSession([ - requests.RequestException('boom'), - FakeResponse(200, json_data=[]), - ]) - - out = dl._get_with_backoff(session, 'url', {}, max_retries=3) # type: ignore[arg-type] - - assert out is not None - assert out.status_code == 200 - assert session.calls == 2 - - -# --- _write_atomic --------------------------------------------------------- - -def test_write_atomic_writes_json(tmp_path): - dest = tmp_path / '7.json' - games = [{'Title': 'X', 'ID': 1, 'NumAchievements': 5}] - - dl._write_atomic(str(dest), games) - - assert json.loads(dest.read_text(encoding='utf-8')) == games - assert not (tmp_path / '7.json.tmp').exists() # temp cleaned up - - -# --- download_retroachievements ------------------------------------------- - -@pytest.fixture() -def ra_dir(tmp_path, monkeypatch): - d = tmp_path / 'data_ra' - monkeypatch.setattr(dl, 'DATA_DIR', str(d)) - return d - - -def _set_creds(monkeypatch): - monkeypatch.setenv('RA_API_USER', 'user') - monkeypatch.setenv('RA_API_KEY', 'key') - - -def test_download_noop_without_credentials(monkeypatch, ra_dir): - monkeypatch.delenv('RA_API_USER', raising=False) - monkeypatch.delenv('RA_API_KEY', raising=False) - - def boom(*a, **k): - raise AssertionError('should not hit the network without creds') - - monkeypatch.setattr(dl, '_get_with_backoff', boom) - - dl.download_retroachievements() - - assert not ra_dir.exists() - - -def test_download_writes_console_file(monkeypatch, ra_dir): - _set_creds(monkeypatch) - monkeypatch.setattr(dl, 'RA_CONSOLES', {'nes': 7}) - games = [{'Title': 'Super Mario Bros.', 'ID': 111, 'NumAchievements': 30}] - monkeypatch.setattr( - dl, '_get_with_backoff', - lambda *a, **k: FakeResponse(200, json_data=games), - ) - - dl.download_retroachievements() - - written = json.loads((ra_dir / '7.json').read_text(encoding='utf-8')) - assert written == games - - -def test_download_skips_empty_or_invalid_response(monkeypatch, ra_dir): - _set_creds(monkeypatch) - monkeypatch.setattr(dl, 'RA_CONSOLES', {'nes': 7, 'smd': 1}) - - responses = { - 7: FakeResponse(200, json_data=[]), # empty list - 1: FakeResponse(200, bad_json=True), # not JSON - } - - def fake_get(session, url, params, **kwargs): - return responses[params['i']] - - monkeypatch.setattr(dl, '_get_with_backoff', fake_get) - - dl.download_retroachievements() - - # Neither console produced a cache file. - assert not (ra_dir / '7.json').exists() - assert not (ra_dir / '1.json').exists() - - -def test_download_failed_fetch_does_not_write(monkeypatch, ra_dir): - _set_creds(monkeypatch) - monkeypatch.setattr(dl, 'RA_CONSOLES', {'nes': 7}) - monkeypatch.setattr(dl, '_get_with_backoff', lambda *a, **k: None) - - dl.download_retroachievements() - - assert not (ra_dir / '7.json').exists() - - -def test_download_use_cached_skips_network(monkeypatch, ra_dir): - _set_creds(monkeypatch) - monkeypatch.setattr(dl, 'RA_CONSOLES', {'nes': 7}) - ra_dir.mkdir(parents=True) - (ra_dir / '7.json').write_text('[]', encoding='utf-8') - - def boom(*a, **k): - raise AssertionError('use_cached must not hit the network') - - monkeypatch.setattr(dl, '_get_with_backoff', boom) - - dl.download_retroachievements(use_cached=True) - # File untouched, no exception raised. - assert (ra_dir / '7.json').read_text(encoding='utf-8') == '[]' diff --git a/db/tests/test_grouping.py b/db/tests/test_grouping.py deleted file mode 100644 index e47684da..00000000 --- a/db/tests/test_grouping.py +++ /dev/null @@ -1,212 +0,0 @@ -""" -Entry-grouping guarantees: - -- The disc strategy recognizes the positioned ``(Disc N)`` family and derives - a stable, region-aware key with the token stripped. -- Real-catalog traps (bare ``(Disc)``, ``(Disk Writer)``, ``(Side N)``) are - NOT treated as disc positions. -- Letter and roman disc labels order correctly. -- build_groups only promotes buckets with enough members AND >1 distinct - position; region and platform partition groups. -- Strategy discovery finds the disc plugin. - -Run from db/: python -m pytest tests -""" -from __future__ import annotations - -import sys -from pathlib import Path -from typing import Any - -DB_ROOT = Path(__file__).resolve().parent.parent -if str(DB_ROOT) not in sys.path: - sys.path.insert(0, str(DB_ROOT)) - -from grouping import EntryRef, build_groups, load_strategies # noqa: E402 -from grouping.strategies.disc import ( # noqa: E402 - DiscStrategy, - _parse_index, -) - - -def _entry(title, platform="ps1", regions=("us",)): - return EntryRef(slug=title.lower().replace(" ", "-"), title=title, - platform=platform, regions=tuple(regions)) - - -# -- disc token recognition -------------------------------------------------- - -def test_matches_numeric_disc(): - m = DiscStrategy().membership(_entry("Final Fantasy VII (Disc 1)")) - assert m is not None - assert m.index == 1 - assert m.label == "Disc 1" - assert m.title == "Final Fantasy VII" - - -def test_key_stable_across_discs(): - s = DiscStrategy() - a = s.membership(_entry("Baten Kaitos (Disc 1)", platform="gc")) - b = s.membership(_entry("Baten Kaitos (Disc 2)", platform="gc")) - assert a is not None and b is not None - assert a.key == b.key - assert a.index != b.index - - -def test_zero_padded_and_of_tail(): - s = DiscStrategy() - padded = s.membership(_entry("X (Disc 01)")) - assert padded is not None and padded.index == 1 - tail = s.membership(_entry("X (Disc 1 of 3)")) - assert tail is not None and tail.index == 1 - - -# -- traps (must NOT be treated as disc positions) --------------------------- - -def test_bare_disc_is_ignored(): - assert DiscStrategy().membership(_entry("Some Game (Disc)")) is None - - -def test_disk_writer_is_ignored(): - assert DiscStrategy().membership(_entry("Famicom Thing (Disk Writer)")) is None - - -def test_side_token_is_ignored(): - assert DiscStrategy().membership(_entry("Tape Game (Side 1)")) is None - - -def test_plain_title_is_ignored(): - assert DiscStrategy().membership(_entry("Chrono Trigger")) is None - - -# -- index parsing ----------------------------------------------------------- - -def test_parse_numeric(): - assert _parse_index("1") == 1 - assert _parse_index("05") == 5 - - -def test_parse_letter_series(): - assert _parse_index("A") == 1 - assert _parse_index("B") == 2 - assert _parse_index("C") == 3 # letter series, not roman 100 - - -def test_parse_roman(): - assert _parse_index("I") == 1 - assert _parse_index("II") == 2 - assert _parse_index("IV") == 4 - - -def test_letter_and_roman_discs_group(): - s = DiscStrategy() - a = s.membership(_entry("Lunar (Disc A)")) - b = s.membership(_entry("Lunar (Disc B)")) - assert a is not None and b is not None - assert a.index == 1 and b.index == 2 and a.key == b.key - - -# -- build_groups promotion -------------------------------------------------- - -def test_two_discs_form_a_group(): - entries = [ - _entry("Baten Kaitos (Disc 1)", platform="gc"), - _entry("Baten Kaitos (Disc 2)", platform="gc"), - _entry("Chrono Trigger", platform="snes"), - ] - groups = build_groups(entries, load_strategies()) - assert len(groups) == 1 - g = groups[0] - assert g.kind == "disc" - assert g.title == "Baten Kaitos" - assert [m.index for m in g.members] == [1, 2] - assert g.id.startswith("disc:") - - -def test_lone_disc_one_not_promoted(): - entries = [_entry("Solo (Disc 1)")] - assert build_groups(entries, load_strategies()) == [] - - -def test_same_index_twice_not_promoted(): - # Two dumps both labelled Disc 1 (e.g. revisions) — no real second disc. - entries = [ - EntryRef("a", "Game (Disc 1)", "ps1", ("us",)), - EntryRef("b", "Game (Disc 1)", "ps1", ("us",)), - ] - assert build_groups(entries, load_strategies()) == [] - - -def test_region_partitions_groups(): - entries = [ - _entry("Game (Disc 1)", regions=("us",)), - _entry("Game (Disc 2)", regions=("us",)), - _entry("Game (Disc 1)", regions=("eu",)), - _entry("Game (Disc 2)", regions=("eu",)), - ] - groups = build_groups(entries, load_strategies()) - assert len(groups) == 2 - assert len({g.id for g in groups}) == 2 # distinct keys - assert all(len(g.members) == 2 for g in groups) - - -def test_platform_partitions_groups(): - entries = [ - _entry("Game (Disc 1)", platform="ps1"), - _entry("Game (Disc 2)", platform="ps1"), - _entry("Game (Disc 1)", platform="sat"), - _entry("Game (Disc 2)", platform="sat"), - ] - groups = build_groups(entries, load_strategies()) - assert len(groups) == 2 - - -# -- discovery --------------------------------------------------------------- - -def test_disc_strategy_is_discovered(): - kinds = {s.kind for s in load_strategies()} - assert "disc" in kinds - - -# -- end-to-end build write path --------------------------------------------- - -def test_build_entry_groups_persists(tmp_path, monkeypatch): - """Exercise the real build wiring: schema -> insert -> group -> store.""" - monkeypatch.chdir(tmp_path) - from database import db_manager - from make import build_entry_groups - - db_manager.con = None - db_manager.cur = None - db_manager.init_database() - try: - for title in [ - "Baten Kaitos (Disc 1)", - "Baten Kaitos (Disc 2)", - "Chrono Trigger", - ]: - db_manager.insert_entry({ - "title": title, "platform": "gc", "regions": ["us"], "links": [], - }) - - build_entry_groups() - - cur = db_manager.cur - assert cur is not None - groups: list[Any] = cur.execute( - "SELECT id, kind, title, member_count FROM entry_groups" - ).fetchall() - assert len(groups) == 1 - group_id, kind, title, count = groups[0] - assert kind == "disc" - assert title == "Baten Kaitos" - assert count == 2 - - members: list[Any] = cur.execute( - "SELECT member_index FROM entry_group_members " - "WHERE group_id = ? ORDER BY member_index", - (group_id,), - ).fetchall() - assert [m[0] for m in members] == [1, 2] - finally: - db_manager.close_database() diff --git a/db/tests/test_internet_archive_scraper.py b/db/tests/test_internet_archive_scraper.py deleted file mode 100644 index dd6c24b7..00000000 --- a/db/tests/test_internet_archive_scraper.py +++ /dev/null @@ -1,146 +0,0 @@ -"""Tests for sources/internet_archive/scraper.py.""" -from __future__ import annotations - -from pathlib import Path -from typing import Any -from unittest.mock import patch - -from sources.internet_archive import scraper as ia -from sources.internet_archive.scraper import create_entry, extract_entries, fetch_response, get_login_session - -MOCK_HTML_PUBLIC = """ - - - - -
Game One.zip2024-01-011.5M
Game Two.zip2024-01-01256K
Parent Directory
-""" - -MOCK_HTML_RESTRICTED = """ - - - -
Restricted Game.zip2024-01-012.3G
-""" - -MOCK_HTML_MIXED = MOCK_HTML_PUBLIC + MOCK_HTML_RESTRICTED - - -def _source(fmt: str = 'zip', filt: str = r'(.*)\.zip') -> dict[str, Any]: - return { - 'filter': filt, - 'regions': ['us'], - 'type': 'Game', - 'format': fmt, - } - - -BASE_URL = 'https://archive.org/download/test-collection' - - -# -- extract_entries with public links --------------------------------------- - -def test_extract_public_links(): - entries = extract_entries(MOCK_HTML_PUBLIC, _source(), 'nes', BASE_URL) - assert len(entries) == 2 - titles = [e['title'] for e in entries] - assert 'Game One' in titles - assert 'Game Two' in titles - - -def test_extract_applies_filter(): - entries = extract_entries(MOCK_HTML_PUBLIC, _source(filt=r'(.*)\.iso'), 'nes', BASE_URL) - assert len(entries) == 0 - - -def test_extract_skips_parent_dir(): - entries = extract_entries(MOCK_HTML_PUBLIC, _source(), 'nes', BASE_URL) - titles = [e['title'] for e in entries] - assert not any('parent' in t.lower() for t in titles) - - -# -- extract_entries with restricted files ----------------------------------- - -def test_extract_restricted_files(): - entries = extract_entries(MOCK_HTML_RESTRICTED, _source(), 'nes', BASE_URL) - assert len(entries) == 1 - assert entries[0]['title'] == 'Restricted Game' - - -def test_extract_mixed(): - entries = extract_entries(MOCK_HTML_MIXED, _source(), 'nes', BASE_URL) - assert len(entries) == 3 - - -# -- create_entry ------------------------------------------------------------ - -def test_create_entry_structure(): - entry = create_entry('game.zip', 'game.zip', 'Game', '1.5M', _source(), 'nes', BASE_URL) - assert entry['title'] == 'Game' - assert entry['platform'] == 'nes' - assert len(entry['links']) == 1 - link = entry['links'][0] - assert 'url' in link - assert 'filename' in link - assert 'size' in link - - -def test_create_entry_url_join(): - entry = create_entry('subdir/game.zip', 'game.zip', 'Game', '1M', _source(), 'nes', BASE_URL) - assert BASE_URL.split('/')[2] in entry['links'][0]['url'] - - -def test_create_entry_size_conversion(): - entry = create_entry('g.zip', 'g.zip', 'G', '1.5G', _source(), 'nes', BASE_URL) - link = entry['links'][0] - assert link['size'] > 1_000_000_000 - - -# -- get_login_session ------------------------------------------------------- - -def test_login_missing_creds(tmp_path): - result = get_login_session(str(tmp_path / 'nonexistent.json')) - assert result is None - - -def test_login_invalid_json(tmp_path): - bad_file = tmp_path / 'bad.json' - bad_file.write_text('not json') - result = get_login_session(str(bad_file)) - assert result is None - - -# -- fetch_response ---------------------------------------------------------- - -def test_fetch_response_cached(tmp_cache_dir: Path): - from utils import cache_manager - url = 'https://archive.org/download/test' - cache_manager.cache_response(url, 'cached') - result = fetch_response(url, session=None, use_cached=True) - assert result == 'cached' - - -def test_fetch_response_not_cached_fetches(tmp_cache_dir: Path): - with patch.object(ia, 'fetch_url', return_value='fresh') as mock: - result = fetch_response('https://archive.org/download/test', session=None, use_cached=True) - assert result == 'fresh' - mock.assert_called_once() - - -def test_fetch_response_no_cache(tmp_cache_dir: Path): - with patch.object(ia, 'fetch_url', return_value='data'): - result = fetch_response('https://archive.org/download/test', session=None, use_cached=False) - assert result == 'data' - - -# -- extract_entries edge cases ---------------------------------------------- - -def test_extract_html_entities(): - html = '
Rock & Roll.zip2024-01-011M
' - entries = extract_entries(html, _source(), 'nes', BASE_URL) - assert len(entries) == 1 - - -def test_extract_empty_html(): - assert extract_entries('', _source(), 'nes', BASE_URL) == [] - assert extract_entries('', _source(), 'nes', BASE_URL) == [] diff --git a/db/tests/test_make.py b/db/tests/test_make.py deleted file mode 100644 index 66d94249..00000000 --- a/db/tests/test_make.py +++ /dev/null @@ -1,212 +0,0 @@ -"""Tests for make.py — build pipeline helpers.""" -from __future__ import annotations - -from pathlib import Path -from typing import Any -from unittest.mock import MagicMock - -import pytest -import yaml -from make import _build_platform_config, _print_source_header, _tag_links, get_parser, load_platforms - - -# -- get_parser -------------------------------------------------------------- - -def test_get_parser_known(): - assert get_parser('no_intro') is not None - assert get_parser('mame') is not None - assert get_parser('libretro') is not None - - -def test_get_parser_unknown(): - assert get_parser('nonexistent') is None - - -# -- load_platforms ---------------------------------------------------------- - -def test_load_platforms(tmp_path: Path): - data = { - 'nes': [{'source': 'minerva', 'format': 'nes', 'urls': ['/nes/'], 'type': 'Game', 'parsers': {}}], - 'snes': [{'source': 'minerva', 'format': 'sfc', 'urls': ['/snes/'], 'type': 'Game', 'parsers': {}}], - } - yml = tmp_path / 'platforms.yml' - yml.write_text(yaml.dump(data)) - result = load_platforms(yml) - assert 'nes' in result - assert 'snes' in result - assert len(result['nes']) == 1 - - -def test_load_platforms_empty(tmp_path: Path): - yml = tmp_path / 'platforms.yml' - yml.write_text('') - result = load_platforms(yml) - assert result == {} - - -def test_load_platforms_invalid(tmp_path: Path): - yml = tmp_path / 'platforms.yml' - yml.write_text('- just a list') - with pytest.raises(ValueError, match='must be a mapping'): - load_platforms(yml) - - -# -- _build_platform_config -------------------------------------------------- - -def test_build_platform_config(): - entry = { - 'source': 'minerva', - 'format': 'nes', - 'regions': ['us', 'eu'], - 'urls': ['/nes/'], - 'type': 'Game', - 'parsers': {'no_intro': {}}, - 'filter': r'(.*)\.zip', - } - source_id, config = _build_platform_config(entry) - assert source_id == 'minerva' - assert config.format == 'nes' - assert config.regions == ['us', 'eu'] - assert config.urls == ['/nes/'] - assert config.type == 'Game' - assert 'no_intro' in config.parsers - - -def test_build_platform_config_defaults(): - entry = {'source': 'ia', 'format': 'rom'} - source_id, config = _build_platform_config(entry) - assert source_id == 'ia' - assert config.regions == [] - assert config.urls == [] - assert config.filter is None - - -# -- _print_source_header (smoke test) -------------------------------------- - -def test_print_source_header(capsys: pytest.CaptureFixture[str]): - entry = { - 'source': 'minerva', - 'format': 'nes', - 'regions': ['us'], - 'urls': ['/nes/'], - 'type': 'Game', - 'parsers': {}, - } - _, config = _build_platform_config(entry) - _print_source_header(1, 'minerva', config) - captured = capsys.readouterr() - assert 'minerva' in captured.out - assert '[nes]' in captured.out - - -# -- _tag_links -------------------------------------------------------------- - -def test_tag_links_sets_defaults(): - class FakeSource: - class manifest: - id = 'test_source' - auth_required = False - - entries: list[dict[str, Any]] = [ - {'links': [{'name': 'a'}, {'name': 'b'}]}, - {'links': [{'name': 'c'}]}, - ] - ne, nl = _tag_links(entries, FakeSource()) # type: ignore[arg-type] - assert (ne, nl) == (2, 3) - for entry in entries: - for link in entry['links']: - assert link['source_id'] == 'test_source' - assert link['requires_auth'] == 0 - - -def test_tag_links_preserves_overrides(): - class FakeSource: - class manifest: - id = 'ia' - auth_required = True - - entries: list[dict[str, Any]] = [ - {'links': [{'name': 'public', 'requires_auth': 0}, {'name': 'restricted'}]}, - ] - _tag_links(entries, FakeSource()) # type: ignore[arg-type] - assert entries[0]['links'][0]['requires_auth'] == 0 - assert entries[0]['links'][1]['requires_auth'] == 1 - - -# -- process_platforms ------------------------------------------------------- - -def test_process_platforms_runs_pipeline(tmp_path: Path, monkeypatch: pytest.MonkeyPatch): - from database import db_manager - from make import process_platforms - - monkeypatch.chdir(tmp_path) - db_manager.init_database() - - class FakeManifest: - id = 'fake' - name = 'Fake Source' - kind = 'catalog' - homepage = '' - auth_required = False - priority = 0 - capabilities = [] - platforms = ('nes',) - raw = {'id': 'fake', 'name': 'Fake Source', 'kind': 'catalog'} - - class FakeSource: - manifest = FakeManifest() - def scrape(self, platform, config, ctx): - return [{ - 'title': 'Test ROM', - 'platform': 'nes', - 'regions': ['us'], - 'links': [{ - 'name': 'Test ROM', - 'type': 'Game', - 'format': 'nes', - 'url': 'https://example.com/test.zip', - 'filename': 'test.zip', - 'host': 'Test', - 'size': 1024, - 'size_str': '1K', - 'source_url': 'https://example.com', - }], - }] - - db_manager.register_source(FakeManifest()) - - mock_registry = MagicMock() - mock_registry.get.return_value = FakeSource() - - platforms = { - 'nes': [{'source': 'fake', 'format': 'nes', 'urls': [], 'type': 'Game', 'parsers': {}}], - } - - from core.contract import BuildContext - stats = process_platforms(platforms, mock_registry, BuildContext()) - assert 'fake' in stats - assert stats['fake']['entries'] == 1 - assert stats['fake']['links'] == 1 - - db_manager.close_database() - - -def test_process_platforms_source_filter(tmp_path: Path, monkeypatch: pytest.MonkeyPatch): - from database import db_manager - from make import process_platforms - - monkeypatch.chdir(tmp_path) - db_manager.init_database() - - mock_registry = MagicMock() - - platforms = { - 'nes': [{'source': 'minerva', 'format': 'nes', 'urls': [], 'type': 'Game', 'parsers': {}}], - 'snes': [{'source': 'ia', 'format': 'sfc', 'urls': [], 'type': 'Game', 'parsers': {}}], - } - - from core.contract import BuildContext - stats = process_platforms(platforms, mock_registry, BuildContext(), source_filter=['nonexistent']) - assert stats == {} - - db_manager.close_database() diff --git a/db/tests/test_mame_parser.py b/db/tests/test_mame_parser.py deleted file mode 100644 index 60776235..00000000 --- a/db/tests/test_mame_parser.py +++ /dev/null @@ -1,72 +0,0 @@ -"""Tests for parsers/mame.py.""" -from __future__ import annotations - -from pathlib import Path - -import pytest -from parsers import mame - - -@pytest.fixture(autouse=True) -def _reset_roms(): - """Reset module-level roms dict between tests.""" - original = mame.roms - yield - mame.roms = original - - -# -- parse with pre-loaded roms ---------------------------------------------- - -def test_parse_match(): - mame.roms = {'dkong': 'Donkey Kong', 'pacman': 'Pac-Man'} - entries = [{'title': 'dkong'}, {'title': 'pacman'}] - result = mame.parse(entries, {}) - assert result[0]['title'] == 'Donkey Kong' - assert result[0]['rom_id'] == 'dkong' - assert result[1]['title'] == 'Pac-Man' - - -def test_parse_no_match(): - mame.roms = {'dkong': 'Donkey Kong'} - entries = [{'title': 'unknown_rom'}] - result = mame.parse(entries, {}) - assert result[0]['title'] == 'unknown_rom' - assert 'rom_id' not in result[0] - - -def test_parse_empty(): - mame.roms = {} - assert mame.parse([], {}) == [] - - -def test_parse_preserves_other_fields(): - mame.roms = {'dkong': 'Donkey Kong'} - entries = [{'title': 'dkong', 'platform': 'mame', 'regions': ['us']}] - result = mame.parse(entries, {}) - assert result[0]['platform'] == 'mame' - assert result[0]['regions'] == ['us'] - - -# -- load_roms from XML ----------------------------------------------------- - -def test_load_roms_from_xml(tmp_path: Path, monkeypatch: pytest.MonkeyPatch): - xml_content = """ - - Donkey Kong - Pac-Man -""" - xml_file = tmp_path / 'mame_test.xml' - xml_file.write_text(xml_content) - monkeypatch.setattr(mame, 'XMLS_DIR', str(tmp_path)) - mame.roms = {} - mame.load_roms() - assert mame.roms.get('dkong') == 'Donkey Kong' - assert mame.roms.get('pacman') == 'Pac-Man' - - -def test_load_roms_skips_non_xml(tmp_path: Path, monkeypatch: pytest.MonkeyPatch): - (tmp_path / 'readme.txt').write_text('not xml') - monkeypatch.setattr(mame, 'XMLS_DIR', str(tmp_path)) - mame.roms = {} - mame.load_roms() - assert mame.roms == {} diff --git a/db/tests/test_mariocube_scraper.py b/db/tests/test_mariocube_scraper.py deleted file mode 100644 index 81a979f1..00000000 --- a/db/tests/test_mariocube_scraper.py +++ /dev/null @@ -1,91 +0,0 @@ -"""Tests for sources/mariocube/scraper.py.""" -from __future__ import annotations - -from typing import Any - -from sources.mariocube.scraper import create_entry, extract_entries, parse_listing_lines - -MOCK_LISTING = ( - "\x1b[1;34mdrwx\x1b[0m 1.5M Game Title.wad\n" - "\x1b[1;34mdrwx\x1b[0m 256K Another Game.wad\n" - "# comment line\n" - "\n" - "\x1b[1;34mdrwx\x1b[0m 50K Not-A-Wad.zip\n" - "short\n" -) - -BASE_URL = 'https://repo.mariocube.com/WADs/A/' - - -def _source(filt: str = r'(.*)\.wad') -> dict[str, Any]: - return { - 'filter': filt, - 'regions': ['us'], - 'type': 'Game', - 'format': 'wad', - } - - -# -- parse_listing_lines ----------------------------------------------------- - -def test_basic_parsing(): - lines = list(parse_listing_lines(MOCK_LISTING)) - filenames = [f for f, _ in lines] - assert 'Game Title.wad' in filenames - assert 'Another Game.wad' in filenames - - -def test_strips_ansi(): - lines = list(parse_listing_lines(MOCK_LISTING)) - for filename, size in lines: - assert '\x1b' not in filename - assert '\x1b' not in size - - -def test_skips_comments(): - lines = list(parse_listing_lines('# comment\n')) - assert len(lines) == 0 - - -def test_skips_blank_lines(): - lines = list(parse_listing_lines('\n\n\n')) - assert len(lines) == 0 - - -def test_skips_short_lines(): - lines = list(parse_listing_lines('only two\n')) - assert len(lines) == 0 - - -# -- extract_entries --------------------------------------------------------- - -def test_extract_with_filter(): - entries = extract_entries(MOCK_LISTING, _source(), 'wii', BASE_URL) - assert len(entries) == 2 - titles = [e['title'] for e in entries] - assert 'Game Title' in titles - assert 'Another Game' in titles - - -def test_extract_filter_excludes(): - entries = extract_entries(MOCK_LISTING, _source(filt=r'(.*)\.iso'), 'wii', BASE_URL) - assert len(entries) == 0 - - -def test_extract_empty_response(): - assert extract_entries('', _source(), 'wii', BASE_URL) == [] - - -# -- create_entry ------------------------------------------------------------ - -def test_create_entry_structure(): - entry = create_entry('Game.wad', 'Game.wad', 'Game', '1.5M', _source(), 'wii', BASE_URL) - assert entry['title'] == 'Game' - assert entry['platform'] == 'wii' - assert len(entry['links']) == 1 - assert entry['links'][0]['size'] > 0 - - -def test_create_entry_size(): - entry = create_entry('g.wad', 'g.wad', 'G', '256K', _source(), 'wii', BASE_URL) - assert entry['links'][0]['size'] == 256 * 1024 diff --git a/db/tests/test_minerva.py b/db/tests/test_minerva.py deleted file mode 100644 index fea25303..00000000 --- a/db/tests/test_minerva.py +++ /dev/null @@ -1,225 +0,0 @@ -""" -MiNERVA scraper tests. - -Fixture-driven (db/tests/fixtures/minerva_index.txt.gz + -minerva_hashes.db); doesn't touch the 1.76 GB upstream DB. -""" -from __future__ import annotations - -import sqlite3 -import sys -from pathlib import Path - -import pytest - -DB_ROOT = Path(__file__).resolve().parent.parent -if str(DB_ROOT) not in sys.path: - sys.path.insert(0, str(DB_ROOT)) - -from core.contract import BuildContext, PlatformConfig # noqa: E402 -from sources.minerva.scraper import ( # noqa: E402 - MinervaSource, - _extract_infohash, - _extract_trackers, - _load_index, - _select_paths, - scrape_with_artefacts, -) - - -FIXTURES = DB_ROOT / "tests" / "fixtures" -INDEX_FIXTURE = FIXTURES / "minerva_index.txt.gz" -DB_FIXTURE = FIXTURES / "minerva_hashes.db" - -NES_PREFIX = "./No-Intro/Nintendo - Nintendo Entertainment System (Headered)/" -SNES_PREFIX = "./No-Intro/Nintendo - Super Nintendo Entertainment System/" -NES_HASH = "aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa" -SNES_HASH = "bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb" - - -@pytest.fixture(scope="module") -def index() -> list[str]: - assert INDEX_FIXTURE.is_file(), ( - "minerva_index.txt.gz fixture missing; run " - "tests/fixtures/build_minerva_fixture.py" - ) - return _load_index(INDEX_FIXTURE) - - -@pytest.fixture() -def db(): - assert DB_FIXTURE.is_file(), ( - "minerva_hashes.db fixture missing; run " - "tests/fixtures/build_minerva_fixture.py" - ) - con = sqlite3.connect(f"file:{DB_FIXTURE}?mode=ro", uri=True) - yield con - con.close() - - -def _config_for(platform: str, prefixes: list[str]) -> PlatformConfig: - return PlatformConfig( - format=platform, - regions=["us"], - urls=prefixes, - type="Game", - parsers={}, - filter=r"(.*)\.zip", - ) - - -# -- pure helpers ------------------------------------------------------------ - -def test_extract_infohash_hex(): - magnet = f"magnet:?xt=urn:btih:{NES_HASH}&dn=foo" - assert _extract_infohash(magnet) == NES_HASH - - -def test_extract_infohash_returns_none_when_absent(): - assert _extract_infohash("magnet:?dn=foo") is None - assert _extract_infohash("") is None - - -def test_extract_trackers_unwraps_percent_encoding(): - magnet = ( - "magnet:?xt=urn:btih:" + NES_HASH + - "&tr=udp%3A%2F%2Ftracker.example%3A1337%2Fannounce" - ) - assert _extract_trackers(magnet) == [ - "udp://tracker.example:1337/announce" - ] - - -def test_select_paths_prefix_match(): - idx = ["./a/x.zip", "./a/sub/y.zip", "./b/z.zip"] - assert _select_paths(idx, ["./a/"]) == ["./a/x.zip", "./a/sub/y.zip"] - assert _select_paths(idx, ["./b/"]) == ["./b/z.zip"] - - -# -- end-to-end with fixtures ------------------------------------------------ - -def test_scrape_with_artefacts_emits_one_entry_per_file(index, db): - cfg = _config_for("nes", [NES_PREFIX]) - entries = scrape_with_artefacts(cfg, "nes", index=index, db=db) - - titles = sorted(e["title"] for e in entries) - assert titles == [ - "Legend of Zelda, The (USA)", - "Megaman 2 (USA)", - "Super Mario Bros. (USA)", - ] - - -def test_scrape_links_carry_torrent_metadata(index, db): - cfg = _config_for("nes", [NES_PREFIX]) - entries = scrape_with_artefacts(cfg, "nes", index=index, db=db) - - for entry in entries: - link = entry["links"][0] - assert link["host"] == "MiNERVA Archive" - assert link["torrent_infohash"] == NES_HASH - assert link["torrent_file_index"] in {0, 1, 2} - assert link["torrent_file_path"].startswith( - "No-Intro/Nintendo - Nintendo Entertainment System" - ) - meta = link["_torrent_meta"] - assert meta["infohash"] == NES_HASH - assert meta["source_id"] == "minerva" - assert meta["name"] == "nes_pack.torrent" - assert meta["magnet"].startswith("magnet:?xt=urn:btih:") - - -def test_scrape_distinguishes_torrents_per_platform(index, db): - cfg = _config_for("snes", [SNES_PREFIX]) - entries = scrape_with_artefacts(cfg, "snes", index=index, db=db) - titles = sorted(e["title"] for e in entries) - assert titles == ["Final Fantasy III (USA)", "Super Metroid (USA)"] - for entry in entries: - assert entry["links"][0]["torrent_infohash"] == SNES_HASH - - -def test_scrape_filter_excludes_non_matches(index, db): - # filter narrows to .zip; if we provide a filter that matches nothing, - # we get zero entries, not a crash. - cfg = PlatformConfig( - format="nes", - regions=[], - urls=[NES_PREFIX], - type="Game", - parsers={}, - filter=r"(.*)\.never_matches$", - ) - assert scrape_with_artefacts(cfg, "nes", index=index, db=db) == [] - - -def test_scrape_skips_paths_without_db_metadata(index, tmp_path): - """If a path is in the index but missing from hashes.db, drop it.""" - empty_db_path = tmp_path / "empty.db" - con = sqlite3.connect(empty_db_path) - con.execute( - "CREATE TABLE files (full_path TEXT PRIMARY KEY, file_name TEXT, " - "size INTEGER, magnet TEXT, so_id TEXT, torrents TEXT)" - ) - con.commit() - con.close() - - con = sqlite3.connect(f"file:{empty_db_path}?mode=ro", uri=True) - cfg = _config_for("nes", [NES_PREFIX]) - assert scrape_with_artefacts(cfg, "nes", index=index, db=con) == [] - con.close() - - -def test_scrape_returns_empty_when_no_urls(index, db): - cfg = _config_for("nes", []) - assert scrape_with_artefacts(cfg, "nes", index=index, db=db) == [] - - -def test_minerva_source_skips_when_artefacts_missing(monkeypatch, tmp_path): - """Build pipeline must keep going if MiNERVA artefacts aren't mirrored.""" - monkeypatch.delenv("MINERVA_HASHES_DB", raising=False) - monkeypatch.delenv("MINERVA_INDEX_TXT", raising=False) - monkeypatch.chdir(tmp_path) - - src = MinervaSource(_dummy_manifest()) - cfg = _config_for("nes", [NES_PREFIX]) - out = src.scrape("nes", cfg, BuildContext()) - assert out == [] - - -def test_minerva_source_derives_index_when_index_file_is_gone(monkeypatch, tmp_path): - """Upstream removed index.txt.gz; the path index comes from hashes.db.""" - monkeypatch.setenv("MINERVA_HASHES_DB", str(DB_FIXTURE)) - monkeypatch.delenv("MINERVA_INDEX_TXT", raising=False) - monkeypatch.chdir(tmp_path) - - src = MinervaSource(_dummy_manifest()) - cfg = _config_for("nes", [NES_PREFIX]) - out = src.scrape("nes", cfg, BuildContext()) - assert sorted(e["title"] for e in out) == [ - "Legend of Zelda, The (USA)", - "Megaman 2 (USA)", - "Super Mario Bros. (USA)", - ] - - -def test_minerva_source_uses_env_vars(monkeypatch): - monkeypatch.setenv("MINERVA_HASHES_DB", str(DB_FIXTURE)) - monkeypatch.setenv("MINERVA_INDEX_TXT", str(INDEX_FIXTURE)) - - src = MinervaSource(_dummy_manifest()) - cfg = _config_for("nes", [NES_PREFIX]) - out = src.scrape("nes", cfg, BuildContext()) - assert len(out) == 3 - - -def _dummy_manifest(): - from core.contract import SourceManifest - return SourceManifest( - id="minerva", - name="MiNERVA Archive", - kind="catalog", - homepage="https://minerva-archive.org", - priority=200, - platforms=("nes", "snes"), - raw={"id": "minerva", "name": "MiNERVA Archive", "kind": "catalog"}, - ) diff --git a/db/tests/test_no_intro.py b/db/tests/test_no_intro.py deleted file mode 100644 index 91effb68..00000000 --- a/db/tests/test_no_intro.py +++ /dev/null @@ -1,134 +0,0 @@ -"""Tests for parsers/no_intro.py — all pure functions.""" -from __future__ import annotations - -import pytest -from parsers.no_intro import ( - get_clean_title, - move_article, - parse, - parse_regions, - process_entry, - remove_groups_with_contents, -) - - -# -- parse_regions ----------------------------------------------------------- - -@pytest.mark.parametrize('title, expected', [ - ('Game (USA)', ['us']), - ('Game (Europe)', ['eu']), - ('Game (Japan)', ['jp']), - ('Game (USA, Europe)', ['us', 'eu']), - ('Game (Brazil)', ['other']), - ('Game (En,Fr)', []), - ('Game', []), -]) -def test_parse_regions(title: str, expected: list[str]): - assert parse_regions(title) == expected - - -def test_parse_regions_stops_after_first_group(): - regions = parse_regions('Game (USA) (Japan)') - assert regions == ['us'] - - -# -- remove_groups_with_contents --------------------------------------------- - -def test_remove_matching_group(): - assert 'USA' not in remove_groups_with_contents('Title (USA)', ['USA']) - - -def test_remove_preserves_non_matching(): - result = remove_groups_with_contents('Title (Rev 1)', ['USA']) - assert '(Rev 1)' in result - - -def test_remove_comma_separated(): - result = remove_groups_with_contents('Title (En,Fr)', ['En', 'Fr']) - assert '(En,Fr)' not in result - - -def test_remove_partial_match_preserved(): - result = remove_groups_with_contents('Title (En,Rev 1)', ['En']) - assert '(En,Rev 1)' in result - - -# -- move_article ------------------------------------------------------------ - -def test_move_the(): - assert move_article('Legend of Zelda, The') == 'The Legend of Zelda' - - -def test_move_the_with_suffix(): - assert move_article('House of the Dead, The II') == 'The House of the Dead II' - - -def test_move_no_comma(): - assert move_article('Mario') == 'Mario' - - -def test_move_apostrophe_article(): - result = move_article("Jeu d'Arcade, L'") - assert result.startswith("L'") - - -def test_move_parens_in_name(): - result = move_article('Game (USA), The') - assert result == 'Game (USA), The' - - -# -- get_clean_title --------------------------------------------------------- - -def test_clean_removes_region_language(): - result = get_clean_title('Super Mario (USA) (En,Fr)') - assert result == 'Super Mario' - - -def test_clean_preserves_rev(): - result = get_clean_title('Game (Rev 1) (USA)') - assert '(Rev 1)' in result - assert '(USA)' not in result - - -def test_clean_normalizes_spaces(): - result = get_clean_title('Game (USA) Title') - assert ' ' not in result - - -# -- process_entry ----------------------------------------------------------- - -def test_process_entry_all_flags(sample_entry: dict): - sample_entry['title'] = 'Legend of Zelda, The (USA) (En)' - sample_entry['regions'] = [] - process_entry(sample_entry, True, True, True) - assert sample_entry['regions'] == ['us'] - assert sample_entry['title'].startswith('The') - assert '(USA)' not in sample_entry['title'] - - -def test_process_entry_no_flags(sample_entry: dict): - original_title = sample_entry['title'] - original_regions = list(sample_entry['regions']) - process_entry(sample_entry, False, False, False) - assert sample_entry['title'] == original_title - assert sample_entry['regions'] == original_regions - - -def test_process_entry_preserves_existing_regions(sample_entry: dict): - sample_entry['title'] = 'Game (Japan)' - sample_entry['regions'] = ['us'] - process_entry(sample_entry, True, False, False) - assert sample_entry['regions'] == ['us'] - - -# -- parse ------------------------------------------------------------------- - -def test_parse_delegates(): - entries = [ - {'title': 'Game (USA)', 'regions': []}, - {'title': 'Other (Europe)', 'regions': []}, - ] - result = parse(entries, {}) - assert len(result) == 2 - assert result[0]['regions'] == ['us'] - assert result[1]['regions'] == ['eu'] diff --git a/db/tests/test_nopaystation_scraper.py b/db/tests/test_nopaystation_scraper.py deleted file mode 100644 index 078438ea..00000000 --- a/db/tests/test_nopaystation_scraper.py +++ /dev/null @@ -1,206 +0,0 @@ -"""Tests for sources/nopaystation/scraper.py.""" -from __future__ import annotations - -import os -from pathlib import Path -from typing import Any - -import pytest -from sources.nopaystation import scraper as nps -from sources.nopaystation.scraper import ( - add_ps3_links, - add_psv_links, - create_entry, - create_rap_file, - create_zrif_file, - parse_links, - parse_response, -) - -MOCK_TSV = ( - "Title ID\tRegion\tName\tPKG direct link\tContent ID\tRAP\tFile Size\n" - "BLUS12345\tUS\tTest Game\thttps://example.com/BLUS12345.pkg\t" - "UP1234-BLUS12345_00\t0123456789abcdef0123456789abcdef\t1073741824\n" - "BLES67890\tEU\tAnother Game\thttps://example.com/BLES67890.pkg\t" - "EP5678-BLES67890_00\t\t524288000\n" -) - - -def _source() -> dict[str, Any]: - return { - 'filter': None, - 'regions': [], - 'type': 'Game', - 'format': 'pkg', - 'urls': ['https://nopaystation.com/tsv/PS3_GAMES.tsv'], - } - - -def _result( - title_id: str = 'BLUS12345', - region: str = 'US', - name: str = 'Test Game', - url: str = 'https://example.com/test.pkg', - content_id: str = 'UP1234-BLUS12345_00', - rap: str = '', - size: str = '1073741824', -) -> dict[str, str]: - return { - 'Title ID': title_id, - 'Region': region, - 'Name': name, - 'PKG direct link': url, - 'Content ID': content_id, - 'RAP': rap, - 'File Size': size, - } - - -# -- create_entry ------------------------------------------------------------ - -def test_create_entry_basic(): - entry = create_entry(_result(), _source(), 'ps3', 'https://nps.com') - assert entry['rom_id'] == 'BLUS12345' - assert entry['title'] == 'Test Game' - assert entry['platform'] == 'ps3' - assert entry['regions'] == ['us'] - - -def test_create_entry_unknown_region(): - entry = create_entry(_result(region='XX'), _source(), 'ps3', 'https://nps.com') - assert entry['regions'] == ['other'] - - -def test_create_entry_invalid_url_no_links(): - entry = create_entry(_result(url='MISSING'), _source(), 'ps3', 'https://nps.com') - assert entry['links'] == [] - - -# -- parse_links ------------------------------------------------------------- - -def test_parse_links_direct_url(): - links = parse_links(_result(), _source(), 'ps3', 'https://nps.com') - assert len(links) >= 1 - assert links[0]['url'] == 'https://example.com/test.pkg' - - -def test_parse_links_names_file_after_title(): - links = parse_links( - _result(url='https://cdn.example.com/fSndUWIDQSWRWhQnmRat.pkg'), - _source(), 'ps3', 'https://nps.com', - ) - assert links[0]['filename'] == 'Test Game.pkg' - - -def test_parse_links_sanitizes_title_and_defaults_extension(): - links = parse_links( - _result(name='Game: Redux?', url='https://cdn.example.com/tokenwithoutext'), - _source(), 'ps3', 'https://nps.com', - ) - assert links[0]['filename'] == 'Game Redux.pkg' - - -def test_parse_links_zero_size(): - links = parse_links(_result(size=''), _source(), 'ps3', 'https://nps.com') - assert links[0]['size'] == 0 - - -def test_parse_links_missing_url(): - links = parse_links(_result(url=''), _source(), 'ps3', 'https://nps.com') - assert len(links) == 0 - - -# -- parse_response ---------------------------------------------------------- - -def test_parse_response_basic(tmp_path: Path, monkeypatch: pytest.MonkeyPatch): - monkeypatch.setattr(nps, 'PS3_RAPS_DIR', str(tmp_path / 'raps')) - monkeypatch.setattr(nps, 'PSV_ZRIFS_DIR', str(tmp_path / 'zrifs')) - (tmp_path / 'raps').mkdir() - (tmp_path / 'zrifs').mkdir() - entries = parse_response(MOCK_TSV, _source(), 'ps3', 'https://nps.com') - assert len(entries) == 2 - assert entries[0]['rom_id'] == 'BLUS12345' - assert entries[1]['rom_id'] == 'BLES67890' - - -def test_parse_response_empty(): - assert parse_response('', _source(), 'ps3', 'https://nps.com') == [] - - -def test_parse_response_header_only(): - header = "Title ID\tRegion\tName\tPKG direct link\tContent ID\tRAP\tFile Size\n" - assert parse_response(header, _source(), 'ps3', 'https://nps.com') == [] - - -# -- create_rap_file / create_zrif_file -------------------------------------- - -def test_create_rap_file(tmp_path: Path): - filepath = str(tmp_path / 'test.rap') - create_rap_file('0123456789abcdef0123456789abcdef', filepath) - assert os.path.exists(filepath) - with open(filepath, 'rb') as f: - assert len(f.read()) == 16 - - -def test_create_zrif_file(tmp_path: Path): - filepath = str(tmp_path / 'test.zrif') - create_zrif_file('KO5ifR1dQ+eHBw==', filepath) - assert os.path.exists(filepath) - with open(filepath, 'r') as f: - assert f.read() == 'KO5ifR1dQ+eHBw==' - - -# -- add_ps3_links ----------------------------------------------------------- - -def test_add_ps3_links_creates_rap(tmp_path: Path, monkeypatch: pytest.MonkeyPatch): - monkeypatch.setattr(nps, 'PS3_RAPS_DIR', str(tmp_path)) - result = { - 'Name': 'Test Game', - 'RAP': '0123456789abcdef0123456789abcdef', - 'Content ID': 'UP1234-BLUS12345_00', - } - links: list[dict[str, Any]] = [] - add_ps3_links(result, links, 'https://nps.com') - assert len(links) == 1 - assert links[0]['format'] == 'rap' - assert os.path.exists(tmp_path / 'UP1234-BLUS12345_00.rap') - - -def test_add_ps3_links_skips_short_rap(tmp_path: Path, monkeypatch: pytest.MonkeyPatch): - monkeypatch.setattr(nps, 'PS3_RAPS_DIR', str(tmp_path)) - result = {'Name': 'Test', 'RAP': 'tooshort', 'Content ID': 'XX'} - links: list[dict[str, Any]] = [] - add_ps3_links(result, links, 'https://nps.com') - assert len(links) == 0 - - -def test_add_ps3_links_skips_empty_content_id(tmp_path: Path, monkeypatch: pytest.MonkeyPatch): - monkeypatch.setattr(nps, 'PS3_RAPS_DIR', str(tmp_path)) - result = {'Name': 'Test', 'RAP': '0123456789abcdef0123456789abcdef', 'Content ID': ''} - links: list[dict[str, Any]] = [] - add_ps3_links(result, links, 'https://nps.com') - assert len(links) == 0 - - -# -- add_psv_links ----------------------------------------------------------- - -def test_add_psv_links_creates_zrif(tmp_path: Path, monkeypatch: pytest.MonkeyPatch): - monkeypatch.setattr(nps, 'PSV_ZRIFS_DIR', str(tmp_path)) - result = { - 'Name': 'Vita Game', - 'zRIF': 'KO5ifR1dQ+eHBw==', - 'Content ID': 'PCSE12345', - } - links: list[dict[str, Any]] = [] - add_psv_links(result, links, 'https://nps.com') - assert len(links) == 1 - assert links[0]['format'] == 'string' - assert os.path.exists(tmp_path / 'PCSE12345') - - -def test_add_psv_links_skips_empty_zrif(tmp_path: Path, monkeypatch: pytest.MonkeyPatch): - monkeypatch.setattr(nps, 'PSV_ZRIFS_DIR', str(tmp_path)) - result = {'Name': 'Test', 'zRIF': '', 'Content ID': 'PCSE12345'} - links: list[dict[str, Any]] = [] - add_psv_links(result, links, 'https://nps.com') - assert len(links) == 0 diff --git a/db/tests/test_parse_utils.py b/db/tests/test_parse_utils.py deleted file mode 100644 index 17046f8f..00000000 --- a/db/tests/test_parse_utils.py +++ /dev/null @@ -1,180 +0,0 @@ -"""Tests for utils/parse_utils.py — all pure functions, no I/O.""" -from __future__ import annotations - -import pytest -from utils.parse_utils import ( - create_search_key, - create_slug, - join_urls, - normalize_repeated_chars, - remove_ext, - replace_invalid_chars, - size_bytes_to_str, - size_str_to_bytes, -) - - -# -- replace_invalid_chars --------------------------------------------------- - -def test_replace_plus(): - assert 'plus' in replace_invalid_chars('C++') - - -def test_replace_ampersand(): - assert 'and' in replace_invalid_chars('Rock & Roll') - - -def test_replace_trademark_symbols(): - result = replace_invalid_chars('Game™ Title© 2024®') - assert '™' not in result - assert '©' not in result - assert '®' not in result - - -def test_replace_no_op(): - assert replace_invalid_chars('Normal Title') == 'Normal Title' - - -# -- remove_ext -------------------------------------------------------------- - -def test_remove_simple_ext(): - assert remove_ext('game.zip') == 'game' - - -def test_remove_ext_with_path(): - assert remove_ext('some/dir/game.zip') == 'game' - - -def test_remove_ext_dotted_name(): - assert remove_ext('game.v2.zip') == 'game.v2' - - -def test_remove_ext_no_ext(): - assert remove_ext('noext') == 'noext' - - -# -- normalize_repeated_chars ------------------------------------------------ - -def test_collapse_spaces(): - assert normalize_repeated_chars('a b', ' ') == 'a b' - - -def test_collapse_dashes(): - assert normalize_repeated_chars('a---b', '-') == 'a-b' - - -def test_no_repeats(): - assert normalize_repeated_chars('a-b', '-') == 'a-b' - - -def test_collapse_regex_special_char(): - assert normalize_repeated_chars('a...b', '.') == 'a.b' - - -# -- create_slug ------------------------------------------------------------- - -def test_slug_basic(): - slug = create_slug({'title': 'Super Mario', 'platform': 'nes', 'regions': ['us']}) - assert slug == 'super-mario-nes-us' - - -def test_slug_multiple_regions(): - slug = create_slug({'title': 'Game', 'platform': 'snes', 'regions': ['us', 'eu']}) - assert slug.endswith('us-eu') - - -def test_slug_special_chars(): - slug = create_slug({'title': 'Rock & Roll!', 'platform': 'nes', 'regions': ['us']}) - assert '&' not in slug - assert '!' not in slug - - -def test_slug_unicode(): - slug = create_slug({'title': 'Pokémon', 'platform': 'gb', 'regions': ['jp']}) - assert 'pokemon' in slug - - -def test_slug_no_trailing_dashes(): - slug = create_slug({'title': ' Game ', 'platform': 'nes', 'regions': ['us']}) - assert not slug.startswith('-') - assert not slug.endswith('-') - - -def test_slug_repeated_dashes_collapsed(): - slug = create_slug({'title': 'A B', 'platform': 'nes', 'regions': ['us']}) - assert '--' not in slug - - -# -- create_search_key ------------------------------------------------------- - -def test_search_key_basic(): - assert create_search_key('Super Mario Bros') == 'supermariobros' - - -def test_search_key_strips_specials(): - key = create_search_key('Rock & Roll!') - assert key == 'rockandroll' - - -def test_search_key_unicode(): - key = create_search_key('Pokémon') - assert key == 'pokemon' - - -def test_search_key_empty(): - assert create_search_key('') == '' - - -def test_search_key_numbers(): - assert create_search_key('Game 2') == 'game2' - - -# -- size_bytes_to_str ------------------------------------------------------- - -@pytest.mark.parametrize('size, expected', [ - (0, '0B'), - (512, '512B'), - (1024, '1K'), - (1536, '1.5K'), - (1048576, '1M'), - (1073741824, '1G'), - (1099511627776, '1T'), -]) -def test_size_bytes_to_str(size: int, expected: str): - assert size_bytes_to_str(size) == expected - - -# -- size_str_to_bytes ------------------------------------------------------- - -@pytest.mark.parametrize('size_str, expected', [ - ('512B', 512), - ('1K', 1024), - ('1.5K', 1536), - ('1M', 1048576), - ('1G', 1073741824), - ('', 0), - (' ', 0), - ('nodigits', 0), -]) -def test_size_str_to_bytes(size_str: str, expected: int): - assert size_str_to_bytes(size_str) == expected - - -# -- join_urls --------------------------------------------------------------- - -def test_join_simple(): - assert join_urls('https://example.com/dir', 'file.zip') == 'https://example.com/dir/file.zip' - - -def test_join_trailing_slash(): - assert join_urls('https://example.com/dir/', 'file.zip') == 'https://example.com/dir/file.zip' - - -def test_join_multi_segment(): - result = join_urls('https://a.com', 'b', 'c') - assert result == 'https://a.com/b/c' - - -def test_join_encoded_chars(): - result = join_urls('https://a.com/dir', 'file%20name.zip') - assert 'file%20name.zip' in result diff --git a/db/tests/test_parsers.py b/db/tests/test_parsers.py deleted file mode 100644 index baddb487..00000000 --- a/db/tests/test_parsers.py +++ /dev/null @@ -1,258 +0,0 @@ -"""Tests for parsers/libretro.py and parsers/gametdb.py with monkeypatched globals.""" -from __future__ import annotations - -from pathlib import Path -from unittest.mock import MagicMock, patch - -import pytest -from parsers import gametdb, libretro - - -# ============================================================================= -# gametdb tests -# ============================================================================= - -MOCK_TDBS = { - 'wiitdb.xml': [ - {'name': 'Test Wii Game', 'id': 'AAAE01', 'type': 'WiiWare', 'region': 'NTSC-U'}, - {'name': 'JP Wii Game', 'id': 'AAAJ01', 'type': 'WiiWare', 'region': 'NTSC-J'}, - ], - 'dstdb.xml': [ - {'name': 'DS Game', 'id': 'BBBB', 'type': 'DS', 'region': 'NTSC-U'}, - ], - '3dstdb.xml': [], - 'wiiutdb.xml': [], - 'ps3tdb.xml': [ - {'name': 'PS3 Game', 'id': 'BCUS12345', 'type': 'PS3', 'region': 'NTSC-U'}, - ], -} - - -@pytest.fixture(autouse=True) -def _reset_gametdb(): - original = gametdb.tdbs - yield - gametdb.tdbs = original - - -# -- build_boxart_url -------------------------------------------------------- - -@pytest.mark.parametrize('platform, country, game_id, expected_fragment', [ - ('wii', 'US', 'AAAE01', 'wii/cover/US/AAAE01.png'), - ('3ds', 'JA', 'BBBB', '3ds/coverM/JA/BBBB.jpg'), - ('nds', 'EN', 'CCCC', 'ds/coverS/EN/CCCC.png'), - ('ps3', 'US', 'BCUS12345', 'ps3/cover/US/BCUS12345.jpg'), -]) -def test_build_boxart_url(platform: str, country: str, game_id: str, expected_fragment: str): - url = gametdb.build_boxart_url(platform, country, game_id) - assert expected_fragment in url - assert url.startswith('https://art.gametdb.com/') - - -# -- find_full_id ------------------------------------------------------------ - -def test_find_full_id_match(): - gametdb.tdbs = MOCK_TDBS - assert gametdb.find_full_id('AAAE', 'wii') == 'AAAE01' - - -def test_find_full_id_no_match(): - gametdb.tdbs = MOCK_TDBS - assert gametdb.find_full_id('ZZZZ', 'wii') is None - - -def test_find_full_id_unknown_platform(): - gametdb.tdbs = MOCK_TDBS - assert gametdb.find_full_id('AAAE', 'unknownplat') is None - - -def test_find_full_id_tdbs_none(): - gametdb.tdbs = None - assert gametdb.find_full_id('AAAE', 'wii') is None - - -# -- gametdb parse ----------------------------------------------------------- - -def test_gametdb_parse_enriches_by_rom_id(): - gametdb.tdbs = MOCK_TDBS - entries = [{ - 'title': 'Test Wii Game', - 'platform': 'wii', - 'regions': ['us'], - 'rom_id': 'AAAE01', - }] - result = gametdb.parse(entries, {'parse_boxart': True, 'parse_name': False}) - assert 'boxart_url' in result[0] - - -def test_gametdb_parse_skips_unknown_platform(): - gametdb.tdbs = MOCK_TDBS - entries = [{'title': 'Game', 'platform': 'unknownplat', 'regions': ['us']}] - result = gametdb.parse(entries, {}) - assert 'boxart_url' not in result[0] - - -def test_gametdb_parse_empty(): - gametdb.tdbs = MOCK_TDBS - assert gametdb.parse([], {}) == [] - - -def test_gametdb_get_boxart_url_by_id_valid(): - gametdb.tdbs = MOCK_TDBS - url = gametdb.get_boxart_url_by_id('AAAE01', 'wii') - assert url is not None - assert 'AAAE01' in url - assert 'art.gametdb.com' in url - - -def test_gametdb_get_boxart_url_by_id_unknown_platform(): - gametdb.tdbs = MOCK_TDBS - assert gametdb.get_boxart_url_by_id('AAAE01', 'unknownplat') is None - - -def test_gametdb_get_boxart_url_by_id_no_match(): - gametdb.tdbs = MOCK_TDBS - assert gametdb.get_boxart_url_by_id('ZZZZZZ', 'wii') is None - - -def test_gametdb_load_tdbs(tmp_path: Path, monkeypatch: pytest.MonkeyPatch): - xml_content = """ - - - AAAE01 - WiiWare - NTSC-U - - - BBBB - -""" - data_dir = tmp_path / 'data' / 'gametdb' - data_dir.mkdir(parents=True) - for xml_name in gametdb.XML_FILENAMES: - (data_dir / xml_name).write_text('') - (data_dir / 'wiitdb.xml').write_text(xml_content) - monkeypatch.chdir(tmp_path) - gametdb.tdbs = None - gametdb.load_tdbs() - assert gametdb.tdbs is not None - assert len(gametdb.tdbs['wiitdb.xml']) == 1 - assert gametdb.tdbs['wiitdb.xml'][0]['name'] == 'Test Game' - - -def test_gametdb_load_tdbs_missing_file(tmp_path: Path, monkeypatch: pytest.MonkeyPatch): - monkeypatch.chdir(tmp_path) - gametdb.tdbs = None - gametdb.load_tdbs() - assert gametdb.tdbs is not None - for xml_name in gametdb.XML_FILENAMES: - assert gametdb.tdbs[xml_name] == [] - - -def test_gametdb_parse_title_search(): - gametdb.tdbs = MOCK_TDBS - entries = [{ - 'title': 'Test Wii Game', - 'platform': 'wii', - 'regions': ['us'], - }] - result = gametdb.parse(entries, {'parse_boxart': True, 'parse_name': False}) - assert 'boxart_url' in result[0] - - -def test_gametdb_parse_name_update(): - gametdb.tdbs = MOCK_TDBS - entries = [{ - 'title': 'Test Wii Game', - 'platform': 'wii', - 'regions': ['us'], - 'rom_id': 'AAAE01', - }] - result = gametdb.parse(entries, {'parse_boxart': False, 'parse_name': True}) - assert result[0]['title'] == 'Test Wii Game' - - -# ============================================================================= -# libretro tests -# ============================================================================= - -MOCK_DBS: dict[str, dict[str, str]] = { - 'nes': {'Super Mario Bros. (USA)': 'NES-SM-USA'}, - 'snes': {'Super Metroid (USA)': 'SNES-SM-USA'}, -} - - -@pytest.fixture(autouse=True) -def _reset_libretro(): - original = libretro.dbs - yield - libretro.dbs = original - - -def test_libretro_parse_sets_rom_id(): - libretro.dbs = MOCK_DBS - entries = [{'title': 'Super Mario Bros. (USA)', 'platform': 'nes'}] - with patch('requests.get') as mock_get: - mock_get.return_value = MagicMock(text='') - result = libretro.parse(entries, {}) - assert result[0]['rom_id'] == 'NES-SM-USA' - - -def test_libretro_parse_no_match(): - libretro.dbs = MOCK_DBS - entries = [{'title': 'Unknown Game', 'platform': 'nes'}] - with patch('requests.get') as mock_get: - mock_get.return_value = MagicMock(text='') - result = libretro.parse(entries, {}) - assert result[0]['rom_id'] is None - - -def test_libretro_parse_unknown_platform(): - libretro.dbs = MOCK_DBS - entries = [{'title': 'Game', 'platform': 'unknownplat'}] - result = libretro.parse(entries, {}) - assert result[0]['title'] == 'Game' - - -def test_libretro_parse_empty(): - libretro.dbs = MOCK_DBS - assert libretro.parse([], {}) == [] - - -def test_libretro_parse_boxart_found(): - libretro.dbs = MOCK_DBS - # Clear cached boxarts so the mock HTTP call is used - if 'available_boxarts' in libretro.PLATFORMS.get('nes', {}): - del libretro.PLATFORMS['nes']['available_boxarts'] - entries = [{'title': 'Super Mario Bros. (USA)', 'platform': 'nes'}] - mock_html = '[IMG]Super Mario Bros. (USA).png' - with patch('requests.get') as mock_get: - mock_get.return_value = MagicMock(text=mock_html) - result = libretro.parse(entries, {}) - assert 'boxart_url' in result[0] - - -def test_libretro_load_dbs_from_dat(tmp_path: Path, monkeypatch: pytest.MonkeyPatch): - dat_content = 'game (\n\tname "Super Mario Bros. (USA)"\n\tserial "NES-SM-USA"\n\trom ( name "smb.nes" )\n)\n' - # Create ALL DAT files that load_dbs expects (empty for most, real content for NES) - for platform_info in libretro.PLATFORMS.values(): - for dat_path in platform_info['dats']: - full_path = tmp_path / 'data' / 'libretro' / dat_path - full_path.parent.mkdir(parents=True, exist_ok=True) - full_path.write_text('') - # Write real content for one NES DAT - nes_dat = tmp_path / 'data' / 'libretro' / libretro.PLATFORMS['nes']['dats'][0] - nes_dat.write_text(dat_content) - monkeypatch.chdir(tmp_path) - libretro.dbs = None - libretro.load_dbs() - assert libretro.dbs is not None - assert libretro.dbs['nes']['Super Mario Bros. (USA)'] == 'NES-SM-USA' - - -def test_libretro_parse_request_exception(): - libretro.dbs = MOCK_DBS - entries = [{'title': 'Super Mario Bros. (USA)', 'platform': 'nes'}] - with patch('requests.get', side_effect=Exception('network error')): - result = libretro.parse(entries, {}) - assert result[0]['title'] == 'Super Mario Bros. (USA)' diff --git a/db/tests/test_registry.py b/db/tests/test_registry.py deleted file mode 100644 index ecac02f3..00000000 --- a/db/tests/test_registry.py +++ /dev/null @@ -1,97 +0,0 @@ -""" -Source plugin contract guarantees: - -- Every plugin folder under db/sources/ is discoverable. -- Each plugin's manifest matches its folder name. -- Each plugin satisfies the Source protocol. -- platforms.yml only references known source ids and has no empty lists. - -Run from the db/ directory: python -m pytest tests -""" -from __future__ import annotations - -import sys -from pathlib import Path - -import pytest -import yaml - -# Make sure db/ is on sys.path no matter how the test is invoked. -DB_ROOT = Path(__file__).resolve().parent.parent -if str(DB_ROOT) not in sys.path: - sys.path.insert(0, str(DB_ROOT)) - -from core import load_registry # noqa: E402 -from core.contract import Source # noqa: E402 - - -EXPECTED_SOURCES = {"internet_archive", "mariocube", "minerva", "nopaystation"} -VALID_KINDS = {"catalog", "host", "hybrid"} - - -@pytest.fixture(scope="module") -def registry(): - return load_registry(DB_ROOT) - - -def test_all_expected_plugins_discovered(registry): - assert set(registry.ids()) == EXPECTED_SOURCES, ( - f"Expected exactly {sorted(EXPECTED_SOURCES)}; " - f"discovered {registry.ids()}" - ) - - -def test_each_plugin_satisfies_source_protocol(registry): - for source_id, source in registry.sources.items(): - assert isinstance(source, Source), ( - f"Plugin '{source_id}' does not satisfy the Source protocol" - ) - assert callable(source.scrape), ( - f"Plugin '{source_id}'.scrape is not callable" - ) - - -def test_manifest_id_matches_folder(registry): - for source_id, manifest in registry.manifests.items(): - assert manifest.id == source_id - - -def test_manifest_kinds_are_valid(registry): - for source_id, manifest in registry.manifests.items(): - assert manifest.kind in VALID_KINDS, ( - f"Plugin '{source_id}' has invalid kind: {manifest.kind!r}" - ) - - -def test_platforms_yml_only_references_known_sources(registry): - with (DB_ROOT / "platforms.yml").open("r", encoding="utf-8") as f: - platforms = yaml.safe_load(f) or {} - - referenced = set() - for entries in platforms.values(): - for entry in entries: - referenced.add(entry["source"]) - - unknown = referenced - set(registry.ids()) - assert not unknown, f"platforms.yml references unknown sources: {sorted(unknown)}" - - -def test_platforms_yml_has_no_empty_lists(): - with (DB_ROOT / "platforms.yml").open("r", encoding="utf-8") as f: - platforms = yaml.safe_load(f) or {} - - empty = [p for p, entries in platforms.items() if not entries] - assert not empty, f"platforms.yml has empty entry lists for: {empty}" - - -def test_each_platforms_yml_entry_has_required_keys(): - with (DB_ROOT / "platforms.yml").open("r", encoding="utf-8") as f: - platforms = yaml.safe_load(f) or {} - - required = {"source", "format", "urls", "type", "parsers"} - for platform, entries in platforms.items(): - for i, entry in enumerate(entries): - missing = required - entry.keys() - assert not missing, ( - f"platforms.yml[{platform}][{i}] missing keys: {sorted(missing)}" - ) diff --git a/db/tests/test_retroachievements.py b/db/tests/test_retroachievements.py deleted file mode 100644 index 288931b5..00000000 --- a/db/tests/test_retroachievements.py +++ /dev/null @@ -1,187 +0,0 @@ -"""Tests for the RetroAchievements parser. - -Covers: building the normalized title index from per-console JSON, dropping -sets with no achievements, exact + normalized title matching, collision -handling, the min_achievements flag, and graceful no-op when data is absent. -""" -from __future__ import annotations - -import json -import sys -from pathlib import Path - -import pytest - -DB_ROOT = Path(__file__).resolve().parent.parent -if str(DB_ROOT) not in sys.path: - sys.path.insert(0, str(DB_ROOT)) - -from parsers import retroachievements as ra # noqa: E402 - - -@pytest.fixture(autouse=True) -def reset_dbs(): - """The parser caches its index in a module global; isolate each test.""" - ra.dbs = None - yield - ra.dbs = None - - -@pytest.fixture() -def ra_data_dir(tmp_path, monkeypatch): - d = tmp_path / 'retroachievements' - d.mkdir() - monkeypatch.setattr(ra, 'DATA_DIR', str(d)) - return d - - -def write_console(data_dir: Path, console_id: int, games: list) -> None: - (data_dir / f'{console_id}.json').write_text( - json.dumps(games), encoding='utf-8') - - -# console ids used in tests (from ra.RA_CONSOLES): nes=7, smd=1 -NES = ra.RA_CONSOLES['nes'] -SMD = ra.RA_CONSOLES['smd'] - - -def test_load_dbs_builds_index_and_drops_zero_achievements(ra_data_dir): - write_console(ra_data_dir, NES, [ - {'Title': 'Super Mario Bros.', 'ID': 111, 'NumAchievements': 30}, - {'Title': 'Has No Achievements', 'ID': 222, 'NumAchievements': 0}, - ]) - - ra.load_dbs() - - assert ra.dbs is not None - index = ra.dbs['nes'] - assert index[ra.ra_normalize('Super Mario Bros.')] == (111, 30) - # The zero-achievement game is excluded entirely. - assert ra.ra_normalize('Has No Achievements') not in index - - -def test_parse_exact_match_sets_fields(ra_data_dir): - write_console(ra_data_dir, NES, [ - {'Title': 'Super Mario Bros.', 'ID': 111, 'NumAchievements': 30}, - ]) - - entries = [{'title': 'Super Mario Bros.', 'platform': 'nes'}] - out = ra.parse(entries, {}) - - assert out[0]['ra_game_id'] == 111 - assert out[0]['ra_num_achievements'] == 30 - - -def test_parse_normalized_match(ra_data_dir): - # RA title and catalog title differ in punctuation and '&' vs 'and'; - # create_search_key collapses both to the same key. - write_console(ra_data_dir, NES, [ - {'Title': 'Pokémon: Red & Blue', 'ID': 7, 'NumAchievements': 12}, - ]) - - entries = [{'title': 'Pokemon - Red and Blue', 'platform': 'nes'}] - out = ra.parse(entries, {}) - - assert out[0]['ra_game_id'] == 7 - assert out[0]['ra_num_achievements'] == 12 - - -def test_parse_collision_keeps_richer_set(ra_data_dir): - # Both titles normalize to 'sonic'; the higher-achievement set wins. - write_console(ra_data_dir, SMD, [ - {'Title': 'Sonic', 'ID': 1, 'NumAchievements': 5}, - {'Title': 'Sonic!', 'ID': 2, 'NumAchievements': 20}, - ]) - - entries = [{'title': 'Sonic', 'platform': 'smd'}] - out = ra.parse(entries, {}) - - assert out[0]['ra_game_id'] == 2 - assert out[0]['ra_num_achievements'] == 20 - - -def test_parse_unsupported_platform_is_untouched(ra_data_dir): - write_console(ra_data_dir, NES, [ - {'Title': 'Super Mario Bros.', 'ID': 111, 'NumAchievements': 30}, - ]) - - # 'wii' is not in RA_CONSOLES. - entries = [{'title': 'Wii Sports', 'platform': 'wii'}] - out = ra.parse(entries, {}) - - assert 'ra_game_id' not in out[0] - assert 'ra_num_achievements' not in out[0] - - -def test_parse_no_match_leaves_keys_absent(ra_data_dir): - write_console(ra_data_dir, NES, [ - {'Title': 'Super Mario Bros.', 'ID': 111, 'NumAchievements': 30}, - ]) - - entries = [{'title': 'Some Unknown Game', 'platform': 'nes'}] - out = ra.parse(entries, {}) - - assert 'ra_game_id' not in out[0] - - -def test_parse_respects_min_achievements_flag(ra_data_dir): - write_console(ra_data_dir, NES, [ - {'Title': 'Tiny Set', 'ID': 9, 'NumAchievements': 3}, - ]) - - entries = [{'title': 'Tiny Set', 'platform': 'nes'}] - out = ra.parse(entries, {'min_achievements': 10}) - - assert 'ra_game_id' not in out[0] - - -def test_parse_no_data_dir_is_noop(ra_data_dir): - # Empty data dir: load_dbs finds nothing, parse returns entries unchanged. - entries = [{'title': 'Super Mario Bros.', 'platform': 'nes'}] - out = ra.parse(entries, {}) - - assert out == [{'title': 'Super Mario Bros.', 'platform': 'nes'}] - - -def test_parse_reports_match_counts(ra_data_dir, capsys): - write_console(ra_data_dir, NES, [ - {'Title': 'Super Mario Bros.', 'ID': 111, 'NumAchievements': 30}, - ]) - - entries = [ - {'title': 'Super Mario Bros.', 'platform': 'nes'}, # matches - {'title': 'Unknown Game', 'platform': 'nes'}, # no match - ] - ra.parse(entries, {}) - - assert 'RetroAchievements: matched 1/2 nes entries' in capsys.readouterr().out - - -def test_parse_supported_platform_without_data_reports_zero(ra_data_dir, capsys): - # smd has data loaded; nes is supported but has no data file, so its - # entries are counted as unmatched rather than skipped silently. - write_console(ra_data_dir, SMD, [ - {'Title': 'Sonic', 'ID': 1, 'NumAchievements': 20}, - ]) - - entries = [ - {'title': 'Super Mario Bros.', 'platform': 'nes'}, - {'title': 'Metroid', 'platform': 'nes'}, - ] - out = ra.parse(entries, {}) - - assert 'ra_game_id' not in out[0] - assert 'RetroAchievements: matched 0/2 nes entries' in capsys.readouterr().out - - -def test_load_dbs_skips_malformed_file(ra_data_dir): - (ra_data_dir / f'{NES}.json').write_text('{not json', encoding='utf-8') - write_console(ra_data_dir, SMD, [ - {'Title': 'Sonic', 'ID': 1, 'NumAchievements': 20}, - ]) - - ra.load_dbs() - - assert ra.dbs is not None - assert 'nes' not in ra.dbs # malformed file skipped - assert ra.dbs['smd'][ra.ra_normalize('Sonic')] == (1, 20) diff --git a/db/tests/test_schema.py b/db/tests/test_schema.py deleted file mode 100644 index 93fb72ec..00000000 --- a/db/tests/test_schema.py +++ /dev/null @@ -1,285 +0,0 @@ -""" -Catalog schema guarantees: - -- All expected tables and link columns exist. -- register_source persists every manifest field. -- record_source_health upserts. -- insert_entry writes link fields with sane defaults. -- _tag_links defaults link.source_id to the manifest id and honours - per-link overrides (e.g. IA marking a link as requires_auth). -- user_sources stays untouched by the build pipeline. - -Run: python -m pytest tests -""" -from __future__ import annotations - -import json -import sqlite3 -import sys -from pathlib import Path - -import pytest - -DB_ROOT = Path(__file__).resolve().parent.parent -if str(DB_ROOT) not in sys.path: - sys.path.insert(0, str(DB_ROOT)) - -from core import load_registry # noqa: E402 -from database import db_manager # noqa: E402 -from make import _tag_links # noqa: E402 - - -@pytest.fixture() -def fresh_db(tmp_path, monkeypatch): - """Run db_manager against a tmp working dir so we don't clobber db/.""" - monkeypatch.chdir(tmp_path) - db_manager.con = None - db_manager.cur = None - db_manager.init_database() - yield tmp_path - db_manager.close_database() - - -def _table_columns(db_path: Path, table: str) -> list[str]: - con = sqlite3.connect(db_path) - try: - return [r[1] for r in con.execute(f"PRAGMA table_info({table})").fetchall()] - finally: - con.close() - - -def test_schema_version_is_4(): - assert db_manager.SCHEMA_VERSION == 4 - - -def test_init_creates_v2_tables(fresh_db): - expected = { - "platforms", "entries", "entries_fts", "regions", - "regions_entries", "links", - "sources", "source_health", "user_sources", "torrents", - "entry_groups", "entry_group_members", - } - cur = db_manager.cur - assert cur is not None - rows = cur.execute("SELECT name FROM sqlite_master WHERE type='table'").fetchall() - actual = {r[0] for r in rows} - missing = expected - actual - assert not missing, f"Missing tables: {sorted(missing)}" - - -def test_entries_table_has_ra_columns(fresh_db): - cur = db_manager.cur - assert cur is not None - cols = [r[1] for r in cur.execute("PRAGMA table_info(entries)").fetchall()] - for required in ("ra_game_id", "ra_num_achievements"): - assert required in cols, f"entries is missing column {required}" - - -def test_insert_entry_round_trips_ra_fields(fresh_db): - registry = load_registry(DB_ROOT) - for manifest in registry.manifests.values(): - db_manager.register_source(manifest) - - db_manager.insert_entry({ - "title": "Super Mario Bros.", - "platform": "nes", - "regions": ["us"], - "links": [], - "ra_game_id": 111, - "ra_num_achievements": 30, - }) - - cur = db_manager.cur - assert cur is not None - row = cur.execute( - "SELECT ra_game_id, ra_num_achievements FROM entries" - ).fetchone() - assert row == (111, 30) - - -def test_ra_fields_not_clobbered_on_reinsert(fresh_db): - registry = load_registry(DB_ROOT) - for manifest in registry.manifests.values(): - db_manager.register_source(manifest) - - base = {"title": "Super Mario Bros.", "platform": "nes", - "regions": ["us"], "links": []} - - # First insert sets RA fields; a later insert of the same slug without - # them must not COALESCE them back to NULL. - db_manager.insert_entry({**base, "ra_game_id": 111, "ra_num_achievements": 30}) - db_manager.insert_entry(dict(base)) - - cur = db_manager.cur - assert cur is not None - row = cur.execute( - "SELECT ra_game_id, ra_num_achievements FROM entries" - ).fetchone() - assert row == (111, 30) - - -def test_links_table_has_v2_columns(fresh_db): - cur = db_manager.cur - assert cur is not None - cols = [r[1] for r in cur.execute("PRAGMA table_info(links)").fetchall()] - for required in ( - "source_id", "requires_auth", - "torrent_infohash", "torrent_file_index", "torrent_file_path", - ): - assert required in cols, f"links is missing column {required}" - - -def test_register_source_persists_manifest(fresh_db): - registry = load_registry(DB_ROOT) - for manifest in registry.manifests.values(): - db_manager.register_source(manifest) - - cur = db_manager.cur - assert cur is not None - rows = cur.execute( - "SELECT id, name, kind, auth_required, priority, manifest_json " - "FROM sources ORDER BY id" - ).fetchall() - - assert {r[0] for r in rows} == set(registry.ids()) - for source_id, name, kind, auth_required, priority, manifest_json in rows: - m = registry.manifests[source_id] - assert name == m.name - assert kind == m.kind - assert auth_required == int(bool(m.auth_required)) - assert priority == int(m.priority) - # manifest_json should round-trip the raw dict - parsed = json.loads(manifest_json) - assert parsed["id"] == m.id - assert parsed["name"] == m.name - - -def test_record_source_health_upserts(fresh_db): - registry = load_registry(DB_ROOT) - for manifest in registry.manifests.values(): - db_manager.register_source(manifest) - - db_manager.record_source_health( - "minerva", "ok", entry_count=5, link_count=10, last_checked=1000 - ) - cur = db_manager.cur - assert cur is not None - row = cur.execute( - "SELECT status, entry_count, link_count, last_checked " - "FROM source_health WHERE source_id='minerva'" - ).fetchone() - assert row == ("ok", 5, 10, 1000) - - # upsert: same source_id replaces - db_manager.record_source_health( - "minerva", "down", reason="404", last_checked=2000, - ) - row = cur.execute( - "SELECT status, reason, last_checked FROM source_health " - "WHERE source_id='minerva'" - ).fetchone() - assert row == ("down", "404", 2000) - - -def test_insert_entry_writes_v2_link_columns(fresh_db): - registry = load_registry(DB_ROOT) - for manifest in registry.manifests.values(): - db_manager.register_source(manifest) - - entry = { - "title": "Sample", - "platform": "nes", - "regions": ["us"], - "links": [ - { - "name": "Sample", - "type": "Game", - "format": "nes", - "url": "magnet:?xt=urn:btih:" + ("a" * 40), - "filename": "x.zip", - "host": "MiNERVA Archive", - "size": 100, - "size_str": "100", - "source_url": "https://minerva-archive.org/", - "source_id": "minerva", - "requires_auth": 0, - }, - { - "name": "Sample (restricted)", - "type": "Game", - "format": "nes", - "url": "https://archive.org/x.zip", - "filename": "x.zip", - "host": "Internet Archive", - "size": 100, - "size_str": "100", - "source_url": "https://archive.org/", - "source_id": "internet_archive", - "requires_auth": 1, - }, - ], - } - db_manager.insert_entry(entry) - - cur = db_manager.cur - assert cur is not None - rows = cur.execute( - "SELECT source_id, requires_auth, torrent_infohash " - "FROM links ORDER BY source_id" - ).fetchall() - assert rows == [ - ("internet_archive", 1, None), - ("minerva", 0, None), - ] - - -def test_tag_links_defaults_to_manifest_id(): - registry = load_registry(DB_ROOT) - source = registry.get("minerva") - assert source is not None - - entries = [ - {"links": [{"name": "a"}, {"name": "b"}]}, - {"links": [{"name": "c"}]}, - ] - n_entries, n_links = _tag_links(entries, source) - assert (n_entries, n_links) == (2, 3) - for entry in entries: - for link in entry["links"]: - assert link["source_id"] == "minerva" - assert link["requires_auth"] == 0 - - -def test_tag_links_respects_per_link_override(): - registry = load_registry(DB_ROOT) - source = registry.get("internet_archive") - assert source is not None - - entries = [{ - "links": [ - {"name": "public"}, - {"name": "restricted", "requires_auth": 1}, - ], - }] - _tag_links(entries, source) - assert entries[0]["links"][0]["requires_auth"] == 0 - assert entries[0]["links"][1]["requires_auth"] == 1 - for link in entries[0]["links"]: - assert link["source_id"] == "internet_archive" - - -def test_user_sources_table_is_empty_after_build(fresh_db): - registry = load_registry(DB_ROOT) - for manifest in registry.manifests.values(): - db_manager.register_source(manifest) - cur = db_manager.cur - assert cur is not None - count = cur.execute("SELECT COUNT(*) FROM user_sources").fetchone()[0] - assert count == 0, "user_sources is reserved; build pipeline must not write to it" - - -def test_torrents_table_starts_empty(fresh_db): - cur = db_manager.cur - assert cur is not None - count = cur.execute("SELECT COUNT(*) FROM torrents").fetchone()[0] - assert count == 0 diff --git a/db/tests/test_scrape_utils.py b/db/tests/test_scrape_utils.py deleted file mode 100644 index a0951a3e..00000000 --- a/db/tests/test_scrape_utils.py +++ /dev/null @@ -1,103 +0,0 @@ -"""Tests for utils/scrape_utils.py.""" -from __future__ import annotations - -from unittest.mock import MagicMock, patch - -import utils.scrape_utils as scrape_utils -from utils.scrape_utils import ( - _needs_playwright, - close_browser, - create_scraper_session, - fetch_url, - BROWSER_HEADERS, - PLAYWRIGHT_REQUIRED_HOSTS, -) - - -# -- _needs_playwright ------------------------------------------------------- - -def test_needs_playwright_matching(): - for host in PLAYWRIGHT_REQUIRED_HOSTS: - assert _needs_playwright(f'https://{host}/page') is True - - -def test_needs_playwright_no_match(): - assert _needs_playwright('https://example.com/page') is False - - -# -- create_scraper_session -------------------------------------------------- - -def test_create_session_default_headers(): - session = create_scraper_session() - for key in BROWSER_HEADERS: - assert key in session.headers - - -def test_create_session_custom_headers(): - session = create_scraper_session({'Custom-Header': 'value'}) - assert session.headers.get('Custom-Header') == 'value' - - -# -- fetch_url --------------------------------------------------------------- - -def test_fetch_url_success(tmp_cache_dir): - with patch('utils.scrape_utils.create_scraper_session') as mock_session_factory: - mock_session = MagicMock() - mock_response = MagicMock() - mock_response.ok = True - mock_response.text = 'content' - mock_session.get.return_value = mock_response - mock_session_factory.return_value = mock_session - - result = fetch_url('https://example.com/page') - assert result == 'content' - - -def test_fetch_url_http_error(tmp_cache_dir): - with patch('utils.scrape_utils.create_scraper_session') as mock_session_factory: - mock_session = MagicMock() - mock_response = MagicMock() - mock_response.ok = False - mock_response.status_code = 403 - mock_session.get.return_value = mock_response - mock_session_factory.return_value = mock_session - - result = fetch_url('https://example.com/blocked') - assert result is None - - -def test_fetch_url_exception(tmp_cache_dir): - with patch('utils.scrape_utils.create_scraper_session') as mock_session_factory: - mock_session = MagicMock() - mock_session.get.side_effect = Exception('timeout') - mock_session_factory.return_value = mock_session - - result = fetch_url('https://example.com/timeout') - assert result is None - - -def test_fetch_url_with_existing_session(tmp_cache_dir): - mock_session = MagicMock() - mock_response = MagicMock() - mock_response.ok = True - mock_response.text = 'data' - mock_session.get.return_value = mock_response - - result = fetch_url('https://example.com/page', session=mock_session) - assert result == 'data' - mock_session.get.assert_called_once() - - -def test_fetch_url_playwright_route(tmp_cache_dir): - url = 'https://pw-only.example.com/page' - with patch.object(scrape_utils, 'PLAYWRIGHT_REQUIRED_HOSTS', ['pw-only.example.com']), \ - patch('utils.scrape_utils._fetch_with_playwright', return_value='pw') as mock_pw: - result = fetch_url(url) - assert result == 'pw' - mock_pw.assert_called_once_with(url) - - -# -- close_browser (smoke test) --------------------------------------------- - -def test_close_browser_no_crash(): - close_browser() diff --git a/db/tests/test_wii_rom_set.py b/db/tests/test_wii_rom_set.py deleted file mode 100644 index dfe86b64..00000000 --- a/db/tests/test_wii_rom_set.py +++ /dev/null @@ -1,60 +0,0 @@ -"""Tests for parsers/wii_rom_set_by_ghostware.py — all pure functions.""" -from __future__ import annotations - -import pytest -from parsers.wii_rom_set_by_ghostware import get_clean_title, parse, parse_id, process_entry - - -# -- parse_id ---------------------------------------------------------------- - -@pytest.mark.parametrize('title, expected', [ - ('Game Title [RMGP01]', 'RMGP01'), - ('Game Title_RMGP01.wbfs', 'RMGP01'), - ('Game Title (RMGP01)', 'RMGP01'), - ('Game Title {RMGP01}', 'RMGP01'), - ('No ID Here', None), - ('Short [AB]', None), -]) -def test_parse_id(title: str, expected: str | None): - assert parse_id(title) == expected - - -# -- get_clean_title --------------------------------------------------------- - -def test_clean_strips_bracket_id(): - assert get_clean_title('Super Mario Galaxy [RMGP01]') == 'Super Mario Galaxy' - - -def test_clean_strips_underscore_id(): - result = get_clean_title('Zelda_SOUPX2.wbfs') - assert 'SOUPX2' not in result - - -def test_clean_no_id(): - assert get_clean_title('No ID') == 'No ID' - - -# -- process_entry / parse --------------------------------------------------- - -def test_process_entry_sets_fields(): - entry = {'title': 'Game [ABCDE1]'} - process_entry(entry) - assert entry['rom_id'] == 'ABCDE1' - assert entry['title'] == 'Game' - - -def test_process_entry_no_id(): - entry = {'title': 'No ID'} - process_entry(entry) - assert entry.get('rom_id') is None - - -def test_parse_processes_list(): - entries = [ - {'title': 'Game One [AAAAAA]'}, - {'title': 'Game Two [BBBBBB]'}, - ] - result = parse(entries, {}) - assert len(result) == 2 - assert result[0]['rom_id'] == 'AAAAAA' - assert result[1]['rom_id'] == 'BBBBBB' diff --git a/db/utils/cache_manager.py b/db/utils/cache_manager.py deleted file mode 100644 index b1c897a7..00000000 --- a/db/utils/cache_manager.py +++ /dev/null @@ -1,67 +0,0 @@ -""" -This module provides utility functions for caching HTTP responses to a local directory. -It includes functionality to sanitize URLs into valid filenames, save responses to cache, -and retrieve cached responses with optional expiration. -""" -import os -import re -import time - -# Directory name where cached responses will be stored -CACHE_DIRNAME = 'cache' - -# Cache expiration in days (0 = never expire) -CACHE_MAX_AGE_DAYS = 7 - -# Ensure the cache directory exists -if not os.path.exists(CACHE_DIRNAME): - os.mkdir(CACHE_DIRNAME) - - -def get_cached_response_filename(url: str) -> str: - """Generate a safe filename for caching a response based on the given URL.""" - return re.sub(r"[\\/:\*\?\"<>|]", '_', url) - - -def cache_response(url: str, response: str) -> None: - """Cache the response content for a given URL.""" - filename = get_cached_response_filename(url) - with open(f'{CACHE_DIRNAME}/{filename}', 'w', encoding='utf-8') as f: - f.write(response) - - -def get_cached_response(url: str, max_age_days: int = CACHE_MAX_AGE_DAYS) -> str | None: - """Retrieve the cached response if it exists and is not expired. - - Args: - url: The URL to look up in cache - max_age_days: Maximum age in days (0 = never expire) - - Returns: - Cached response string or None if not found/expired - """ - filename = get_cached_response_filename(url) - filepath = f'{CACHE_DIRNAME}/{filename}' - - if not os.path.exists(filepath): - return None - - # Check cache age if expiration is enabled - if max_age_days > 0: - file_age_days = (time.time() - os.path.getmtime(filepath)) / 86400 - if file_age_days > max_age_days: - return None # Cache expired - - with open(filepath, encoding='utf-8') as f: - return f.read() - - -def get_cache_age_days(url: str) -> float | None: - """Get the age of a cached response in days, or None if not cached.""" - filename = get_cached_response_filename(url) - filepath = f'{CACHE_DIRNAME}/{filename}' - - if not os.path.exists(filepath): - return None - - return (time.time() - os.path.getmtime(filepath)) / 86400 diff --git a/db/utils/parse_utils.py b/db/utils/parse_utils.py deleted file mode 100644 index 0a8c9b86..00000000 --- a/db/utils/parse_utils.py +++ /dev/null @@ -1,124 +0,0 @@ -""" -This module provides utility functions for parsing, normalizing, and manipulating strings, filenames, -URLs, and file sizes. These functions are designed to handle common tasks such as sanitizing input, -creating slugs, and converting between human-readable and byte-based file sizes. -""" -import os -import re -import urllib.parse -from typing import Any -from unidecode import unidecode - - -def replace_invalid_chars(title: str) -> str: - """Replace invalid characters in a string with valid substitutes.""" - for value1, value2 in { - '+': 'plus', - '&': 'and', - '™': '', - '©': '', - '®': '' - }.items(): - title = title.replace(value1, f' {value2} ') - return title - - -def remove_ext(filename: str) -> str: - """Remove the file extension from a given filename.""" - basename = os.path.basename(filename) - name, ext = os.path.splitext(basename) - - # If no extension exists, return the original filename - if ext == '': - return filename - return name - - -def normalize_repeated_chars(text: str, char: str) -> str: - """Replace consecutive occurrences of a character with a single instance.""" - escaped_char = re.escape(char) # Escape special characters for regex - return re.sub(f'{escaped_char}+', char, text).strip() - - -def create_slug(entry: dict[str, Any]) -> str: - """Create a URL-friendly slug from an entry dictionary.""" - title = entry['title'] - - title = replace_invalid_chars(title) - title = unidecode(title) - platform = entry['platform'] - regions = '-'.join(entry['regions']) - slug = f"{title}-{platform}-{regions}" - - slug = re.sub(r"[^a-zA-Z0-9-]", '-', slug).lower() - slug = normalize_repeated_chars(slug, '-').strip('-') - - return slug - - -def create_search_key(title: str) -> str: - """Generate a search-friendly key from the given title by normalizing and sanitizing it.""" - title = replace_invalid_chars(title) - title = unidecode(title) - title = title.lower() - title = re.sub(r"[^a-z0-9]", '', title) - title = title.strip() - - return title - - -def size_bytes_to_str(size: int | float) -> str: - """Convert a size in bytes to a human-readable string with appropriate units.""" - suffixes = ['B', 'K', 'M', 'G', 'T', 'P'] - i = 0 - while size >= 1024 and i < len(suffixes) - 1: - size /= 1024.0 - i += 1 - - # Format the size to two decimal places and remove trailing zeros - f = ('%.2f' % size).rstrip('0').rstrip('.') - return '%s%s' % (f, suffixes[i]) - - -def size_str_to_bytes(size_str: str) -> int: - """Convert a human-readable size string to bytes.""" - if not size_str or not size_str.strip(): - return 0 - - # Extract the first alphabetic character as the unit - unit = 'B' # Default to bytes - for character in size_str.upper(): - if character in 'BKMGT': - unit = character - break - - # Extract the numeric part of the size string - numeric_str = re.sub(r'[^\d\.]', '', size_str) - if not numeric_str: - return 0 - - try: - size = float(numeric_str) - except ValueError: - return 0 - - if unit == 'B': - pass - elif unit == 'K': - size = size * 1024 - elif unit == 'M': - size = size * 1024 * 1024 - elif unit == 'G': - size = size * 1024 * 1024 * 1024 - elif unit == 'T': - size = size * 1024 * 1024 * 1024 * 1024 - - return int(size) - - -def join_urls(url: str, *links: str) -> str: - """Join a base URL with one or more relative links.""" - for link in links: - # Ensure proper joining of URLs by stripping and appending slashes - url = urllib.parse.urljoin(url.rstrip('/') + '/', link.lstrip('/')) - return url diff --git a/db/utils/scrape_utils.py b/db/utils/scrape_utils.py deleted file mode 100644 index 2eb4c601..00000000 --- a/db/utils/scrape_utils.py +++ /dev/null @@ -1,152 +0,0 @@ -""" -This module provides utilities for scraping web content and caching responses. -""" -import time -from typing import Any - -import cloudscraper -from playwright.sync_api import sync_playwright - -from utils import cache_manager - -# Use browser-like headers instead of curl -BROWSER_HEADERS = { - 'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36', - 'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8', - 'Accept-Language': 'en-US,en;q=0.5', - 'Accept-Encoding': 'gzip, deflate, br', - 'DNT': '1', - 'Connection': 'keep-alive', - 'Upgrade-Insecure-Requests': '1' -} - -# Sites that require Playwright (real browser) due to TLS fingerprinting. -# repo.mariocube.com flipped: Cloudflare now challenges headless Chromium -# but passes cloudscraper, so it uses the default path again. -PLAYWRIGHT_REQUIRED_HOSTS = [] - -# Rate limiting settings -MAX_RETRIES = 5 -RETRY_DELAY = 3 # seconds between retries -REQUEST_DELAY = 1 # seconds between requests to avoid rate limiting - -# Global Playwright browser instance for efficiency -_playwright = None -_browser = None -_last_request_time = 0 - - -def _get_browser(): - """Get or create the Playwright browser instance.""" - global _playwright, _browser - if _browser is None: - _playwright = sync_playwright().start() - _browser = _playwright.chromium.launch(headless=True) - return _browser - - -def close_browser(): - """Close the Playwright browser when done.""" - global _playwright, _browser - if _browser: - _browser.close() - _browser = None - if _playwright: - _playwright.stop() - _playwright = None - - -def _needs_playwright(url: str) -> bool: - """Check if a URL requires Playwright for TLS fingerprinting bypass.""" - return any(host in url for host in PLAYWRIGHT_REQUIRED_HOSTS) - - -def _rate_limit() -> None: - """Enforce rate limiting between requests.""" - global _last_request_time - elapsed = time.time() - _last_request_time - if elapsed < REQUEST_DELAY: - time.sleep(REQUEST_DELAY - elapsed) - _last_request_time = time.time() - - -def _fetch_with_playwright(url: str) -> str | None: - """Fetch URL using Playwright (real browser) with retry logic.""" - browser = _get_browser() - - # Show progress for slow Playwright fetches - short_url = url.split('/')[-2] if url.endswith('/') else url.split('/')[-1] - print(f" Fetching {short_url[:50]}... ", end='', flush=True) - - for attempt in range(MAX_RETRIES): - _rate_limit() - page = browser.new_page() - try: - response = page.goto(url, wait_until='domcontentloaded', timeout=60000) - if response and response.ok: - content = page.content() - page.close() - print("OK") - return content - page.close() - print("failed") - return None - except Exception as e: - page.close() - if attempt < MAX_RETRIES - 1: - wait_time = RETRY_DELAY * (attempt + 1) # Exponential backoff - print(f"retry {attempt + 1}... ", end='', flush=True) - time.sleep(wait_time) - else: - print(f"failed ({e})") - return None - return None - - -def create_scraper_session(headers: dict[str, str] | None = None) -> Any: - """Create a scraper session and optionally apply custom headers.""" - session = cloudscraper.create_scraper( - browser={ - 'browser': 'chrome', - 'platform': 'windows', - 'mobile': False - } - ) - applied_headers = headers or BROWSER_HEADERS - if applied_headers: - session.headers.update(applied_headers) - return session - - -def fetch_url(url: str, session: Any = None) -> str | None: - """Fetch the content of a URL and cache the response.""" - # Get short URL for display (handle trailing slashes) - url_stripped = url.rstrip('/') - short_url = url_stripped.split('/')[-1][:50] if '/' in url_stripped else url_stripped[:50] - - # Use Playwright for sites with strict TLS fingerprinting - if _needs_playwright(url): - response = _fetch_with_playwright(url) - if response: - cache_manager.cache_response(url, response) - return response - - # Use cloudscraper for other sites - if not session: - session = create_scraper_session(BROWSER_HEADERS) - - try: - r = session.get(url, timeout=60) - - if not r.ok: - print(f" {short_url}... HTTP {r.status_code}") - return None - - response = r.text - cache_manager.cache_response(url, response) - print(f" {short_url}... OK") - - return response - except Exception as e: - print(f" {short_url}... error: {e}") - return None diff --git a/db/workflow.py b/db/workflow.py deleted file mode 100644 index 4b54797d..00000000 --- a/db/workflow.py +++ /dev/null @@ -1,42 +0,0 @@ -#!/usr/bin/env python -""" -Generate the ROM database. - -Usage: - python workflow.py # Fresh download of everything - python workflow.py --use-cached # Reuse cached HTTP responses - python workflow.py --skip-minerva # Skip the 1.7 GB hashes.db download - python workflow.py --skip-ra # Skip RetroAchievements lookups -""" -import os -import sys -from make import make -from scripts.download_gametdb_xmls import download_gametdb_xmls -from scripts.download_libretro_dats import download_libretro_dats -from scripts.download_mame_hashes import download_mame_hashes -from scripts.download_minerva_artefacts import download_minerva_artefacts -from scripts.download_retroachievements import download_retroachievements - -if __name__ == '__main__': - os.chdir(os.path.dirname(os.path.realpath(__file__))) - - args = sys.argv[1:] - use_cached = '--use-cached' in args - skip_minerva = '--skip-minerva' in args - skip_ra = '--skip-ra' in args - - if use_cached: - print("Using cached HTTP responses where available.\n") - - download_gametdb_xmls() - download_libretro_dats() - download_mame_hashes() - if not skip_minerva: - download_minerva_artefacts() - else: - print('Skipping MiNERVA artefacts (--skip-minerva); plugin will no-op.') - if not skip_ra: - download_retroachievements(use_cached=use_cached) - else: - print('Skipping RetroAchievements (--skip-ra); RA columns will be empty.') - make(use_cached=use_cached)