From dda01de481fa97a680df627be95b83b44beeb197 Mon Sep 17 00:00:00 2001 From: Seungpyo1007 Date: Wed, 30 Sep 2026 15:43:06 +0900 Subject: [PATCH] perf(dump): dump only the collections a change can affect `app.dump --only COLLECTION` (repeatable) and `--changed-since BASE` restrict the replay to the affected collections and keep the other manifest entries instead of overwriting them. brand changes still mean everything; soc/cpu/gpu add the collections that embed them; scored collections are always dumped whole because scores are partly population-relative. Refs GetTechAPI/TechAPI#350 --- app/dump.py | 78 +++++++++++++++++++++++++++++++--- tests/integration/test_dump.py | 28 +++++++++++- 2 files changed, 100 insertions(+), 6 deletions(-) diff --git a/app/dump.py b/app/dump.py index 8049ed9..7989438 100644 --- a/app/dump.py +++ b/app/dump.py @@ -19,6 +19,7 @@ from fastapi.testclient import TestClient from app.categories import COLLECTIONS as CATEGORY_COLLECTIONS +from app.validate import changed_paths OUTPUT_DIR = Path(__file__).resolve().parent.parent / "dump" @@ -29,12 +30,21 @@ PAGE_LIMIT = 100 # API max page size (ยง7.3) -def resolve_collections(exclude: list[str] | None = None) -> list[str]: - """Return the collections to dump, minus ``exclude``. +def resolve_collections( + exclude: list[str] | None = None, only: list[str] | None = None +) -> list[str]: + """Return the collections to dump: just ``only`` if given, minus ``exclude``. Unknown names raise instead of being ignored, so a typo in a workflow fails loudly rather than silently dumping everything. """ + if only: + unknown = sorted(set(only) - set(COLLECTIONS)) + if unknown: + raise ValueError( + f"unknown collection(s) {unknown}; valid names: {', '.join(COLLECTIONS)}" + ) + return [r for r in COLLECTIONS if r in set(only) and r not in set(exclude or [])] if not exclude: return list(COLLECTIONS) unknown = sorted(set(exclude) - set(COLLECTIONS)) @@ -45,6 +55,31 @@ def resolve_collections(exclude: list[str] | None = None) -> list[str]: return [resource for resource in COLLECTIONS if resource not in set(exclude)] +# Extra collections whose pages embed another category's data. ``None`` = everything. +# Scored collections (smartphones/cpus/gpus/socs) are always re-dumped whole: scores +# are partly relative to the population, so one record can move its neighbours. +DEPENDENTS: dict[str, list[str] | None] = { + "brand": None, + "soc": ["socs", "smartphones", "tablets", "watches", "pdas"], + "cpu": ["cpus", "laptops"], + "gpu": ["gpus", "laptops"], +} + + +def collections_for_changes(paths: set[str]) -> list[str]: + """Collections a set of changed seed paths (``data/`` stripped) can affect.""" + affected: set[str] = set() + for path in paths: + category = path.split("/", 1)[0] + if category not in CATEGORY_COLLECTIONS: + continue # e.g. _verify/: not part of the dump + deps = DEPENDENTS.get(category, [CATEGORY_COLLECTIONS[category]]) + if deps is None: + return list(COLLECTIONS) + affected.update(deps) + return [resource for resource in COLLECTIONS if resource in affected] + + def _write_json(path: Path, data: object) -> None: text = json.dumps(data, indent=2, ensure_ascii=False) + "\n" # Leave identical pages untouched: rewriting ~1M unchanged files resets @@ -105,6 +140,14 @@ def generate( """Write the full static dump. Returns the number of detail files per collection.""" counts: dict[str, int] = {} manifest: dict[str, object] = {"version": "v1", "collections": {}} + if collections is not None and set(collections) != set(COLLECTIONS): + # Partial run: keep the entries of collections we are not touching. + try: + previous = json.loads((output_dir / "v1" / "index.json").read_text(encoding="utf-8")) + if isinstance(previous.get("collections"), dict): + manifest["collections"] = previous["collections"] + except (FileNotFoundError, json.JSONDecodeError): + pass for resource in collections or COLLECTIONS: count, items = _fetch_all(client, resource) @@ -142,14 +185,21 @@ def generate( return counts -def run(output_dir: Path = OUTPUT_DIR, exclude: list[str] | None = None) -> None: +def run( + output_dir: Path = OUTPUT_DIR, + exclude: list[str] | None = None, + only: list[str] | None = None, +) -> None: from sqlmodel import Session from app.database import create_db_and_tables, engine from app.main import app from app.seed import seed - collections = resolve_collections(exclude) + collections = resolve_collections(exclude, only) + if not collections: + print("Nothing to dump.") + return create_db_and_tables() with Session(engine) as session: @@ -176,5 +226,23 @@ def run(output_dir: Path = OUTPUT_DIR, exclude: list[str] | None = None) -> None "writing hundreds of thousands of files that are discarded anyway." ), ) + parser.add_argument( + "--only", + action="append", + default=[], + metavar="COLLECTION", + help="dump just this collection, repeatable; other manifest entries are kept", + ) + parser.add_argument( + "--changed-since", + metavar="BASE", + help="dump only the collections affected by data changed since BASE", + ) args = parser.parse_args() - run(args.output, args.exclude) + only = args.only + if args.changed_since: + only = only + collections_for_changes(changed_paths(args.changed_since)) + if not only: + print("No dump-relevant data changed.") + raise SystemExit(0) + run(args.output, args.exclude, only) diff --git a/tests/integration/test_dump.py b/tests/integration/test_dump.py index 1f2c121..403ed5a 100644 --- a/tests/integration/test_dump.py +++ b/tests/integration/test_dump.py @@ -8,7 +8,13 @@ import pytest from fastapi.testclient import TestClient -from app.dump import COLLECTIONS, _prune_orphaned_pages, generate, resolve_collections +from app.dump import ( + COLLECTIONS, + _prune_orphaned_pages, + collections_for_changes, + generate, + resolve_collections, +) from tests.integration.mobile_device_fixtures import ensure_mobile_device_fixtures @@ -137,3 +143,23 @@ def test_prune_orphaned_pages_leaves_files_and_valid_slugs(tmp_path: Path) -> No def test_prune_orphaned_pages_noop_when_dir_missing(tmp_path: Path) -> None: assert _prune_orphaned_pages(tmp_path / "does-not-exist", valid_slugs=set()) == [] + + +def test_partial_dump_keeps_other_manifest_entries(client: TestClient, tmp_path: Path) -> None: + generate(client, output_dir=tmp_path, collections=["cpus", "gpus"]) + generate(client, output_dir=tmp_path, collections=["cpus"]) + manifest = json.loads((tmp_path / "v1" / "index.json").read_text()) + assert set(manifest["collections"]) == {"cpus", "gpus"} + + +def test_resolve_collections_only() -> None: + assert resolve_collections(only=["laptops", "cpus"]) == ["cpus", "laptops"] + with pytest.raises(ValueError): + resolve_collections(only=["nope"]) + + +def test_collections_for_changes_maps_dependents() -> None: + assert collections_for_changes({"website/a/x.json"}) == ["websites"] + assert collections_for_changes({"cpu/i/1/x.json"}) == ["cpus", "laptops"] + assert collections_for_changes({"_verify/status.json"}) == [] + assert collections_for_changes({"brand/us/acme.json"}) == COLLECTIONS