diff --git a/app/ingest/enrich.py b/app/ingest/enrich.py index 2e56990..194d1ac 100644 --- a/app/ingest/enrich.py +++ b/app/ingest/enrich.py @@ -149,7 +149,12 @@ def enrich( break processed += 1 name = rec.get("name", "") - out = resolver(client, name, overrides.get(name)) + if resolver is topcpu.resolve or resolver is topcpu.resolve_gpu: + out = resolver( + client, name, overrides.get(name), architecture=rec.get("architecture") + ) + else: + out = resolver(client, name, overrides.get(name)) if sleep: time.sleep(sleep) if out is None: diff --git a/app/ingest/sources/topcpu.py b/app/ingest/sources/topcpu.py index 1a001e8..12635ca 100644 --- a/app/ingest/sources/topcpu.py +++ b/app/ingest/sources/topcpu.py @@ -60,6 +60,32 @@ _DIGITS = re.compile(r"[^0-9]") _NUM = re.compile(r"[\d,]+\.?\d*") +# Architecture families, including versioned labels such as Tesla 2.0 and +# TeraScale 3. These cannot provide the DX12 driver required by Time Spy (FL11_0). +# AMD's Windows driver support matrix separates HD5000/6000 (DX11) from GCN: +# https://www.amd.com/en/resources/support-articles/faqs/GPU-615.html +# NVIDIA's DX12 support starts at Fermi, INCLUDING Fermi and early Kepler: +# https://nvidia.custhelp.com/app/answers/detail/a_id/3711 +# https://support.benchmarks.ul.com/support/solutions/articles/44002136075 +_PRE_DX12_ARCHITECTURE = re.compile( + r"\b(?:TeraScale|Tesla|Curie|Rankine|Kelvin|Celsius|Ultra[- ]Threaded SE|" + r"R100|R200|R300|R400)\b", + re.IGNORECASE, +) +_DX12_FIELDS = frozenset({"timespy_score", "timespy_extreme_score", "speedway_score"}) + + +def is_pre_dx12(architecture: str | None) -> bool: + """Recognize known unsupported architectures; unknown labels are not guessed. + + Use the record's microarchitecture, never its release year or product brand + (e.g. a Tesla-branded card may have a newer, DX12-capable architecture). + This is a pre-DX12 exclusion, not a complete Speed Way/Ultimate eligibility + check. OctaneBench uses CUDA, and FP32 is a spec, so neither is DX12-filtered. + """ + return bool(architecture and _PRE_DX12_ARCHITECTURE.search(architecture)) + + # Cached normalized score maps, keyed by (url, normalizer name). _caches: dict[str, dict[str, float]] = {} @@ -110,9 +136,15 @@ def reset_cache() -> None: def resolve( - client: httpx.Client, name: str, id_override: str | None = None + client: httpx.Client, + name: str, + id_override: str | None = None, + *, + architecture: str | None = None, ) -> tuple[dict[str, int], str] | None: """GPU Time Spy resolver: ``({"timespy_score": score}, url)`` or None.""" + if is_pre_dx12(architecture): + return None hit = _load_map(client, TIMESPY_URL, normalize_gpu).get(normalize_gpu(name)) if hit is None: return None @@ -140,23 +172,26 @@ def resolve_cpu( def resolve_gpu( - client: httpx.Client, name: str, id_override: str | None = None + client: httpx.Client, + name: str, + id_override: str | None = None, + *, + architecture: str | None = None, ) -> tuple[dict[str, float], str] | None: """GPU breadth resolver: Time Spy Extreme / Speed Way / OctaneBench / FP32. - WARNING: topcpu publishes unreliable *estimated* 3DMark/Octane scores for - pre-DX12 cards that can't actually run them (e.g. Radeon HD 5670 "Time Spy" - 3897 — physically impossible; contradicts its PassMark G3D). The same applies - to ``resolve`` (Time Spy). When enriching, GUARD on DX12 capability - (release year >= 2011 / GCN/Kepler+) before writing timespy*/speedway/ - octanebench — only fp32_tflops (a spec) is era-safe. See - TechAPI/.claude/benchmark_fill_progress.md pt.7. + topcpu publishes estimated DX12 scores for cards that cannot run those + benchmarks. Pass the record's architecture to suppress Time Spy Extreme + and Speed Way for known pre-DX12 families, regardless of release year. + Non-DX12 dimensions retain their existing behavior. """ key = normalize_gpu(name) if not key: return None scores: dict[str, float] = {} for url, field, as_float in _GPU_FAMILIES: + if field in _DX12_FIELDS and is_pre_dx12(architecture): + continue v = _load_map(client, url, normalize_gpu, as_float=as_float).get(key) if v is not None: scores[field] = v diff --git a/tests/unit/test_gpu_sources.py b/tests/unit/test_gpu_sources.py index fc49f2a..0366a1c 100644 --- a/tests/unit/test_gpu_sources.py +++ b/tests/unit/test_gpu_sources.py @@ -5,7 +5,12 @@ import io import json import zipfile +from pathlib import Path +import httpx +import pytest + +from app.ingest import enrich as enrich_mod from app.ingest.sources import blender, topcpu, videocardbenchmark # --- shared GPU name normalization (variant safety) --------------------------- @@ -238,3 +243,96 @@ def test_topcpu_gpu_breadth_int_and_float() -> None: } assert "gpu-r" in url assert topcpu.resolve_gpu(_RoutingClient(routes), "Radeon RX 9999") is None + + +@pytest.mark.parametrize("architecture", [ + "TeraScale", "TeraScale 2", "terascale 3", "Tesla", "Tesla 2.0", + "Tesla 2.0 | Tesla", "Curie", "Rankine", "Kelvin", "Celsius", + "Ultra-Threaded SE", "R100", "R200", "R300", "R400", +]) +def test_topcpu_rejects_pre_dx12_time_spy(architecture: str) -> None: + topcpu.reset_cache() + client = _HtmlClient(_gpu_row("Legacy GPU", "3897")) + assert topcpu.resolve(client, "Legacy GPU", architecture=architecture) is None + + +@pytest.mark.parametrize("architecture", [ + "Fermi", "Fermi 2.0", "Kepler", "Kepler 2.0", "GCN 1.0", "RDNA 2", "Ampere", +]) +def test_topcpu_keeps_dx12_time_spy(architecture: str) -> None: + topcpu.reset_cache() + client = _HtmlClient(_gpu_row("Supported GPU", "1000")) + assert topcpu.resolve(client, "Supported GPU", architecture=architecture) == ( + {"timespy_score": 1000}, topcpu.TIMESPY_URL, + ) + + +def test_topcpu_unknown_architecture_is_not_guessed() -> None: + assert not topcpu.is_pre_dx12(None) + assert not topcpu.is_pre_dx12("") + assert not topcpu.is_pre_dx12("Unclassified") + + +@pytest.mark.parametrize("resolver,primary_field", [ + (topcpu.resolve, "timespy_score"), (topcpu.resolve_gpu, None), +]) +def test_topcpu_enrichment_uses_architecture_not_year( + tmp_path: Path, monkeypatch, resolver, primary_field, +) -> None: + topcpu.reset_cache() + # GCN and TeraScale coexisted in 2012; Fermi can run Time Spy despite 2010. + records = [ + ("Radeon HD 7350", "TeraScale 2", "2012-01-01", False), + ("Radeon HD 7970", "GCN 1.0", "2012-01-01", True), + ("GeForce GTX 480", "Fermi", "2010-03-26", True), + ("Tesla K20", "Kepler", "2012-11-12", True), + ("GeForce GT 330", "Tesla 2.0", "2024-01-01", False), + ] + gpu_dir = tmp_path / "gpu" + gpu_dir.mkdir() + for i, (name, architecture, release_date, _) in enumerate(records): + (gpu_dir / f"{i}.json").write_text(json.dumps({ + "name": name, "architecture": architecture, "release_date": release_date, + "timespy_score": None, "timespy_extreme_score": None, "speedway_score": None, + "octanebench_score": 55, "fp32_tflops": None, "source_urls": [], + }), encoding="utf-8") + + def handle(request: httpx.Request) -> httpx.Response: + value = "0.25" if "fp32-float" in str(request.url) else "1000" + return httpx.Response(200, text="".join(_gpu_row(r[0], value) for r in records)) + + monkeypatch.setattr(enrich_mod, "make_client", lambda: httpx.Client( + transport=httpx.MockTransport(handle), + )) + result = enrich_mod.enrich( + data_root=tmp_path, component="gpu", resolver=resolver, + primary_field=primary_field, sleep=0, + ) + for i, (_, _, _, supported) in enumerate(records): + written = json.loads((gpu_dir / f"{i}.json").read_text(encoding="utf-8")) + if resolver is topcpu.resolve: + assert written["timespy_score"] == (1000 if supported else None) + else: + assert written["timespy_extreme_score"] == (1000 if supported else None) + assert written["speedway_score"] == (1000 if supported else None) + assert written["fp32_tflops"] == 0.25 + # Fill-only-nulls must still preserve existing data. + assert written["octanebench_score"] == 55 + if resolver is topcpu.resolve: + assert result.unresolved == ["Radeon HD 7350", "GeForce GT 330"] + else: + assert result.unresolved == [] + + +def test_topcpu_pre_dx12_breadth_keeps_non_dx12_dimensions() -> None: + topcpu.reset_cache() + name = "GeForce GT 330" + routes = { + "3dmark-time-spy-extreme": _gpu_row(name, "2000"), + "3dmark-speed-way": _gpu_row(name, "3000"), + "octanebench": _gpu_row(name, "20"), + "fp32-float": _gpu_row(name, "0.25"), + } + out = topcpu.resolve_gpu(_RoutingClient(routes), name, architecture="Tesla 2.0") + assert out is not None + assert out[0] == {"octanebench_score": 20, "fp32_tflops": 0.25}