From dffb6e0783aa4b700ffa70a1a4713b61f5f366fc Mon Sep 17 00:00:00 2001 From: Seungpyo1007 Date: Tue, 29 Sep 2026 19:19:57 +0900 Subject: [PATCH] fix(verify): make era advisory rule cores-aware The era-vs-score check in integrity_check.py used a flat per-chip ceiling (PassMark>1500 before 2006, R23>3000 before 2011) and ignored core/thread count. Cinebench R23 and PassMark run on the physical silicon regardless of launch date, so legitimate pre-2011 high-core enthusiast parts sail past a flat gate: a 6c/12t Gulftown (i7-980X/990X) posts ~6,000-6,500 R23 today and a 4c/8t Bloomfield (i7-920) ~3,000-3,800. Today's CPU advisory re-check (Refs #98) found all 8 era findings were exactly this population of false positives (i7-920/965/870/970/980X/990X, Phenom II X6 1090T/1100T). Scale the ceiling by thread count instead (R23 1000/thread, PassMark 900/thread; threads default to cores, then 1). Pre-2011 microarchitectures top out around ~600 R23 and well under 900 PassMark per thread, so the known-good chips stop flagging, while a genuinely implausible old-chip/modern-score combo still trips it (e.g. a 2c/2009 part claiming 20,000 R23 = 10,000/thread). Extract the rule into an importable, side-effect-free era_score_outliers() helper and guard the scan under __main__ so it can be unit-tested. Add tests/unit/test_integrity_era.py covering the 8 known false positives, the thread-scaling boundary, the thread-count fallback, the pre-2006 PassMark path, and a genuine implausible case that must still flag. Verified against a live TechAPI develop checkout: the era-vs-score section now reports 0 findings (was 8) with the ratio/structural sections unchanged. Refs #98 --- integrity_check.py | 209 ++++++++++++++++++------------- tests/unit/test_integrity_era.py | 130 +++++++++++++++++++ 2 files changed, 255 insertions(+), 84 deletions(-) create mode 100644 tests/unit/test_integrity_era.py diff --git a/integrity_check.py b/integrity_check.py index fbd8f08..0dfe7e3 100644 --- a/integrity_check.py +++ b/integrity_check.py @@ -43,6 +43,45 @@ def hard(msg: str) -> None: HARD.append(msg) print(msg) +# Era-vs-score: catch wrong-variant contamination (an old chip carrying a score +# that belongs to a newer part). The original rule used a flat per-chip ceiling +# (PassMark>1500 before 2006, R23>3000 before 2011) and ignored core/thread count. +# That mis-fires on legitimate pre-2011 high-core enthusiast parts: Cinebench R23 +# and PassMark run on the physical silicon regardless of launch date, so a 6c/12t +# Gulftown (i7-980X/990X) genuinely posts ~6,000-6,500 R23 today and a 4c/8t +# Bloomfield (i7-920) ~3,000-3,800 — all above a flat 3,000 gate. The fix scales +# the ceiling by thread count: pre-2011 microarchitectures (Nehalem/Westmere/K10) +# top out around ~600 R23 and ~450 PassMark *per thread*, whereas a genuinely +# implausible "old chip, modern score" combo (e.g. a 2c/2009 part claiming 20,000 +# R23 = 10,000/thread) sits far above the per-thread ceiling and still flags. +ERA_R23_PER_THREAD = 1000 # R23 multi per thread; pre-2011 real parts are ~400-600 +ERA_PASSMARK_PER_THREAD = 900 # PassMark per thread; pre-2006 real parts are well under this + +def era_score_outliers(rec: dict) -> list[str]: + """Return era-vs-score finding messages for one CPU record (empty if none). + + Cores/threads-aware: the score ceiling scales with thread count so that + high-core-for-their-era chips are not flagged, while a per-thread score that + is implausible for the release era still is. Threads default to cores, then 1. + """ + findings: list[str] = [] + year = (rec.get("release_date") or "0")[:4] + threads = rec.get("threads") or rec.get("cores") or 1 + name = rec.get("name", "?") + pm = rec.get("passmark_cpu_mark") + r23 = rec.get("cinebench_r23_multi") + if year < "2006" and pm and pm > ERA_PASSMARK_PER_THREAD * threads: + findings.append( + f" {name!r} ({year}): passmark {pm} too high for era " + f"({pm / threads:.0f}/thread over {ERA_PASSMARK_PER_THREAD}, {threads}T)" + ) + if year < "2011" and r23 and r23 > ERA_R23_PER_THREAD * threads: + findings.append( + f" {name!r} ({year}): r23 {r23} too high for era " + f"({r23 / threads:.0f}/thread over {ERA_R23_PER_THREAD}, {threads}T)" + ) + return findings + def load(comp): recs = [] for dp, _, fs in os.walk(os.path.join(ROOT, comp)): @@ -63,89 +102,91 @@ def mad_outliers(pairs, lo=0.34, hi=3.0): def section(t): print(f"\n### {t}") -records = {category: load(category) for category in CATEGORIES} -cpus = records["cpu"]; gpus = records["gpu"] -print(f"scope: {len(CATEGORIES)}/12 categories ? {sum(map(len, records.values()))} records") -print(f"loaded CPU={len(cpus)} GPU={len(gpus)}") - -# --- 1. duplicates + slug/file + verified-no-source --- -section("structural") -for comp, recs in records.items(): - slugs, names = {}, {} - for p, fn, d in recs: - if d.get("verified") is True and not d.get("source_urls"): - hard(f" [{comp}] verified without sources: {fn}") - slugs.setdefault(d.get("slug"), []).append(fn) - names.setdefault(d.get("name"), []).append(fn) - if d.get("slug") != fn: - hard(f" [{comp}] slug!=file: {fn} slug={d.get('slug')}") - for s, fl in slugs.items(): - if len(fl) > 1: hard(f" [{comp}] DUP slug {s}: {sorted(fl)}") - for n, fl in names.items(): - if comp in ("cpu", "gpu") and len(fl) > 1: hard(f" [{comp}] DUP name {n!r}: {sorted(fl)}") - -# --- 2. AMD Ryzen line vs DESKTOP model tier-digit (2nd digit); APU/mobile excepted --- -section("CPU name/tier consistency (desktop mainstream only)") -TIERMAP = {"6": "5", "7": "7", "8": "7", "9": "9"} # 2nd model digit -> expected line -for p, fn, d in cpus: - n = d.get("name", "") - # mainstream desktop: 4-digit model, no G/U/H/HS/HX (APU/mobile) suffix - m = re.match(r"AMD Ryzen (\d) (\d)(\d)\d\d(X3D|X|XT)?$", n) - if m: - line, _gen, tier = m.group(1), m.group(2), m.group(3) - exp = TIERMAP.get(tier) - if exp and exp != line: - print(f" [tier] {n!r}: line Ryzen {line} but tier-digit {tier} → expect Ryzen {exp}") - -# --- 3. benchmark sanity: single>multi (consistent-scale benches) --- -section("CPU single>multi (cinebench/geekbench — should be multi>=single)") -for p, fn, d in cpus: - for s, mu in [("cinebench_r23_single","cinebench_r23_multi"), - ("geekbench_single","geekbench_multi"), - ("cinebench_2024_single","cinebench_2024_multi")]: - a, b = d.get(s), d.get(mu) - if a and b and a > b and (d.get("threads") or 1) > 1: - hard(f" {d['name']!r}: {s}={a} > {mu}={b}") - -# --- 4. era vs score (catch wrong-variant: old chip w/ modern score) --- -section("CPU era-vs-score outliers") -for p, fn, d in cpus: - y = (d.get("release_date") or "0")[:4] - pm = d.get("passmark_cpu_mark"); r23 = d.get("cinebench_r23_multi") - if y < "2006" and pm and pm > 1500: - print(f" {d['name']!r} ({y}): passmark {pm} too high for era") - if y < "2011" and r23 and r23 > 3000: - print(f" {d['name']!r} ({y}): r23 {r23} too high for era") - -# --- 5. cross-source correlation outliers (KEY contamination detector) --- -section("CPU cross-source ratio outliers (possible wrong-variant)") def collect(recs, fa, fb): return [(d["name"], d[fa], d[fb]) for p, fn, d in recs if d.get(fa) and d.get(fb)] -for fa, fb in [("passmark_cpu_mark","cinebench_r23_multi"), - ("passmark_cpu_mark","geekbench_multi"), - ("cinebench_r23_multi","geekbench_multi"), - ("cinebench_2024_multi","cinebench_r23_multi")]: - out = mad_outliers(collect(cpus, fa, fb)) - for label, ratio in out: - print(f" [{fa}/{fb}] {label!r}: ratio={ratio}") - -# --- 6. GPU cross-source + sanity --- -section("GPU cross-source ratio outliers + sanity") -for fa, fb in [("passmark_g3d_mark","timespy_score"), - ("timespy_score","blender_score"), - ("fp32_tflops","timespy_score"), - ("passmark_g3d_mark","fp32_tflops")]: - for label, ratio in mad_outliers(collect(gpus, fa, fb)): - print(f" [{fa}/{fb}] {label!r}: ratio={ratio}") - -print("\n(no lines under a section = clean)") - -if HARD_REPORT: - with open(HARD_REPORT, "w", encoding="utf-8") as report: - json.dump(sorted(set(HARD)), report, ensure_ascii=False, indent=2) - -if STRICT and HARD: - print(f"\n❌ integrity gate: {len(HARD)} hard anomaly(ies) — blocking refresh.") - sys.exit(1) -if STRICT: - print("\n✅ integrity gate: no hard anomalies.") + +def main() -> None: + records = {category: load(category) for category in CATEGORIES} + cpus = records["cpu"]; gpus = records["gpu"] + print(f"scope: {len(CATEGORIES)}/12 categories ? {sum(map(len, records.values()))} records") + print(f"loaded CPU={len(cpus)} GPU={len(gpus)}") + + # --- 1. duplicates + slug/file + verified-no-source --- + section("structural") + for comp, recs in records.items(): + slugs, names = {}, {} + for p, fn, d in recs: + if d.get("verified") is True and not d.get("source_urls"): + hard(f" [{comp}] verified without sources: {fn}") + slugs.setdefault(d.get("slug"), []).append(fn) + names.setdefault(d.get("name"), []).append(fn) + if d.get("slug") != fn: + hard(f" [{comp}] slug!=file: {fn} slug={d.get('slug')}") + for s, fl in slugs.items(): + if len(fl) > 1: hard(f" [{comp}] DUP slug {s}: {sorted(fl)}") + for n, fl in names.items(): + if comp in ("cpu", "gpu") and len(fl) > 1: hard(f" [{comp}] DUP name {n!r}: {sorted(fl)}") + + # --- 2. AMD Ryzen line vs DESKTOP model tier-digit (2nd digit); APU/mobile excepted --- + section("CPU name/tier consistency (desktop mainstream only)") + TIERMAP = {"6": "5", "7": "7", "8": "7", "9": "9"} # 2nd model digit -> expected line + for p, fn, d in cpus: + n = d.get("name", "") + # mainstream desktop: 4-digit model, no G/U/H/HS/HX (APU/mobile) suffix + m = re.match(r"AMD Ryzen (\d) (\d)(\d)\d\d(X3D|X|XT)?$", n) + if m: + line, _gen, tier = m.group(1), m.group(2), m.group(3) + exp = TIERMAP.get(tier) + if exp and exp != line: + print(f" [tier] {n!r}: line Ryzen {line} but tier-digit {tier} → expect Ryzen {exp}") + + # --- 3. benchmark sanity: single>multi (consistent-scale benches) --- + section("CPU single>multi (cinebench/geekbench — should be multi>=single)") + for p, fn, d in cpus: + for s, mu in [("cinebench_r23_single","cinebench_r23_multi"), + ("geekbench_single","geekbench_multi"), + ("cinebench_2024_single","cinebench_2024_multi")]: + a, b = d.get(s), d.get(mu) + if a and b and a > b and (d.get("threads") or 1) > 1: + hard(f" {d['name']!r}: {s}={a} > {mu}={b}") + + # --- 4. era vs score (catch wrong-variant: old chip w/ modern score) --- + section("CPU era-vs-score outliers") + for p, fn, d in cpus: + for msg in era_score_outliers(d): + print(msg) + + # --- 5. cross-source correlation outliers (KEY contamination detector) --- + section("CPU cross-source ratio outliers (possible wrong-variant)") + for fa, fb in [("passmark_cpu_mark","cinebench_r23_multi"), + ("passmark_cpu_mark","geekbench_multi"), + ("cinebench_r23_multi","geekbench_multi"), + ("cinebench_2024_multi","cinebench_r23_multi")]: + out = mad_outliers(collect(cpus, fa, fb)) + for label, ratio in out: + print(f" [{fa}/{fb}] {label!r}: ratio={ratio}") + + # --- 6. GPU cross-source + sanity --- + section("GPU cross-source ratio outliers + sanity") + for fa, fb in [("passmark_g3d_mark","timespy_score"), + ("timespy_score","blender_score"), + ("fp32_tflops","timespy_score"), + ("passmark_g3d_mark","fp32_tflops")]: + for label, ratio in mad_outliers(collect(gpus, fa, fb)): + print(f" [{fa}/{fb}] {label!r}: ratio={ratio}") + + print("\n(no lines under a section = clean)") + + if HARD_REPORT: + with open(HARD_REPORT, "w", encoding="utf-8") as report: + json.dump(sorted(set(HARD)), report, ensure_ascii=False, indent=2) + + if STRICT and HARD: + print(f"\n❌ integrity gate: {len(HARD)} hard anomaly(ies) — blocking refresh.") + sys.exit(1) + if STRICT: + print("\n✅ integrity gate: no hard anomalies.") + + +if __name__ == "__main__": + main() diff --git a/tests/unit/test_integrity_era.py b/tests/unit/test_integrity_era.py new file mode 100644 index 0000000..6a18576 --- /dev/null +++ b/tests/unit/test_integrity_era.py @@ -0,0 +1,130 @@ +"""Unit tests for the cores/threads-aware era-vs-score rule in ``integrity_check``. + +The era rule flags a wrong-variant contamination signal: an old chip carrying a +score that belongs to a newer part. The original rule used a flat per-chip ceiling +(PassMark>1500 before 2006, R23>3000 before 2011) and ignored core/thread count, +which mis-fired on legitimate pre-2011 high-core enthusiast parts. These tests pin +the fix: those known-good chips must not flag, while a genuinely implausible +year/score combination still must. +""" + +from __future__ import annotations + +import integrity_check as ic + + +def _rec(**overrides: object) -> dict: + base: dict = dict( + name="Test CPU", + release_date="2010-01-01", + cores=6, + threads=12, + ) + base.update(overrides) + return base + + +# The 8 pre-2011 high-core-for-their-era chips flagged as false positives during +# the 2026-09-29 CPU advisory re-check (TechEngine #98). Real cores/threads and the +# stored Cinebench R23 multi scores; every one must be treated as plausible. +KNOWN_GOOD_ERA_CHIPS = [ + ("Intel Core i7-920 (Bloomfield)", "2008", 4, 8, 3800), + ("Intel Core i7-965 Extreme Edition (Bloomfield)", "2008", 4, 8, 4500), + ("Intel Core i7-870 (Lynnfield)", "2009", 4, 8, 4200), + ("Intel Core i7-970", "2010", 6, 12, 5800), + ("Intel Core i7-980X Extreme Edition (Gulftown)", "2010", 6, 12, 6500), + ("Intel Core i7-990X Extreme Edition", "2010", 6, 12, 6300), + ("AMD Phenom II X6 1090T Black Edition (Thuban)", "2010", 6, 6, 3400), + ("AMD Phenom II X6 1100T Black Edition (Thuban)", "2010", 6, 6, 3500), +] + + +def test_known_good_pre2011_high_core_chips_are_not_flagged() -> None: + for name, year, cores, threads, r23 in KNOWN_GOOD_ERA_CHIPS: + rec = _rec( + name=name, + release_date=f"{year}-06-01", + cores=cores, + threads=threads, + cinebench_r23_multi=r23, + ) + assert ic.era_score_outliers(rec) == [], f"{name} should not be flagged" + + +def test_genuinely_implausible_old_chip_still_flags() -> None: + # A 2-core/2009 part claiming a 20,000 R23 multi (10,000/thread) is impossible + # for the era and must still be flagged. + rec = _rec( + name="Bogus Dual-Core 2009", + release_date="2009-01-01", + cores=2, + threads=2, + cinebench_r23_multi=20000, + ) + findings = ic.era_score_outliers(rec) + assert len(findings) == 1 + assert "r23 20000 too high for era" in findings[0] + + +def test_r23_ceiling_scales_with_threads() -> None: + # Same per-thread score, different thread counts: a high absolute R23 that is + # reasonable per-thread for a many-threaded part is fine, but the identical + # per-thread rate on very few threads is not what trips the rule -- the rule + # is about total score vs thread budget. Just above / below the per-thread + # ceiling around the boundary. + threads = 12 + ceiling = ic.ERA_R23_PER_THREAD * threads + below = _rec(threads=threads, cinebench_r23_multi=ceiling - 1) + above = _rec(threads=threads, cinebench_r23_multi=ceiling + 1) + assert ic.era_score_outliers(below) == [] + assert len(ic.era_score_outliers(above)) == 1 + + +def test_threads_default_to_cores_then_one() -> None: + # threads missing -> falls back to cores + rec_cores = _rec(threads=None, cores=6, cinebench_r23_multi=6 * ic.ERA_R23_PER_THREAD + 1) + assert len(ic.era_score_outliers(rec_cores)) == 1 + # both missing -> falls back to 1 thread + rec_one = dict(name="No core info", release_date="2010-01-01", + cinebench_r23_multi=ic.ERA_R23_PER_THREAD + 1) + assert len(ic.era_score_outliers(rec_one)) == 1 + + +def test_modern_chips_are_never_flagged_regardless_of_score() -> None: + # The year gate means post-2011 parts are out of scope entirely. + rec = _rec( + name="AMD Ryzen 9 9950X", + release_date="2024-08-15", + cores=16, + threads=32, + cinebench_r23_multi=42000, + passmark_cpu_mark=65756, + ) + assert ic.era_score_outliers(rec) == [] + + +def test_passmark_era_rule_is_thread_aware() -> None: + # Pre-2006 PassMark ceiling also scales by threads. A single-core 2004 chip + # with an era-appropriate PassMark is fine; an absurd one still flags. + ok = _rec(name="Pentium 4 (2004)", release_date="2004-06-01", cores=1, threads=1, + passmark_cpu_mark=ic.ERA_PASSMARK_PER_THREAD - 1) + bad = _rec(name="Pentium 4 (2004)", release_date="2004-06-01", cores=1, threads=1, + passmark_cpu_mark=ic.ERA_PASSMARK_PER_THREAD * 10) + assert ic.era_score_outliers(ok) == [] + findings = ic.era_score_outliers(bad) + assert len(findings) == 1 + assert "passmark" in findings[0] + + +def test_missing_scores_never_flag() -> None: + assert ic.era_score_outliers(_rec(cinebench_r23_multi=None, passmark_cpu_mark=None)) == [] + assert ic.era_score_outliers(_rec()) == [] # no score fields at all + + +def test_importing_integrity_check_runs_no_scan(capsys) -> None: + # The module must be importable without triggering the filesystem scan + # (guarded by ``if __name__ == '__main__'``). Re-importing is a no-op; assert + # the pure helper is present and callable. + assert callable(ic.era_score_outliers) + captured = capsys.readouterr() + assert "scope:" not in captured.out