Skip to content

Commit 2bbcaac

Browse files
committed
fix(verify): make era advisory rule cores-aware
The era-vs-score check in integrity_check.py used a flat per-chip ceiling (PassMark>1500 before 2006, R23>3000 before 2011) and ignored core/thread count. Cinebench R23 and PassMark run on the physical silicon regardless of launch date, so legitimate pre-2011 high-core enthusiast parts sail past a flat gate: a 6c/12t Gulftown (i7-980X/990X) posts ~6,000-6,500 R23 today and a 4c/8t Bloomfield (i7-920) ~3,000-3,800. Today's CPU advisory re-check (Refs #98) found all 8 era findings were exactly this population of false positives (i7-920/965/870/970/980X/990X, Phenom II X6 1090T/1100T). Scale the ceiling by thread count instead (R23 1000/thread, PassMark 900/thread; threads default to cores, then 1). Pre-2011 microarchitectures top out around ~600 R23 and well under 900 PassMark per thread, so the known-good chips stop flagging, while a genuinely implausible old-chip/modern-score combo still trips it (e.g. a 2c/2009 part claiming 20,000 R23 = 10,000/thread). Extract the rule into an importable, side-effect-free era_score_outliers() helper and guard the scan under __main__ so it can be unit-tested. Add tests/unit/test_integrity_era.py covering the 8 known false positives, the thread-scaling boundary, the thread-count fallback, the pre-2006 PassMark path, and a genuine implausible case that must still flag. Verified against a live TechAPI develop checkout: the era-vs-score section now reports 0 findings (was 8) with the ratio/structural sections unchanged. Refs #98
1 parent d93637c commit 2bbcaac

2 files changed

Lines changed: 255 additions & 84 deletions

File tree

‎integrity_check.py‎

Lines changed: 125 additions & 84 deletions
Original file line numberDiff line numberDiff line change
@@ -43,6 +43,45 @@ def hard(msg: str) -> None:
4343
HARD.append(msg)
4444
print(msg)
4545

46+
# Era-vs-score: catch wrong-variant contamination (an old chip carrying a score
47+
# that belongs to a newer part). The original rule used a flat per-chip ceiling
48+
# (PassMark>1500 before 2006, R23>3000 before 2011) and ignored core/thread count.
49+
# That mis-fires on legitimate pre-2011 high-core enthusiast parts: Cinebench R23
50+
# and PassMark run on the physical silicon regardless of launch date, so a 6c/12t
51+
# Gulftown (i7-980X/990X) genuinely posts ~6,000-6,500 R23 today and a 4c/8t
52+
# Bloomfield (i7-920) ~3,000-3,800 — all above a flat 3,000 gate. The fix scales
53+
# the ceiling by thread count: pre-2011 microarchitectures (Nehalem/Westmere/K10)
54+
# top out around ~600 R23 and ~450 PassMark *per thread*, whereas a genuinely
55+
# implausible "old chip, modern score" combo (e.g. a 2c/2009 part claiming 20,000
56+
# R23 = 10,000/thread) sits far above the per-thread ceiling and still flags.
57+
ERA_R23_PER_THREAD = 1000 # R23 multi per thread; pre-2011 real parts are ~400-600
58+
ERA_PASSMARK_PER_THREAD = 900 # PassMark per thread; pre-2006 real parts are well under this
59+
60+
def era_score_outliers(rec: dict) -> list[str]:
61+
"""Return era-vs-score finding messages for one CPU record (empty if none).
62+
63+
Cores/threads-aware: the score ceiling scales with thread count so that
64+
high-core-for-their-era chips are not flagged, while a per-thread score that
65+
is implausible for the release era still is. Threads default to cores, then 1.
66+
"""
67+
findings: list[str] = []
68+
year = (rec.get("release_date") or "0")[:4]
69+
threads = rec.get("threads") or rec.get("cores") or 1
70+
name = rec.get("name", "?")
71+
pm = rec.get("passmark_cpu_mark")
72+
r23 = rec.get("cinebench_r23_multi")
73+
if year < "2006" and pm and pm > ERA_PASSMARK_PER_THREAD * threads:
74+
findings.append(
75+
f" {name!r} ({year}): passmark {pm} too high for era "
76+
f"({pm / threads:.0f}/thread over {ERA_PASSMARK_PER_THREAD}, {threads}T)"
77+
)
78+
if year < "2011" and r23 and r23 > ERA_R23_PER_THREAD * threads:
79+
findings.append(
80+
f" {name!r} ({year}): r23 {r23} too high for era "
81+
f"({r23 / threads:.0f}/thread over {ERA_R23_PER_THREAD}, {threads}T)"
82+
)
83+
return findings
84+
4685
def load(comp):
4786
recs = []
4887
for dp, _, fs in os.walk(os.path.join(ROOT, comp)):
@@ -63,89 +102,91 @@ def mad_outliers(pairs, lo=0.34, hi=3.0):
63102

64103
def section(t): print(f"\n### {t}")
65104

66-
records = {category: load(category) for category in CATEGORIES}
67-
cpus = records["cpu"]; gpus = records["gpu"]
68-
print(f"scope: {len(CATEGORIES)}/12 categories ? {sum(map(len, records.values()))} records")
69-
print(f"loaded CPU={len(cpus)} GPU={len(gpus)}")
70-
71-
# --- 1. duplicates + slug/file + verified-no-source ---
72-
section("structural")
73-
for comp, recs in records.items():
74-
slugs, names = {}, {}
75-
for p, fn, d in recs:
76-
if d.get("verified") is True and not d.get("source_urls"):
77-
hard(f" [{comp}] verified without sources: {fn}")
78-
slugs.setdefault(d.get("slug"), []).append(fn)
79-
names.setdefault(d.get("name"), []).append(fn)
80-
if d.get("slug") != fn:
81-
hard(f" [{comp}] slug!=file: {fn} slug={d.get('slug')}")
82-
for s, fl in slugs.items():
83-
if len(fl) > 1: hard(f" [{comp}] DUP slug {s}: {sorted(fl)}")
84-
for n, fl in names.items():
85-
if comp in ("cpu", "gpu") and len(fl) > 1: hard(f" [{comp}] DUP name {n!r}: {sorted(fl)}")
86-
87-
# --- 2. AMD Ryzen line vs DESKTOP model tier-digit (2nd digit); APU/mobile excepted ---
88-
section("CPU name/tier consistency (desktop mainstream only)")
89-
TIERMAP = {"6": "5", "7": "7", "8": "7", "9": "9"} # 2nd model digit -> expected line
90-
for p, fn, d in cpus:
91-
n = d.get("name", "")
92-
# mainstream desktop: 4-digit model, no G/U/H/HS/HX (APU/mobile) suffix
93-
m = re.match(r"AMD Ryzen (\d) (\d)(\d)\d\d(X3D|X|XT)?$", n)
94-
if m:
95-
line, _gen, tier = m.group(1), m.group(2), m.group(3)
96-
exp = TIERMAP.get(tier)
97-
if exp and exp != line:
98-
print(f" [tier] {n!r}: line Ryzen {line} but tier-digit {tier} → expect Ryzen {exp}")
99-
100-
# --- 3. benchmark sanity: single>multi (consistent-scale benches) ---
101-
section("CPU single>multi (cinebench/geekbench — should be multi>=single)")
102-
for p, fn, d in cpus:
103-
for s, mu in [("cinebench_r23_single","cinebench_r23_multi"),
104-
("geekbench_single","geekbench_multi"),
105-
("cinebench_2024_single","cinebench_2024_multi")]:
106-
a, b = d.get(s), d.get(mu)
107-
if a and b and a > b and (d.get("threads") or 1) > 1:
108-
hard(f" {d['name']!r}: {s}={a} > {mu}={b}")
109-
110-
# --- 4. era vs score (catch wrong-variant: old chip w/ modern score) ---
111-
section("CPU era-vs-score outliers")
112-
for p, fn, d in cpus:
113-
y = (d.get("release_date") or "0")[:4]
114-
pm = d.get("passmark_cpu_mark"); r23 = d.get("cinebench_r23_multi")
115-
if y < "2006" and pm and pm > 1500:
116-
print(f" {d['name']!r} ({y}): passmark {pm} too high for era")
117-
if y < "2011" and r23 and r23 > 3000:
118-
print(f" {d['name']!r} ({y}): r23 {r23} too high for era")
119-
120-
# --- 5. cross-source correlation outliers (KEY contamination detector) ---
121-
section("CPU cross-source ratio outliers (possible wrong-variant)")
122105
def collect(recs, fa, fb):
123106
return [(d["name"], d[fa], d[fb]) for p, fn, d in recs if d.get(fa) and d.get(fb)]
124-
for fa, fb in [("passmark_cpu_mark","cinebench_r23_multi"),
125-
("passmark_cpu_mark","geekbench_multi"),
126-
("cinebench_r23_multi","geekbench_multi"),
127-
("cinebench_2024_multi","cinebench_r23_multi")]:
128-
out = mad_outliers(collect(cpus, fa, fb))
129-
for label, ratio in out:
130-
print(f" [{fa}/{fb}] {label!r}: ratio={ratio}")
131-
132-
# --- 6. GPU cross-source + sanity ---
133-
section("GPU cross-source ratio outliers + sanity")
134-
for fa, fb in [("passmark_g3d_mark","timespy_score"),
135-
("timespy_score","blender_score"),
136-
("fp32_tflops","timespy_score"),
137-
("passmark_g3d_mark","fp32_tflops")]:
138-
for label, ratio in mad_outliers(collect(gpus, fa, fb)):
139-
print(f" [{fa}/{fb}] {label!r}: ratio={ratio}")
140-
141-
print("\n(no lines under a section = clean)")
142-
143-
if HARD_REPORT:
144-
with open(HARD_REPORT, "w", encoding="utf-8") as report:
145-
json.dump(sorted(set(HARD)), report, ensure_ascii=False, indent=2)
146-
147-
if STRICT and HARD:
148-
print(f"\n❌ integrity gate: {len(HARD)} hard anomaly(ies) — blocking refresh.")
149-
sys.exit(1)
150-
if STRICT:
151-
print("\n✅ integrity gate: no hard anomalies.")
107+
108+
def main() -> None:
109+
records = {category: load(category) for category in CATEGORIES}
110+
cpus = records["cpu"]; gpus = records["gpu"]
111+
print(f"scope: {len(CATEGORIES)}/12 categories ? {sum(map(len, records.values()))} records")
112+
print(f"loaded CPU={len(cpus)} GPU={len(gpus)}")
113+
114+
# --- 1. duplicates + slug/file + verified-no-source ---
115+
section("structural")
116+
for comp, recs in records.items():
117+
slugs, names = {}, {}
118+
for p, fn, d in recs:
119+
if d.get("verified") is True and not d.get("source_urls"):
120+
hard(f" [{comp}] verified without sources: {fn}")
121+
slugs.setdefault(d.get("slug"), []).append(fn)
122+
names.setdefault(d.get("name"), []).append(fn)
123+
if d.get("slug") != fn:
124+
hard(f" [{comp}] slug!=file: {fn} slug={d.get('slug')}")
125+
for s, fl in slugs.items():
126+
if len(fl) > 1: hard(f" [{comp}] DUP slug {s}: {sorted(fl)}")
127+
for n, fl in names.items():
128+
if comp in ("cpu", "gpu") and len(fl) > 1: hard(f" [{comp}] DUP name {n!r}: {sorted(fl)}")
129+
130+
# --- 2. AMD Ryzen line vs DESKTOP model tier-digit (2nd digit); APU/mobile excepted ---
131+
section("CPU name/tier consistency (desktop mainstream only)")
132+
TIERMAP = {"6": "5", "7": "7", "8": "7", "9": "9"} # 2nd model digit -> expected line
133+
for p, fn, d in cpus:
134+
n = d.get("name", "")
135+
# mainstream desktop: 4-digit model, no G/U/H/HS/HX (APU/mobile) suffix
136+
m = re.match(r"AMD Ryzen (\d) (\d)(\d)\d\d(X3D|X|XT)?$", n)
137+
if m:
138+
line, _gen, tier = m.group(1), m.group(2), m.group(3)
139+
exp = TIERMAP.get(tier)
140+
if exp and exp != line:
141+
print(f" [tier] {n!r}: line Ryzen {line} but tier-digit {tier} → expect Ryzen {exp}")
142+
143+
# --- 3. benchmark sanity: single>multi (consistent-scale benches) ---
144+
section("CPU single>multi (cinebench/geekbench — should be multi>=single)")
145+
for p, fn, d in cpus:
146+
for s, mu in [("cinebench_r23_single","cinebench_r23_multi"),
147+
("geekbench_single","geekbench_multi"),
148+
("cinebench_2024_single","cinebench_2024_multi")]:
149+
a, b = d.get(s), d.get(mu)
150+
if a and b and a > b and (d.get("threads") or 1) > 1:
151+
hard(f" {d['name']!r}: {s}={a} > {mu}={b}")
152+
153+
# --- 4. era vs score (catch wrong-variant: old chip w/ modern score) ---
154+
section("CPU era-vs-score outliers")
155+
for p, fn, d in cpus:
156+
for msg in era_score_outliers(d):
157+
print(msg)
158+
159+
# --- 5. cross-source correlation outliers (KEY contamination detector) ---
160+
section("CPU cross-source ratio outliers (possible wrong-variant)")
161+
for fa, fb in [("passmark_cpu_mark","cinebench_r23_multi"),
162+
("passmark_cpu_mark","geekbench_multi"),
163+
("cinebench_r23_multi","geekbench_multi"),
164+
("cinebench_2024_multi","cinebench_r23_multi")]:
165+
out = mad_outliers(collect(cpus, fa, fb))
166+
for label, ratio in out:
167+
print(f" [{fa}/{fb}] {label!r}: ratio={ratio}")
168+
169+
# --- 6. GPU cross-source + sanity ---
170+
section("GPU cross-source ratio outliers + sanity")
171+
for fa, fb in [("passmark_g3d_mark","timespy_score"),
172+
("timespy_score","blender_score"),
173+
("fp32_tflops","timespy_score"),
174+
("passmark_g3d_mark","fp32_tflops")]:
175+
for label, ratio in mad_outliers(collect(gpus, fa, fb)):
176+
print(f" [{fa}/{fb}] {label!r}: ratio={ratio}")
177+
178+
print("\n(no lines under a section = clean)")
179+
180+
if HARD_REPORT:
181+
with open(HARD_REPORT, "w", encoding="utf-8") as report:
182+
json.dump(sorted(set(HARD)), report, ensure_ascii=False, indent=2)
183+
184+
if STRICT and HARD:
185+
print(f"\n❌ integrity gate: {len(HARD)} hard anomaly(ies) — blocking refresh.")
186+
sys.exit(1)
187+
if STRICT:
188+
print("\n✅ integrity gate: no hard anomalies.")
189+
190+
191+
if __name__ == "__main__":
192+
main()

‎tests/unit/test_integrity_era.py‎

Lines changed: 130 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,130 @@
1+
"""Unit tests for the cores/threads-aware era-vs-score rule in ``integrity_check``.
2+
3+
The era rule flags a wrong-variant contamination signal: an old chip carrying a
4+
score that belongs to a newer part. The original rule used a flat per-chip ceiling
5+
(PassMark>1500 before 2006, R23>3000 before 2011) and ignored core/thread count,
6+
which mis-fired on legitimate pre-2011 high-core enthusiast parts. These tests pin
7+
the fix: those known-good chips must not flag, while a genuinely implausible
8+
year/score combination still must.
9+
"""
10+
11+
from __future__ import annotations
12+
13+
import integrity_check as ic
14+
15+
16+
def _rec(**overrides: object) -> dict:
17+
base: dict = dict(
18+
name="Test CPU",
19+
release_date="2010-01-01",
20+
cores=6,
21+
threads=12,
22+
)
23+
base.update(overrides)
24+
return base
25+
26+
27+
# The 8 pre-2011 high-core-for-their-era chips flagged as false positives during
28+
# the 2026-09-29 CPU advisory re-check (TechEngine #98). Real cores/threads and the
29+
# stored Cinebench R23 multi scores; every one must be treated as plausible.
30+
KNOWN_GOOD_ERA_CHIPS = [
31+
("Intel Core i7-920 (Bloomfield)", "2008", 4, 8, 3800),
32+
("Intel Core i7-965 Extreme Edition (Bloomfield)", "2008", 4, 8, 4500),
33+
("Intel Core i7-870 (Lynnfield)", "2009", 4, 8, 4200),
34+
("Intel Core i7-970", "2010", 6, 12, 5800),
35+
("Intel Core i7-980X Extreme Edition (Gulftown)", "2010", 6, 12, 6500),
36+
("Intel Core i7-990X Extreme Edition", "2010", 6, 12, 6300),
37+
("AMD Phenom II X6 1090T Black Edition (Thuban)", "2010", 6, 6, 3400),
38+
("AMD Phenom II X6 1100T Black Edition (Thuban)", "2010", 6, 6, 3500),
39+
]
40+
41+
42+
def test_known_good_pre2011_high_core_chips_are_not_flagged() -> None:
43+
for name, year, cores, threads, r23 in KNOWN_GOOD_ERA_CHIPS:
44+
rec = _rec(
45+
name=name,
46+
release_date=f"{year}-06-01",
47+
cores=cores,
48+
threads=threads,
49+
cinebench_r23_multi=r23,
50+
)
51+
assert ic.era_score_outliers(rec) == [], f"{name} should not be flagged"
52+
53+
54+
def test_genuinely_implausible_old_chip_still_flags() -> None:
55+
# A 2-core/2009 part claiming a 20,000 R23 multi (10,000/thread) is impossible
56+
# for the era and must still be flagged.
57+
rec = _rec(
58+
name="Bogus Dual-Core 2009",
59+
release_date="2009-01-01",
60+
cores=2,
61+
threads=2,
62+
cinebench_r23_multi=20000,
63+
)
64+
findings = ic.era_score_outliers(rec)
65+
assert len(findings) == 1
66+
assert "r23 20000 too high for era" in findings[0]
67+
68+
69+
def test_r23_ceiling_scales_with_threads() -> None:
70+
# Same per-thread score, different thread counts: a high absolute R23 that is
71+
# reasonable per-thread for a many-threaded part is fine, but the identical
72+
# per-thread rate on very few threads is not what trips the rule -- the rule
73+
# is about total score vs thread budget. Just above / below the per-thread
74+
# ceiling around the boundary.
75+
threads = 12
76+
ceiling = ic.ERA_R23_PER_THREAD * threads
77+
below = _rec(threads=threads, cinebench_r23_multi=ceiling - 1)
78+
above = _rec(threads=threads, cinebench_r23_multi=ceiling + 1)
79+
assert ic.era_score_outliers(below) == []
80+
assert len(ic.era_score_outliers(above)) == 1
81+
82+
83+
def test_threads_default_to_cores_then_one() -> None:
84+
# threads missing -> falls back to cores
85+
rec_cores = _rec(threads=None, cores=6, cinebench_r23_multi=6 * ic.ERA_R23_PER_THREAD + 1)
86+
assert len(ic.era_score_outliers(rec_cores)) == 1
87+
# both missing -> falls back to 1 thread
88+
rec_one = dict(name="No core info", release_date="2010-01-01",
89+
cinebench_r23_multi=ic.ERA_R23_PER_THREAD + 1)
90+
assert len(ic.era_score_outliers(rec_one)) == 1
91+
92+
93+
def test_modern_chips_are_never_flagged_regardless_of_score() -> None:
94+
# The year gate means post-2011 parts are out of scope entirely.
95+
rec = _rec(
96+
name="AMD Ryzen 9 9950X",
97+
release_date="2024-08-15",
98+
cores=16,
99+
threads=32,
100+
cinebench_r23_multi=42000,
101+
passmark_cpu_mark=65756,
102+
)
103+
assert ic.era_score_outliers(rec) == []
104+
105+
106+
def test_passmark_era_rule_is_thread_aware() -> None:
107+
# Pre-2006 PassMark ceiling also scales by threads. A single-core 2004 chip
108+
# with an era-appropriate PassMark is fine; an absurd one still flags.
109+
ok = _rec(name="Pentium 4 (2004)", release_date="2004-06-01", cores=1, threads=1,
110+
passmark_cpu_mark=ic.ERA_PASSMARK_PER_THREAD - 1)
111+
bad = _rec(name="Pentium 4 (2004)", release_date="2004-06-01", cores=1, threads=1,
112+
passmark_cpu_mark=ic.ERA_PASSMARK_PER_THREAD * 10)
113+
assert ic.era_score_outliers(ok) == []
114+
findings = ic.era_score_outliers(bad)
115+
assert len(findings) == 1
116+
assert "passmark" in findings[0]
117+
118+
119+
def test_missing_scores_never_flag() -> None:
120+
assert ic.era_score_outliers(_rec(cinebench_r23_multi=None, passmark_cpu_mark=None)) == []
121+
assert ic.era_score_outliers(_rec()) == [] # no score fields at all
122+
123+
124+
def test_importing_integrity_check_runs_no_scan(capsys) -> None:
125+
# The module must be importable without triggering the filesystem scan
126+
# (guarded by ``if __name__ == '__main__'``). Re-importing is a no-op; assert
127+
# the pure helper is present and callable.
128+
assert callable(ic.era_score_outliers)
129+
captured = capsys.readouterr()
130+
assert "scope:" not in captured.out

0 commit comments

Comments
 (0)