From c313e0057955a7b84f6f79ff98e47ef537295ec7 Mon Sep 17 00:00:00 2001 From: Seungpyo1007 Date: Tue, 29 Sep 2026 22:43:09 +0900 Subject: [PATCH] fix(verify): make cross-source ratio outliers cores-aware MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Audit follow-up to #111 (cores-aware era rule): swept every rule in integrity_check.py for the same shape of bug — a value compared against a fixed reference that ignores a variable which legitimately shifts what "normal" looks like — and found one more instance. The CPU cross-source ratio detector ran a single global median±MAD over the whole catalog. But ratios like cinebench_r23_multi/geekbench_multi are confounded by core count: R23 multi scales near-linearly with cores while Geekbench multicore compresses, so the ratio climbs monotonically with thread count (measured on live data: ~1.05 at 1-4T up to ~1.48 at 65T+, Pearson corr(threads, log-ratio) +0.52; PassMark/R23 falls, corr -0.59). A global median therefore flagged entire legitimate core-count strata — 90 of 739 R23/GB pairs, median 56 threads vs 16 overall, the whole EPYC/Threadripper/ Xeon many-core cluster with genuine scores — as contamination. Same shape as the flat era ceiling. Fix mirrors #111 (which divided score by threads): regress the confounder out. mad_outliers now accepts an optional per-part covariate, fits a robust Theil-Sen line of log-ratio vs log(covariate), and runs the median±MAD test on the residuals, so each part is judged against the ratio expected for its own core count. CPU pairs pass thread count; the systematic core-count gradient no longer flags while a part anomalous for its own class still does (live R23/GB flags 90 -> 10, survivors are genuine per-class outliers). Callers without a covariate (GPUs) are unchanged. Coarse thread-banding was rejected: it removes the between-band trend but shrinks the within-band envelope, netting more false positives. The GPU cross-source ratios were checked and left as a single population on purpose: they mix a theoretical spec (fp32_tflops) with empirical benchmarks across gaming vs. compute cards and many hardware eras, with no single clean stratifying variable, and stay advisory-only. Adds tests/unit/test_integrity_cross_source.py. Refs #98 --- integrity_check.py | 98 +++++++++++++++-- tests/unit/test_integrity_cross_source.py | 123 ++++++++++++++++++++++ 2 files changed, 212 insertions(+), 9 deletions(-) create mode 100644 tests/unit/test_integrity_cross_source.py diff --git a/integrity_check.py b/integrity_check.py index 0dfe7e3..30ee2e3 100644 --- a/integrity_check.py +++ b/integrity_check.py @@ -15,7 +15,10 @@ slug/filename mismatches, and physically-impossible single>multi benchmarks. The statistical cross-source/era outliers stay advisory (a heterogeneous catalog of server + desktop + mobile parts legitimately produces many ratio outliers), so -they are printed for review but never fail the gate. +they are printed for review but never fail the gate. The CPU cross-source ratio +check regresses out the core-count trend so it compares each part against the +ratio expected for its own core count instead of a desktop-dominated global +median (see mad_outliers). ``--hard-report`` writes a JSON list of hard anomalies for baseline comparison. """ from __future__ import annotations @@ -91,20 +94,88 @@ def load(comp): recs.append((p, fn[:-5], json.load(open(p, encoding="utf-8")))) return recs -def mad_outliers(pairs, lo=0.34, hi=3.0): - """pairs: list of (label, a, b); flag log(a/b) outliers via median±3*MAD.""" - rs = [(l, math.log(a / b)) for l, a, b in pairs if a and b] - if len(rs) < 8: +# Cross-source ratio outliers: same wrong-variant detector as the era rule, but +# statistical. The original computed ONE global median±MAD over the whole CPU +# catalog for ratios like cinebench_r23_multi/geekbench_multi. That ratio is not +# scale-free across the catalog — it is confounded by core/thread count, because +# the two benchmarks scale differently with parallelism: Cinebench R23 multi +# scales near-linearly with cores while Geekbench multicore compresses at high +# core counts, so the R23/GB ratio climbs monotonically with thread count +# (measured on live data: ~1.05 at 1-4T rising to ~1.48 at 65T+, Pearson +# corr(threads, log-ratio) ≈ +0.52; the PassMark/R23 ratio falls with threads, +# corr ≈ -0.59). A single global median therefore flags entire legitimate +# core-count strata — every many-core EPYC/Threadripper/Xeon and, at the other +# end, the low-core parts — as "contamination" (90 of 739 R23/GB pairs, median +# 56 threads vs 16 overall, all with genuine scores). That is the same shape of +# bug as the flat era ceiling: a fixed reference blind to a variable that +# legitimately shifts what "normal" looks like. +# +# Fix (mirrors PR #111's cores-aware era rule, which divided the score by thread +# count): regress the confounder out. When a per-part covariate is supplied we +# fit a robust (Theil–Sen) line of log-ratio against log(covariate) and run the +# median±MAD test on the *residuals*, so a part is measured against the ratio +# expected for its own core count rather than a desktop-dominated global median. +# The systematic core-count gradient no longer flags; a part whose ratio is +# anomalous for its own class — the real wrong-variant signal — still does +# (live R23/GB flags drop 90 → 10, and the survivors are genuine per-class +# outliers). Coarse banding was rejected: it removes the between-band trend but +# shrinks the within-band MAD envelope, netting *more* false positives. +# Callers with no meaningful covariate (e.g. GPUs) omit it and get the original +# single-population behaviour unchanged. +def _theil_sen(xs: list[float], ys: list[float]) -> tuple[float, float]: + """Robust slope/intercept via median of pairwise slopes (Theil–Sen).""" + slopes = [ + (ys[j] - ys[i]) / (xs[j] - xs[i]) + for i in range(len(xs)) for j in range(i + 1, len(xs)) + if xs[j] != xs[i] + ] + slope = statistics.median(slopes) if slopes else 0.0 + intercept = statistics.median(y - slope * x for x, y in zip(xs, ys, strict=True)) + return slope, intercept + +def mad_outliers(pairs): + """Flag log(a/b) outliers via median±4*MAD. + + ``pairs`` is a list of ``(label, a, b)`` or ``(label, a, b, covariate)``. + With a positive per-part ``covariate`` (e.g. thread count) the log-ratio is + first detrended against ``log(covariate)`` with a robust Theil–Sen fit and + the outlier test runs on the residuals, so a variable that legitimately + shifts the ratio does not turn a whole stratum into false positives. Fewer + than 8 usable points returns nothing — too few to estimate a robust + median/MAD. Omitting the covariate reproduces the original global test. + """ + rows: list[tuple[str, float, float | None]] = [] + for item in pairs: + label, a, b = item[0], item[1], item[2] + cov = item[3] if len(item) > 3 else None + if a and b: + rows.append((label, math.log(a / b), cov)) + if len(rows) < 8: return [] - med = statistics.median(r for _, r in rs) - mad = statistics.median(abs(r - med) for _, r in rs) or 1e-9 - return [(l, round(math.exp(r), 2)) for l, r in rs if abs(r - med) > 4 * mad] + ys = [r for _, r, _ in rows] + xs = [math.log(c) for _, _, c in rows if c and c > 0] + if len(xs) == len(rows) and len({round(x, 9) for x in xs}) > 1: + slope, intercept = _theil_sen(xs, ys) + scores = [y - (slope * x + intercept) for x, y in zip(xs, ys, strict=True)] + else: + scores = ys # no usable covariate -> original global behaviour + med = statistics.median(scores) + mad = statistics.median(abs(s - med) for s in scores) or 1e-9 + return [ + (rows[i][0], round(math.exp(ys[i]), 2)) + for i in range(len(rows)) if abs(scores[i] - med) > 4 * mad + ] def section(t): print(f"\n### {t}") def collect(recs, fa, fb): return [(d["name"], d[fa], d[fb]) for p, fn, d in recs if d.get(fa) and d.get(fb)] +def collect_cpu(recs, fa, fb): + """Like ``collect`` but tags each pair with its thread-count covariate.""" + return [(d["name"], d[fa], d[fb], d.get("threads") or d.get("cores") or 1) + for p, fn, d in recs if d.get(fa) and d.get(fb)] + def main() -> None: records = {category: load(category) for category in CATEGORIES} cpus = records["cpu"]; gpus = records["gpu"] @@ -157,16 +228,25 @@ def main() -> None: print(msg) # --- 5. cross-source correlation outliers (KEY contamination detector) --- + # Thread-count-aware (see mad_outliers): the ratio between two CPU benchmarks + # is confounded by core count, so the core-count trend is regressed out and + # each part is judged against the ratio expected for its own core count + # rather than a desktop-dominated global median. section("CPU cross-source ratio outliers (possible wrong-variant)") for fa, fb in [("passmark_cpu_mark","cinebench_r23_multi"), ("passmark_cpu_mark","geekbench_multi"), ("cinebench_r23_multi","geekbench_multi"), ("cinebench_2024_multi","cinebench_r23_multi")]: - out = mad_outliers(collect(cpus, fa, fb)) + out = mad_outliers(collect_cpu(cpus, fa, fb)) for label, ratio in out: print(f" [{fa}/{fb}] {label!r}: ratio={ratio}") # --- 6. GPU cross-source + sanity --- + # Left as a single population on purpose: unlike the CPU thread-count + # confounder, the GPU ratios mix a theoretical spec (fp32_tflops) with + # empirical benchmarks across gaming vs. compute cards and many hardware + # eras, so there is no single clean stratifying variable. These stay + # advisory-only and are surfaced for human review rather than gated. section("GPU cross-source ratio outliers + sanity") for fa, fb in [("passmark_g3d_mark","timespy_score"), ("timespy_score","blender_score"), diff --git a/tests/unit/test_integrity_cross_source.py b/tests/unit/test_integrity_cross_source.py new file mode 100644 index 0000000..d56b70a --- /dev/null +++ b/tests/unit/test_integrity_cross_source.py @@ -0,0 +1,123 @@ +"""Unit tests for the core-count-aware cross-source ratio outlier detector in +``integrity_check``. + +Cross-source ratios (e.g. cinebench_r23_multi / geekbench_multi) are the key +wrong-variant contamination signal, but the raw ratio is confounded by core +count: the two benchmarks scale differently with parallelism, so the ratio drifts +monotonically with thread count. The original single global median±MAD therefore +flagged entire legitimate core-count strata (every many-core EPYC/Threadripper/ +Xeon) as outliers — the same shape of bug as the flat era ceiling that PR #111 +fixed. The fix regresses the core-count trend out and runs the outlier test on +the residuals. These tests pin that behaviour: a whole core-count band that only +follows the natural trend must not flag, while a part that is anomalous *for its +own core count* still must, and callers without a covariate keep the original +global behaviour. +""" + +from __future__ import annotations + +import math + +import integrity_check as ic + + +def _clean_pair_at( + threads: int, trend_slope: float = 0.11, base: float = 0.05 +) -> tuple[float, float]: + """Build an (a, b) pair whose log-ratio sits exactly on the core-count trend. + + log(a/b) = base + trend_slope * log(threads); b fixed at 1000. + """ + log_ratio = base + trend_slope * math.log(threads) + b = 1000.0 + a = b * math.exp(log_ratio) + return a, b + + +# A realistic thread-count ladder spanning desktop -> HEDT -> server, each part's +# ratio lying on the same gentle upward core-count trend (the confounder). The +# population is desktop-dominated (like the real catalog: median ~16 threads), so +# a global median is pulled toward the low-core parts and the legitimate high-core +# trend-followers look like outliers to it. +TREND_THREADS = ( + [4] * 6 + [8] * 8 + [12] * 10 + [16] * 12 + [24] * 6 + [32] * 4 + + [48, 64, 96, 128, 192, 256] +) + + +def test_pairs_on_the_core_count_trend_do_not_flag() -> None: + # Every pair follows the same log-ratio-vs-log-threads line, so after the + # trend is regressed out the residuals are ~0 and nothing is an outlier -- + # even though the raw ratios span a wide range (the old global test flagged + # the extremes). + pairs = [ + (f"part-{t}-{i}", *_clean_pair_at(t), t) + for i, t in enumerate(TREND_THREADS) + ] + assert ic.mad_outliers(pairs) == [] + + +def test_raw_global_test_would_have_flagged_the_trend_extremes() -> None: + # Guard/contrast: the SAME data, fed WITHOUT the covariate, reproduces the + # old behaviour and flags the high-core extremes. This documents that the fix + # -- not a change in the data -- is what removes the false positives. + pairs_no_cov = [(f"part-{t}-{i}", *_clean_pair_at(t)) for i, t in enumerate(TREND_THREADS)] + flagged = [label for label, _ in ic.mad_outliers(pairs_no_cov)] + # The many-core trend-followers (which are perfectly legitimate) are what the + # covariate-free test wrongly flags. + assert flagged, "global (covariate-free) test should still flag trend extremes" + assert any("part-256" in f or "part-192" in f for f in flagged) + + +def test_part_anomalous_for_its_own_core_count_still_flags() -> None: + # A part whose ratio is far off the trend line *for its thread count* is a + # genuine wrong-variant candidate and must survive the detrending. + pairs = [ + (f"part-{t}-{i}", *_clean_pair_at(t), t) + for i, t in enumerate(TREND_THREADS) + ] + # Inject a 32-thread part with a wildly wrong ratio (e.g. a swapped variant): + bogus_a, bogus_b = _clean_pair_at(32) + pairs.append(("Swapped-variant 32T", bogus_a * 4.0, bogus_b, 32)) + flagged = [label for label, _ in ic.mad_outliers(pairs)] + assert "Swapped-variant 32T" in flagged + + +def test_missing_covariate_recovers_global_behaviour() -> None: + # Three-tuples (no covariate) must behave exactly like the original detector: + # a uniform cluster with one gross outlier flags only the outlier. + pairs = [(f"p{i}", 1000.0 + i, 1000.0) for i in range(20)] + pairs.append(("gross", 100000.0, 1000.0)) + flagged = [label for label, _ in ic.mad_outliers(pairs)] + assert flagged == ["gross"] + + +def test_fewer_than_eight_points_never_flags() -> None: + pairs = [(f"p{i}", 1000.0 * (i + 1), 1000.0, 8) for i in range(7)] + assert ic.mad_outliers(pairs) == [] + + +def test_zero_or_missing_values_are_skipped() -> None: + # a or b of 0/None must not raise and must be dropped from the population. + pairs = [(f"p{i}", 1000.0, 1000.0, 8) for i in range(10)] + pairs += [("zero-a", 0, 1000.0, 8), ("none-b", 1000.0, None, 8)] + # No real outlier among the valid points -> empty, and no exception. + assert ic.mad_outliers(pairs) == [] + + +def test_returned_ratio_is_the_raw_ratio_not_the_residual() -> None: + # The reported number must remain the human-readable a/b ratio so reviewers + # can eyeball it, even though the test runs on residuals. + pairs = [(f"p{i}", *_clean_pair_at(t), t) for i, t in enumerate(TREND_THREADS)] + a, b = _clean_pair_at(32) + pairs.append(("Swapped-variant 32T", a * 4.0, b, 32)) + result = dict(ic.mad_outliers(pairs)) + assert math.isclose(result["Swapped-variant 32T"], round((a * 4.0) / b, 2), rel_tol=1e-6) + + +def test_theil_sen_recovers_a_known_slope() -> None: + xs = [math.log(t) for t in (1, 2, 4, 8, 16, 32, 64)] + ys = [0.5 + 0.3 * x for x in xs] # perfect line, slope 0.3 + slope, intercept = ic._theil_sen(xs, ys) + assert math.isclose(slope, 0.3, rel_tol=1e-9) + assert math.isclose(intercept, 0.5, rel_tol=1e-9)