Skip to content

Commit 0044ab2

Browse files
committed
perf(integrity): scope the PR integrity gate to changed records
`integrity_check.py --only FILE` checks just the listed data-relative paths: per-record hard anomalies, plus duplicate slug/name against the whole catalog via file names (and a full cpu/gpu name index). The population-based advisory sections are skipped when scoped. The PR validation comment runs it on head and base with the same changed-path list; the weekly refresh still runs the full scan. Refs GetTechAPI/TechAPI#350
1 parent 890d3d9 commit 0044ab2

2 files changed

Lines changed: 48 additions & 6 deletions

File tree

‎.github/workflows/techapi-pr-validation-comment.yml‎

Lines changed: 6 additions & 3 deletions
Original file line numberDiff line numberDiff line change
@@ -169,12 +169,15 @@ jobs:
169169
} > validation.log 2>&1
170170
app_status=$(grep "app_validate_status=" validation.log | tail -n 1 | cut -d= -f2)
171171
172+
# Scope the integrity scan to the records this PR touches (hard anomalies
173+
# are per-record + duplicate checks); the same list is used on the base.
174+
git -C TechAPI diff --name-only --diff-filter=ACMR "$TECHAPI_DIFF_BASE...HEAD" -- data/ | sed 's|^data/||' | grep '\.json$' > changed-paths.txt || true
172175
{
173176
echo
174-
echo "## integrity_check.py (PR head compared with PR base)"
175-
python integrity_check.py TechAPI/data --hard-report head-hard.json
177+
echo "## integrity_check.py (PR head compared with PR base, changed records only)"
178+
python integrity_check.py TechAPI/data --only changed-paths.txt --hard-report head-hard.json
176179
echo "head_integrity_scan_status=$?"
177-
python integrity_check.py TechAPI-base/data --hard-report base-hard.json > baseline-integrity.log
180+
python integrity_check.py TechAPI-base/data --only changed-paths.txt --hard-report base-hard.json > baseline-integrity.log
178181
echo "base_integrity_scan_status=$?"
179182
} >> validation.log 2>&1
180183
integrity_status=$(python - <<'PY'

‎integrity_check.py‎

Lines changed: 42 additions & 3 deletions
Original file line numberDiff line numberDiff line change
@@ -37,6 +37,13 @@
3737
HARD_REPORT = _argv[_report_index + 1] if _report_index >= 0 else None
3838
if _report_index >= 0:
3939
del _argv[_report_index:_report_index + 2]
40+
# --only FILE: scope to the data-relative paths listed in FILE (one per line).
41+
# Only the per-record hard checks (+ duplicate slug/name) run; the population-based
42+
# advisory sections need the whole catalog and are skipped.
43+
_only_index = _argv.index("--only") if "--only" in _argv else -1
44+
ONLY_FILE = _argv[_only_index + 1] if _only_index >= 0 else None
45+
if _only_index >= 0:
46+
del _argv[_only_index:_only_index + 2]
4047
_positional = [a for a in _argv if not a.startswith("-")]
4148
ROOT = _positional[0] if _positional else r"C:\Users\29\Desktop\TechAPI\data"
4249

@@ -85,8 +92,20 @@ def era_score_outliers(rec: dict) -> list[str]:
8592
)
8693
return findings
8794

88-
def load(comp):
95+
ONLY: set[str] | None = None
96+
if ONLY_FILE:
97+
with open(ONLY_FILE, encoding="utf-8") as _f:
98+
ONLY = {ln.strip() for ln in _f if ln.strip()}
99+
100+
101+
def load(comp, full=False):
89102
recs = []
103+
if ONLY is not None and not full:
104+
for rel in sorted(ONLY):
105+
p = os.path.join(ROOT, rel)
106+
if rel.startswith(comp + "/") and os.path.isfile(p) and not os.path.basename(p).startswith("_"):
107+
recs.append((p, os.path.basename(p)[:-5], json.load(open(p, encoding="utf-8"))))
108+
return recs
90109
for dp, _, fs in os.walk(os.path.join(ROOT, comp)):
91110
for fn in fs:
92111
if fn.endswith(".json") and not fn.startswith("_"):
@@ -179,6 +198,7 @@ def collect_cpu(recs, fa, fb):
179198
def main() -> None:
180199
records = {category: load(category) for category in CATEGORIES}
181200
cpus = records["cpu"]; gpus = records["gpu"]
201+
scoped = ONLY is not None
182202
print(f"scope: {len(CATEGORIES)}/12 categories ? {sum(map(len, records.values()))} records")
183203
print(f"loaded CPU={len(cpus)} GPU={len(gpus)}")
184204

@@ -197,6 +217,25 @@ def main() -> None:
197217
if len(fl) > 1: hard(f" [{comp}] DUP slug {s}: {sorted(fl)}")
198218
for n, fl in names.items():
199219
if comp in ("cpu", "gpu") and len(fl) > 1: hard(f" [{comp}] DUP name {n!r}: {sorted(fl)}")
220+
if scoped and recs:
221+
# Changed records vs the rest of the catalog. File names equal slugs by
222+
# convention, so listing names finds a slug clash without parsing.
223+
changed_fn = {fn for _, fn, _ in recs}
224+
stems = {}
225+
for dp, _, fs in os.walk(os.path.join(ROOT, comp)):
226+
for f in fs:
227+
if f.endswith(".json") and not f.startswith("_"):
228+
stems.setdefault(f[:-5], []).append(os.path.join(dp, f))
229+
for s, fl in slugs.items():
230+
if len(stems.get(s, [])) > len(fl):
231+
hard(f" [{comp}] DUP slug {s}: {sorted(os.path.basename(x) for x in stems[s])}")
232+
if comp in ("cpu", "gpu") and names:
233+
known = {}
234+
for p, fn, d in load(comp, full=True):
235+
known.setdefault(d.get("name"), []).append(fn)
236+
for n in names:
237+
if len(known.get(n, [])) > len(names[n]):
238+
hard(f" [{comp}] DUP name {n!r}: {sorted(known[n])}")
200239

201240
# --- 2. AMD Ryzen line vs DESKTOP model tier-digit (2nd digit); APU/mobile excepted ---
202241
section("CPU name/tier consistency (desktop mainstream only)")
@@ -233,7 +272,7 @@ def main() -> None:
233272
# each part is judged against the ratio expected for its own core count
234273
# rather than a desktop-dominated global median.
235274
section("CPU cross-source ratio outliers (possible wrong-variant)")
236-
for fa, fb in [("passmark_cpu_mark","cinebench_r23_multi"),
275+
for fa, fb in [] if scoped else [("passmark_cpu_mark","cinebench_r23_multi"),
237276
("passmark_cpu_mark","geekbench_multi"),
238277
("cinebench_r23_multi","geekbench_multi"),
239278
("cinebench_2024_multi","cinebench_r23_multi")]:
@@ -248,7 +287,7 @@ def main() -> None:
248287
# eras, so there is no single clean stratifying variable. These stay
249288
# advisory-only and are surfaced for human review rather than gated.
250289
section("GPU cross-source ratio outliers + sanity")
251-
for fa, fb in [("passmark_g3d_mark","timespy_score"),
290+
for fa, fb in [] if scoped else [("passmark_g3d_mark","timespy_score"),
252291
("timespy_score","blender_score"),
253292
("fp32_tflops","timespy_score"),
254293
("passmark_g3d_mark","fp32_tflops")]:

0 commit comments

Comments
 (0)