From 2fff3f3b928baeb5cd31eee5b9624957a5bf7bfd Mon Sep 17 00:00:00 2001 From: Abhishek Shivakumar Date: Mon, 17 Aug 2026 12:09:39 +0100 Subject: [PATCH] Track benchmark disparity history and restore data-driven plots --- .github/workflows/pages.yml | 6 +- .github/workflows/remaining-issues.yml | 31 +++++- benchmarks/check_disparity_regression.py | 105 ++++++++++++++++++++ benchmarks/disparity_regression_policy.json | 20 ++++ benchmarks/publish_headline_v2.py | 60 +++++++---- docs/benchmarks.html | 2 +- docs/disparity-history.html | 1 + 7 files changed, 201 insertions(+), 24 deletions(-) create mode 100644 benchmarks/check_disparity_regression.py create mode 100644 benchmarks/disparity_regression_policy.json create mode 100644 docs/disparity-history.html diff --git a/.github/workflows/pages.yml b/.github/workflows/pages.yml index 59e1851..3b07301 100644 --- a/.github/workflows/pages.yml +++ b/.github/workflows/pages.yml @@ -47,7 +47,11 @@ jobs: run: python tools/build_browser_wasm.py --flow-repo .flow-toolchain - name: Smoke direct browser WebAssembly API run: node tools/smoke_browser_wasm.mjs docs/wasm/browser_mlir.wasm - - name: Generate canonical v2 benchmark headline + - name: Regenerate canonical disparity evidence + run: python benchmarks/generate_disparity_report.py + - name: Materialize disparity history for this Pages build + run: python benchmarks/check_disparity_regression.py --baseline /tmp/no-disparity-baseline.json --commit "${GITHUB_SHA}" + - name: Publish canonical benchmark, disparity, history and architecture evidence run: python benchmarks/publish_headline_v2.py - name: Normalize legacy sklearn timings to milliseconds run: python benchmarks/normalize_published_timings.py diff --git a/.github/workflows/remaining-issues.yml b/.github/workflows/remaining-issues.yml index 8b3e8df..20f43cb 100644 --- a/.github/workflows/remaining-issues.yml +++ b/.github/workflows/remaining-issues.yml @@ -14,6 +14,11 @@ jobs: runs-on: ubuntu-latest steps: - uses: actions/checkout@v4 + - name: Preserve frozen disparity baseline + run: | + if [ -f benchmarks/disparity_report.json ]; then + cp benchmarks/disparity_report.json /tmp/disparity_report.baseline.json + fi - uses: actions/checkout@v4 with: repository: flooooooooooow/flow @@ -38,18 +43,35 @@ jobs: run: python benchmarks/run_headline.py --repeats 3 - name: Apply evidence-backed KMeans equivalence gate run: python benchmarks/finalize_kmeans_parity.py - - name: Require all 19 rows parity eligible + - name: Generate persistent disparity report + run: | + cp benchmarks/headline_summary.json benchmarks/headline_result_v2.json + python benchmarks/generate_disparity_report.py + - name: Gate disparity regressions and append history + run: | + if [ -f /tmp/disparity_report.baseline.json ]; then + python benchmarks/check_disparity_regression.py --baseline /tmp/disparity_report.baseline.json + else + python benchmarks/check_disparity_regression.py --baseline /tmp/no-disparity-baseline.json + fi + - name: Require all 19 rows parity eligible and disparities retained run: | python - <<'PY' import json from pathlib import Path result = json.loads(Path('benchmarks/headline_summary.json').read_text()) + disparity = json.loads(Path('benchmarks/disparity_report.json').read_text()) + history = json.loads(Path('benchmarks/disparity_history.json').read_text()) c = result['counts'] assert c['total_rows'] == 19, c assert c['eligible_comparisons'] == 19, c assert c['parity_unresolved'] == 0, c assert c['measurement_unresolved'] == 0, c - print(c) + assert disparity['counts']['rows'] == 19, disparity['counts'] + assert disparity['counts']['rows_with_tracked_disparity'] > 0, disparity['counts'] + assert disparity['counts']['strict_final_status_disagreements'] >= 1, disparity['counts'] + assert history['snapshots'], history + print({'headline': c, 'disparity': disparity['counts'], 'history_snapshots': len(history['snapshots'])}) PY - uses: actions/upload-artifact@v4 with: @@ -60,7 +82,10 @@ jobs: benchmarks/flow_results_v2.txt benchmarks/headline_rows.json benchmarks/headline_summary.json + benchmarks/headline_result_v2.json benchmarks/parity_diagnostics.json + benchmarks/disparity_report.json + benchmarks/disparity_history.json benchmarks/headline_environment.json architecture-map: @@ -149,5 +174,5 @@ jobs: echo 'Generated evidence already current.' exit 0 fi - git commit -m "Freeze validated sklearn architecture and 19/19 benchmark evidence [skip ci]" + git commit -m "Freeze validated benchmark, disparity history, and architecture evidence [skip ci]" git push origin HEAD:main diff --git a/benchmarks/check_disparity_regression.py b/benchmarks/check_disparity_regression.py new file mode 100644 index 0000000..e263818 --- /dev/null +++ b/benchmarks/check_disparity_regression.py @@ -0,0 +1,105 @@ +#!/usr/bin/env python3 +from __future__ import annotations + +import argparse +import json +import os +from pathlib import Path + +ROOT = Path(__file__).resolve().parent + + +def key(row: dict) -> tuple[str, str, str]: + return row["algorithm"], row["dataset"], row["metric"] + + +def main() -> int: + p = argparse.ArgumentParser() + p.add_argument("--current", type=Path, default=ROOT / "disparity_report.json") + p.add_argument("--baseline", type=Path, default=ROOT / "disparity_report.baseline.json") + p.add_argument("--history", type=Path, default=ROOT / "disparity_history.json") + p.add_argument("--policy", type=Path, default=ROOT / "disparity_regression_policy.json") + p.add_argument("--commit", default=os.environ.get("GITHUB_SHA", "unknown")) + args = p.parse_args() + + current = json.loads(args.current.read_text()) + policy = json.loads(args.policy.read_text()) + failures: list[str] = [] + + baseline = None + if args.baseline.exists(): + baseline = json.loads(args.baseline.read_text()) + old = {key(r): r for r in baseline["rows"]} + same_env = baseline.get("environment_id") == current.get("environment_id") + for row in current["rows"]: + previous = old.get(key(row)) + if previous is None: + continue + + old_frac = previous.get("score_tolerance_fraction") + new_frac = row.get("score_tolerance_fraction") + if old_frac is not None and new_frac is not None: + limit = float(policy["score_tolerance_fraction"]["max_absolute_increase"]) + if new_frac - old_frac > limit: + failures.append(f"{key(row)} tolerance fraction regressed {old_frac:.6g} -> {new_frac:.6g}") + + old_abs = previous.get("score_abs_diff") + new_abs = row.get("score_abs_diff") + if old_abs is not None and new_abs is not None: + floor = float(policy["score_abs_diff"]["absolute_noise_floor"]) + rel = float(policy["score_abs_diff"]["max_relative_increase"]) + if new_abs > max(floor, old_abs * (1.0 + rel)): + failures.append(f"{key(row)} score |delta| regressed {old_abs:.6g} -> {new_abs:.6g}") + + for field, policy_name in (("configuration_differences", "configuration_difference_count"), ("semantic_differences", "semantic_difference_count")): + old_n = len(previous.get(field, [])) + new_n = len(row.get(field, [])) + if new_n - old_n > int(policy[policy_name]["max_increase"]): + failures.append(f"{key(row)} {field} increased {old_n} -> {new_n}") + + if same_env or not policy["runtime_log2_ratio"].get("same_environment_only", True): + old_rt = previous.get("runtime_log2_ratio") + new_rt = row.get("runtime_log2_ratio") + if old_rt is not None and new_rt is not None: + limit = float(policy["runtime_log2_ratio"]["max_absolute_change"]) + if abs(new_rt - old_rt) > limit: + failures.append(f"{key(row)} runtime log2 ratio changed {old_rt:.4f} -> {new_rt:.4f}") + + history = {"schema_version": 1, "snapshots": []} + if args.history.exists(): + history = json.loads(args.history.read_text()) + compact_rows = [] + for row in current["rows"]: + compact_rows.append({ + "algorithm": row["algorithm"], + "dataset": row["dataset"], + "metric": row["metric"], + "score_abs_diff": row.get("score_abs_diff"), + "score_tolerance_fraction": row.get("score_tolerance_fraction"), + "runtime_log2_ratio": row.get("runtime_log2_ratio"), + "configuration_difference_count": len(row.get("configuration_differences", [])), + "semantic_difference_count": len(row.get("semantic_differences", [])), + "strict_diagnostic_status": row.get("strict_diagnostic_status"), + "final_parity_status": row.get("final_parity_status"), + }) + snapshot = { + "commit": args.commit, + "environment_id": current.get("environment_id"), + "rows": compact_rows, + } + snapshots = [s for s in history.get("snapshots", []) if not (s.get("commit") == snapshot["commit"] and s.get("environment_id") == snapshot["environment_id"])] + snapshots.append(snapshot) + history["snapshots"] = snapshots[-50:] + args.history.write_text(json.dumps(history, indent=2) + "\n") + + if failures: + print("Disparity regression gate failed:") + for failure in failures: + print(" -", failure) + return 1 + print(f"disparity regression gate passed; history snapshots={len(history['snapshots'])}") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/benchmarks/disparity_regression_policy.json b/benchmarks/disparity_regression_policy.json new file mode 100644 index 0000000..490abb4 --- /dev/null +++ b/benchmarks/disparity_regression_policy.json @@ -0,0 +1,20 @@ +{ + "schema_version": 1, + "score_tolerance_fraction": { + "max_absolute_increase": 0.1 + }, + "score_abs_diff": { + "max_relative_increase": 0.5, + "absolute_noise_floor": 1e-7 + }, + "runtime_log2_ratio": { + "same_environment_only": true, + "max_absolute_change": 1.0 + }, + "configuration_difference_count": { + "max_increase": 0 + }, + "semantic_difference_count": { + "max_increase": 0 + } +} diff --git a/benchmarks/publish_headline_v2.py b/benchmarks/publish_headline_v2.py index 618668d..3db26ef 100644 --- a/benchmarks/publish_headline_v2.py +++ b/benchmarks/publish_headline_v2.py @@ -1,10 +1,5 @@ #!/usr/bin/env python3 -"""Validate and publish committed evidence artifacts into the static Pages tree. - -The public site renders the committed JSON directly. This keeps benchmark and -architecture presentation mechanically coupled to the repository source of -truth instead of patching historical hard-coded HTML. -""" +"""Validate and publish committed evidence artifacts into the static Pages tree.""" from __future__ import annotations import argparse @@ -17,8 +12,12 @@ DOCS = ROOT / "docs" RESULT = BENCH / "headline_result_v2.json" ARCH = BENCH / "architecture_performance_map.json" +DISPARITY = BENCH / "disparity_report.json" +HISTORY = BENCH / "disparity_history.json" DOC_RESULT = DOCS / "headline-result-v2.json" DOC_ARCH = DOCS / "architecture-performance-map.json" +DOC_DISPARITY = DOCS / "disparity-report.json" +DOC_HISTORY = DOCS / "disparity-history.json" def validate_headline(result: dict) -> None: @@ -36,16 +35,31 @@ def validate_headline(result: dict) -> None: def validate_architecture(architecture: dict, total_rows: int) -> None: coverage = architecture["coverage"] - if coverage["headline_rows"] != total_rows: - raise SystemExit("architecture map headline coverage disagrees with canonical result") - if coverage["headline_rows_with_substrate"] != total_rows: - raise SystemExit("not every headline row has an execution-substrate classification") - if coverage["headline_rows_with_speedup"] != total_rows: - raise SystemExit("not every headline row joins to benchmark speedup evidence") + if coverage["headline_rows"] != total_rows or coverage["headline_rows_with_substrate"] != total_rows or coverage["headline_rows_with_speedup"] != total_rows: + raise SystemExit("architecture map does not cover every canonical row") if not architecture.get("speedup_by_execution_substrate"): raise SystemExit("architecture map has no substrate/speedup aggregation") +def validate_disparity(disparity: dict, total_rows: int) -> None: + if disparity["counts"]["rows"] != total_rows: + raise SystemExit("disparity report does not cover every canonical row") + if disparity["counts"]["rows_with_tracked_disparity"] <= 0: + raise SystemExit("disparity report unexpectedly contains no tracked disparities") + keys = {(r["algorithm"], r["dataset"], r["metric"]) for r in disparity["rows"]} + if len(keys) != total_rows: + raise SystemExit("disparity report row keys are incomplete or duplicated") + + +def validate_history(history: dict, total_rows: int) -> None: + snapshots = history.get("snapshots", []) + if not snapshots: + raise SystemExit("disparity history has no snapshots") + for snapshot in snapshots: + if len(snapshot.get("rows", [])) != total_rows: + raise SystemExit("disparity history snapshot does not cover every canonical row") + + def main() -> int: parser = argparse.ArgumentParser() parser.add_argument("--check", action="store_true") @@ -56,18 +70,26 @@ def main() -> int: validate_headline(result) validate_architecture(architecture, result["counts"]["total_rows"]) + disparity = json.loads(DISPARITY.read_text()) if DISPARITY.exists() else None + history = json.loads(HISTORY.read_text()) if HISTORY.exists() else None + if disparity is not None: + validate_disparity(disparity, result["counts"]["total_rows"]) + elif not args.check: + raise SystemExit("disparity_report.json must be generated before Pages publication") + if history is not None: + validate_history(history, result["counts"]["total_rows"]) + elif not args.check: + raise SystemExit("disparity_history.json must be generated before Pages publication") + if not args.check: shutil.copyfile(RESULT, DOC_RESULT) shutil.copyfile(ARCH, DOC_ARCH) + shutil.copyfile(DISPARITY, DOC_DISPARITY) + shutil.copyfile(HISTORY, DOC_HISTORY) counts = result["counts"] - print( - "published evidence: " - f"{counts['flow_wins']}/{counts['eligible_comparisons']} Flow wins; " - f"{counts['parity_unresolved']} parity unresolved; " - f"{architecture['coverage']['inventory_operations']} estimator-operation inventory rows" - ) + print(f"published evidence: {counts['flow_wins']}/{counts['eligible_comparisons']} Flow wins; {disparity['counts']['rows_with_tracked_disparity']} rows with tracked disparities; {len(history['snapshots'])} history snapshots") else: - print("canonical benchmark and architecture evidence are internally consistent") + print("canonical benchmark, architecture, disparity, and available history evidence are internally consistent") return 0 diff --git a/docs/benchmarks.html b/docs/benchmarks.html index ecf2eba..e90d800 100644 --- a/docs/benchmarks.html +++ b/docs/benchmarks.html @@ -1 +1 @@ -Benchmarks — flow-scikit

canonical v2 / parity-gated benchmark

All 19 rows are measured. All 19 are parity-eligible.

This page renders the committed headline_result_v2.json artifact directly. Competitive timings use explicit millisecond units, persisted identical train/test fixtures, repeated measurements and a separate numerical-parity contract.

Flow wins—

End-to-end fit + predict comparisons won by Flow.

sklearn wins—

End-to-end comparisons won by scikit-learn.

parity eligible—

Rows admitted to the competitive denominator.

unresolved—

Parity or measurement failures. Canonical v2 currently requires zero.

TIMING_UNIT|msend-to-endseed=4280/20 persisted split2% practical tie threshold
KMeans note: Digits KMeans is classified as approximately equivalent rather than bit-identical. The semantic audit identifies initialization as the first divergence; the final objective remains closely matched, so the row is admitted under the explicit clustering-equivalence contract rather than by loosening measurement rules.

all canonical rows

No selected-win table.

Every row is shown below. Speedup is sklearn_ms / flow_ms; values above 1× favor Flow.

AlgorithmDatasetParityWinnersklearn msFlow msspeedup

methodology

Correctness and timing are separate gates.

The benchmark consumes the same persisted train/test indices in Python and Flow. Python uses high-resolution adaptive timing and the canonical runner aggregates repeated process measurements with medians and IQR. Flow timings are emitted in milliseconds and aggregated by the same runner. A row enters the headline only after its estimator-specific parity contract succeeds.

Supervised rows compare predictive metrics under declared tolerances. PCA additionally checks explained variance, singular values, reconstruction error and sign-aligned components. KMeans uses permutation-invariant clustering quality and inertia rather than label-mapped classification accuracy.

historical deployment evidence

Footprint and startup remain separate experiments.

The repository also contains a historical deployment comparison recording a roughly 1.4 MB Flow native executable and a roughly 65× cold-start advantage (33 ms versus 2160 ms). Those figures come from a different deployment experiment and are intentionally not mixed into the canonical estimator timing denominator.

reproduce

Read the source artifacts.

Canonical result ↗ Benchmark methodology ↗

\ No newline at end of file +Benchmarks — flow-scikit

canonical v2 / parity + disparity benchmark

Eligibility never means identity.

All 19 canonical rows are measured and currently eligible for comparison, but numerical, semantic and runtime disparities remain first-class evidence. This page renders the committed benchmark and disparity artifacts directly so differences cannot disappear merely because a row passes its contract.

Flow wins—

End-to-end fit + predict comparisons won by Flow.

sklearn wins—

End-to-end comparisons won by scikit-learn.

parity eligible—

Rows admitted to the competitive denominator.

tracked disparities—

Rows with non-zero numerical or explicit semantic/configuration differences.

TIMING_UNIT|msend-to-endseed=4280/20 persisted split2% practical tie thresholddisparity retained after eligibility
KMeans note: Digits KMeans is eligible as approximately equivalent, not bit-identical. Its strict diagnostic disparity, different initialization trajectory and convergence semantics remain visible separately from the final eligibility decision.

runtime overview

The plots are generated from the canonical JSON.

Each runtime plot shows end-to-end fit + predict time on a log scale. The plots use the same rows as the table below and therefore update whenever the frozen canonical result changes.

All 19 speed ratios

scikit-learn total time divided by Flow total time. The vertical 1× line separates Flow wins from scikit-learn wins.

Iris total runtime

scikit-learnFlow

Digits total runtime

scikit-learnFlow

Diabetes total runtime

scikit-learnFlow

persistent disparity

Passing parity does not erase the gap.

The disparity plot normalizes each row's principal numerical difference against its effective tolerance where a tolerance is available. A value near 1 means the row is close to the acceptance boundary. Semantic/configuration differences are tracked in the same artifact and remain visible in the table.

Numerical disparity relative to tolerance

The dashed line is the acceptance boundary. Values can remain non-zero even for eligible rows.

all canonical rows

No selected-win table.

Every row is shown below. Speedup is sklearn_ms / flow_ms; values above 1× favor Flow. Strict diagnostic status is kept separate from final eligibility.

AlgorithmDatasetFinal parityStrict diagnosticWinnerscore |Δ|sklearn msFlow msspeedup

methodology

Correctness, disparity and timing are separate dimensions.

The benchmark consumes the same persisted train/test indices in Python and Flow. Python uses high-resolution adaptive timing and the canonical runner aggregates repeated process measurements with medians and IQR. Flow timings are emitted in milliseconds and aggregated by the same runner.

Supervised rows compare predictive metrics under declared tolerances. PCA additionally checks explained variance, singular values, reconstruction error and sign-aligned components. KMeans uses permutation-invariant clustering quality and inertia. The persistent disparity artifact preserves raw numerical gaps and known semantic/configuration differences even after the estimator-specific eligibility contract succeeds.

historical deployment evidence

Footprint and startup remain separate experiments.

The repository also contains a historical deployment comparison recording a roughly 1.4 MB Flow native executable and a roughly 65× cold-start advantage (33 ms versus 2160 ms). Those figures come from a different deployment experiment and are intentionally not mixed into the canonical estimator timing denominator.

reproduce

Read the source artifacts.

Canonical result ↗ Disparity report ↗

\ No newline at end of file diff --git a/docs/disparity-history.html b/docs/disparity-history.html new file mode 100644 index 0000000..dcfec6e --- /dev/null +++ b/docs/disparity-history.html @@ -0,0 +1 @@ +Disparity history — flow-scikit

persistent benchmark evidence

Disparity is tracked across freezes, not only at the current commit.

Each canonical benchmark freeze stores score disparity, tolerance usage, runtime ratio and semantic/configuration counts. CI compares the new snapshot with the previous frozen report before accepting it.

Worst tolerance-normalized numerical disparity by snapshot

The dashed line is the acceptance boundary. A rising line means at least one canonical row is moving closer to its numerical tolerance even if parity still passes.

Latest snapshot

AlgorithmDataset|Δ| / toleranceruntime log2 ratioconfig diffssemantic diffs