Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
6 changes: 5 additions & 1 deletion .github/workflows/pages.yml
Original file line number Diff line number Diff line change
Expand Up @@ -47,7 +47,11 @@ jobs:
run: python tools/build_browser_wasm.py --flow-repo .flow-toolchain
- name: Smoke direct browser WebAssembly API
run: node tools/smoke_browser_wasm.mjs docs/wasm/browser_mlir.wasm
- name: Generate canonical v2 benchmark headline
- name: Regenerate canonical disparity evidence
run: python benchmarks/generate_disparity_report.py
- name: Materialize disparity history for this Pages build
run: python benchmarks/check_disparity_regression.py --baseline /tmp/no-disparity-baseline.json --commit "${GITHUB_SHA}"
- name: Publish canonical benchmark, disparity, history and architecture evidence
run: python benchmarks/publish_headline_v2.py
- name: Normalize legacy sklearn timings to milliseconds
run: python benchmarks/normalize_published_timings.py
Expand Down
31 changes: 28 additions & 3 deletions .github/workflows/remaining-issues.yml
Original file line number Diff line number Diff line change
Expand Up @@ -14,6 +14,11 @@ jobs:
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v4
- name: Preserve frozen disparity baseline
run: |
if [ -f benchmarks/disparity_report.json ]; then
cp benchmarks/disparity_report.json /tmp/disparity_report.baseline.json
fi
- uses: actions/checkout@v4
with:
repository: flooooooooooow/flow
Expand All @@ -38,18 +43,35 @@ jobs:
run: python benchmarks/run_headline.py --repeats 3
- name: Apply evidence-backed KMeans equivalence gate
run: python benchmarks/finalize_kmeans_parity.py
- name: Require all 19 rows parity eligible
- name: Generate persistent disparity report
run: |
cp benchmarks/headline_summary.json benchmarks/headline_result_v2.json
python benchmarks/generate_disparity_report.py
- name: Gate disparity regressions and append history
run: |
if [ -f /tmp/disparity_report.baseline.json ]; then
python benchmarks/check_disparity_regression.py --baseline /tmp/disparity_report.baseline.json
else
python benchmarks/check_disparity_regression.py --baseline /tmp/no-disparity-baseline.json
fi
- name: Require all 19 rows parity eligible and disparities retained
run: |
python - <<'PY'
import json
from pathlib import Path
result = json.loads(Path('benchmarks/headline_summary.json').read_text())
disparity = json.loads(Path('benchmarks/disparity_report.json').read_text())
history = json.loads(Path('benchmarks/disparity_history.json').read_text())
c = result['counts']
assert c['total_rows'] == 19, c
assert c['eligible_comparisons'] == 19, c
assert c['parity_unresolved'] == 0, c
assert c['measurement_unresolved'] == 0, c
print(c)
assert disparity['counts']['rows'] == 19, disparity['counts']
assert disparity['counts']['rows_with_tracked_disparity'] > 0, disparity['counts']
assert disparity['counts']['strict_final_status_disagreements'] >= 1, disparity['counts']
assert history['snapshots'], history
print({'headline': c, 'disparity': disparity['counts'], 'history_snapshots': len(history['snapshots'])})
PY
- uses: actions/upload-artifact@v4
with:
Expand All @@ -60,7 +82,10 @@ jobs:
benchmarks/flow_results_v2.txt
benchmarks/headline_rows.json
benchmarks/headline_summary.json
benchmarks/headline_result_v2.json
benchmarks/parity_diagnostics.json
benchmarks/disparity_report.json
benchmarks/disparity_history.json
benchmarks/headline_environment.json

architecture-map:
Expand Down Expand Up @@ -149,5 +174,5 @@ jobs:
echo 'Generated evidence already current.'
exit 0
fi
git commit -m "Freeze validated sklearn architecture and 19/19 benchmark evidence [skip ci]"
git commit -m "Freeze validated benchmark, disparity history, and architecture evidence [skip ci]"
git push origin HEAD:main
105 changes: 105 additions & 0 deletions benchmarks/check_disparity_regression.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,105 @@
#!/usr/bin/env python3
from __future__ import annotations

import argparse
import json
import os
from pathlib import Path

ROOT = Path(__file__).resolve().parent


def key(row: dict) -> tuple[str, str, str]:
return row["algorithm"], row["dataset"], row["metric"]


def main() -> int:
p = argparse.ArgumentParser()
p.add_argument("--current", type=Path, default=ROOT / "disparity_report.json")
p.add_argument("--baseline", type=Path, default=ROOT / "disparity_report.baseline.json")
p.add_argument("--history", type=Path, default=ROOT / "disparity_history.json")
p.add_argument("--policy", type=Path, default=ROOT / "disparity_regression_policy.json")
p.add_argument("--commit", default=os.environ.get("GITHUB_SHA", "unknown"))
args = p.parse_args()

current = json.loads(args.current.read_text())
policy = json.loads(args.policy.read_text())
failures: list[str] = []

baseline = None
if args.baseline.exists():
baseline = json.loads(args.baseline.read_text())
old = {key(r): r for r in baseline["rows"]}
same_env = baseline.get("environment_id") == current.get("environment_id")
for row in current["rows"]:
previous = old.get(key(row))
if previous is None:
continue

old_frac = previous.get("score_tolerance_fraction")
new_frac = row.get("score_tolerance_fraction")
if old_frac is not None and new_frac is not None:
limit = float(policy["score_tolerance_fraction"]["max_absolute_increase"])
if new_frac - old_frac > limit:
failures.append(f"{key(row)} tolerance fraction regressed {old_frac:.6g} -> {new_frac:.6g}")

old_abs = previous.get("score_abs_diff")
new_abs = row.get("score_abs_diff")
if old_abs is not None and new_abs is not None:
floor = float(policy["score_abs_diff"]["absolute_noise_floor"])
rel = float(policy["score_abs_diff"]["max_relative_increase"])
if new_abs > max(floor, old_abs * (1.0 + rel)):
failures.append(f"{key(row)} score |delta| regressed {old_abs:.6g} -> {new_abs:.6g}")

for field, policy_name in (("configuration_differences", "configuration_difference_count"), ("semantic_differences", "semantic_difference_count")):
old_n = len(previous.get(field, []))
new_n = len(row.get(field, []))
if new_n - old_n > int(policy[policy_name]["max_increase"]):
failures.append(f"{key(row)} {field} increased {old_n} -> {new_n}")

if same_env or not policy["runtime_log2_ratio"].get("same_environment_only", True):
old_rt = previous.get("runtime_log2_ratio")
new_rt = row.get("runtime_log2_ratio")
if old_rt is not None and new_rt is not None:
limit = float(policy["runtime_log2_ratio"]["max_absolute_change"])
if abs(new_rt - old_rt) > limit:
failures.append(f"{key(row)} runtime log2 ratio changed {old_rt:.4f} -> {new_rt:.4f}")

history = {"schema_version": 1, "snapshots": []}
if args.history.exists():
history = json.loads(args.history.read_text())
compact_rows = []
for row in current["rows"]:
compact_rows.append({
"algorithm": row["algorithm"],
"dataset": row["dataset"],
"metric": row["metric"],
"score_abs_diff": row.get("score_abs_diff"),
"score_tolerance_fraction": row.get("score_tolerance_fraction"),
"runtime_log2_ratio": row.get("runtime_log2_ratio"),
"configuration_difference_count": len(row.get("configuration_differences", [])),
"semantic_difference_count": len(row.get("semantic_differences", [])),
"strict_diagnostic_status": row.get("strict_diagnostic_status"),
"final_parity_status": row.get("final_parity_status"),
})
snapshot = {
"commit": args.commit,
"environment_id": current.get("environment_id"),
"rows": compact_rows,
}
snapshots = [s for s in history.get("snapshots", []) if not (s.get("commit") == snapshot["commit"] and s.get("environment_id") == snapshot["environment_id"])]
snapshots.append(snapshot)
history["snapshots"] = snapshots[-50:]
args.history.write_text(json.dumps(history, indent=2) + "\n")

if failures:
print("Disparity regression gate failed:")
for failure in failures:
print(" -", failure)
return 1
print(f"disparity regression gate passed; history snapshots={len(history['snapshots'])}")
return 0


if __name__ == "__main__":
raise SystemExit(main())
20 changes: 20 additions & 0 deletions benchmarks/disparity_regression_policy.json
Original file line number Diff line number Diff line change
@@ -0,0 +1,20 @@
{
"schema_version": 1,
"score_tolerance_fraction": {
"max_absolute_increase": 0.1
},
"score_abs_diff": {
"max_relative_increase": 0.5,
"absolute_noise_floor": 1e-7
},
"runtime_log2_ratio": {
"same_environment_only": true,
"max_absolute_change": 1.0
},
"configuration_difference_count": {
"max_increase": 0
},
"semantic_difference_count": {
"max_increase": 0
}
}
60 changes: 41 additions & 19 deletions benchmarks/publish_headline_v2.py
Original file line number Diff line number Diff line change
@@ -1,10 +1,5 @@
#!/usr/bin/env python3
"""Validate and publish committed evidence artifacts into the static Pages tree.

The public site renders the committed JSON directly. This keeps benchmark and
architecture presentation mechanically coupled to the repository source of
truth instead of patching historical hard-coded HTML.
"""
"""Validate and publish committed evidence artifacts into the static Pages tree."""
from __future__ import annotations

import argparse
Expand All @@ -17,8 +12,12 @@
DOCS = ROOT / "docs"
RESULT = BENCH / "headline_result_v2.json"
ARCH = BENCH / "architecture_performance_map.json"
DISPARITY = BENCH / "disparity_report.json"
HISTORY = BENCH / "disparity_history.json"
DOC_RESULT = DOCS / "headline-result-v2.json"
DOC_ARCH = DOCS / "architecture-performance-map.json"
DOC_DISPARITY = DOCS / "disparity-report.json"
DOC_HISTORY = DOCS / "disparity-history.json"


def validate_headline(result: dict) -> None:
Expand All @@ -36,16 +35,31 @@ def validate_headline(result: dict) -> None:

def validate_architecture(architecture: dict, total_rows: int) -> None:
coverage = architecture["coverage"]
if coverage["headline_rows"] != total_rows:
raise SystemExit("architecture map headline coverage disagrees with canonical result")
if coverage["headline_rows_with_substrate"] != total_rows:
raise SystemExit("not every headline row has an execution-substrate classification")
if coverage["headline_rows_with_speedup"] != total_rows:
raise SystemExit("not every headline row joins to benchmark speedup evidence")
if coverage["headline_rows"] != total_rows or coverage["headline_rows_with_substrate"] != total_rows or coverage["headline_rows_with_speedup"] != total_rows:
raise SystemExit("architecture map does not cover every canonical row")
if not architecture.get("speedup_by_execution_substrate"):
raise SystemExit("architecture map has no substrate/speedup aggregation")


def validate_disparity(disparity: dict, total_rows: int) -> None:
if disparity["counts"]["rows"] != total_rows:
raise SystemExit("disparity report does not cover every canonical row")
if disparity["counts"]["rows_with_tracked_disparity"] <= 0:
raise SystemExit("disparity report unexpectedly contains no tracked disparities")
keys = {(r["algorithm"], r["dataset"], r["metric"]) for r in disparity["rows"]}
if len(keys) != total_rows:
raise SystemExit("disparity report row keys are incomplete or duplicated")


def validate_history(history: dict, total_rows: int) -> None:
snapshots = history.get("snapshots", [])
if not snapshots:
raise SystemExit("disparity history has no snapshots")
for snapshot in snapshots:
if len(snapshot.get("rows", [])) != total_rows:
raise SystemExit("disparity history snapshot does not cover every canonical row")


def main() -> int:
parser = argparse.ArgumentParser()
parser.add_argument("--check", action="store_true")
Expand All @@ -56,18 +70,26 @@ def main() -> int:
validate_headline(result)
validate_architecture(architecture, result["counts"]["total_rows"])

disparity = json.loads(DISPARITY.read_text()) if DISPARITY.exists() else None
history = json.loads(HISTORY.read_text()) if HISTORY.exists() else None
if disparity is not None:
validate_disparity(disparity, result["counts"]["total_rows"])
elif not args.check:
raise SystemExit("disparity_report.json must be generated before Pages publication")
if history is not None:
validate_history(history, result["counts"]["total_rows"])
elif not args.check:
raise SystemExit("disparity_history.json must be generated before Pages publication")

if not args.check:
shutil.copyfile(RESULT, DOC_RESULT)
shutil.copyfile(ARCH, DOC_ARCH)
shutil.copyfile(DISPARITY, DOC_DISPARITY)
shutil.copyfile(HISTORY, DOC_HISTORY)
counts = result["counts"]
print(
"published evidence: "
f"{counts['flow_wins']}/{counts['eligible_comparisons']} Flow wins; "
f"{counts['parity_unresolved']} parity unresolved; "
f"{architecture['coverage']['inventory_operations']} estimator-operation inventory rows"
)
print(f"published evidence: {counts['flow_wins']}/{counts['eligible_comparisons']} Flow wins; {disparity['counts']['rows_with_tracked_disparity']} rows with tracked disparities; {len(history['snapshots'])} history snapshots")
else:
print("canonical benchmark and architecture evidence are internally consistent")
print("canonical benchmark, architecture, disparity, and available history evidence are internally consistent")
return 0


Expand Down
Loading
Loading