From df0e12ebda4710c6a3ac44d20f2b34edcd4b3edf Mon Sep 17 00:00:00 2001 From: Jonathan <64296013+Joncallim@users.noreply.github.com> Date: Sun, 27 Sep 2026 18:29:38 +0800 Subject: [PATCH 1/2] test(perf): make the Baseline-4 promotion comparison runnable What: add the perf:promote command and fail-closed composite comparison CLI, with coverage and evidence instructions.\n\nWhy: make reviewed Baseline-4 promotion comparisons directly runnable from composite evidence artifacts.\n\nHow checked: npm run test:perf; npm run typecheck; git diff --check. --- docs/testing/TIME_TO_ANSWER_EVIDENCE.md | 14 +++-- package.json | 1 + tests/perf/promote.test.mjs | 82 +++++++++++++++++++++++++ tests/perf/promote.ts | 81 ++++++++++++++++++++++++ tests/perf/tsconfig.json | 2 +- 5 files changed, 174 insertions(+), 6 deletions(-) create mode 100644 tests/perf/promote.test.mjs create mode 100644 tests/perf/promote.ts diff --git a/docs/testing/TIME_TO_ANSWER_EVIDENCE.md b/docs/testing/TIME_TO_ANSWER_EVIDENCE.md index 69fcb305..10226538 100644 --- a/docs/testing/TIME_TO_ANSWER_EVIDENCE.md +++ b/docs/testing/TIME_TO_ANSWER_EVIDENCE.md @@ -375,11 +375,15 @@ npx tsx tests/perf/assembleCompositeEvidence.ts \ --output /tmp/time-to-answer-baseline.json # 6. recompute summaries from the raw samples (never trust supplied aggregates) npm run perf:summarize -- --artifact /tmp/time-to-answer-baseline.json -# 7. compare a candidate against a reviewed baseline (fails closed) -npm run perf:time-to-answer -- \ - --metadata /tmp/time-to-answer-metadata.json \ - --output /tmp/time-to-answer-candidate.json \ - --baseline /tmp/time-to-answer-baseline.json +# 7. assemble the candidate's end-to-end and Stage-5 captures into its composite +npx tsx tests/perf/assembleCompositeEvidence.ts \ + --general /tmp/time-to-answer-candidate-general.json \ + --stageFive /tmp/time-to-answer-candidate-stage5.json \ + --output /tmp/time-to-answer-candidate.json +# 8. compare the two composite artifacts (fails closed) +npm run perf:promote -- \ + --baseline /tmp/time-to-answer-baseline.json \ + --candidate /tmp/time-to-answer-candidate.json ``` Prerequisites: a release daemon (`cargo build --release -p dockermap-daemon`), diff --git a/package.json b/package.json index 126e17b1..48a28cd4 100644 --- a/package.json +++ b/package.json @@ -32,6 +32,7 @@ "test:version": "node --test scripts/check-version-authority.test.mjs scripts/package-release.test.mjs", "test:perf": "node --test tests/perf/*.test.mjs", "perf:time-to-answer": "tsx tests/perf/capture.ts", + "perf:promote": "tsx tests/perf/promote.ts", "perf:calibrate-time-to-answer": "tsx tests/perf/capture.ts", "perf:preconditioning": "tsx tests/perf/preconditioning.ts", "perf:phase-control": "tsx tests/perf/phaseControl.ts", diff --git a/tests/perf/promote.test.mjs b/tests/perf/promote.test.mjs new file mode 100644 index 00000000..4fb897d3 --- /dev/null +++ b/tests/perf/promote.test.mjs @@ -0,0 +1,82 @@ +import { mkdtempSync, rmSync, writeFileSync } from "node:fs"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; +import { spawnSync } from "node:child_process"; +import assert from "node:assert/strict"; +import test from "node:test"; + +const fixtures = ["reference-25", "reference-100", "reference-250"]; +const stages = [ +["daemonStartToListenerMs", fixtures], +["listenerToFirstDockerModelMs", fixtures], +["dockerObservationMs", fixtures], +["composeEnrichmentMs", [...fixtures, "slow-bounded-compose-projection"]], +["publicationToNodeObservationMs", [...fixtures, "provider-only-revision-change", "docker-topology-change", "unavailable-optional-provider"]], +["notificationToCoherentModelMs", [...fixtures, "provider-only-revision-change", "docker-topology-change", "unavailable-optional-provider"]], +["coherentModelToUsefulRenderMs", [...fixtures, "docker-topology-change"]], +["buildModelMs", fixtures], +["findingsDerivationMs", fixtures], +["legacyTopologyLayoutMs", fixtures], +["commandQueryMs", fixtures], +["productionBundleMs", fixtures] +]; + +const environment = { +runnerClass: "linux-x86_64-dedicated", cpuClass: "cpus-16vcpu", osImage: "ubuntu-26.04", osKernel: "7.0.0-31-generic", +nodeRevision: "22.23.2", rustRevision: "1.88.0", dockerRevision: "29.8.1", ssePollIntervalMs: "2000", +daemonBinarySha256: "eeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeee", +daemonBinaryBuild: "cargo-build-release-locked-p-dockermap-daemon", cargoRevision: "cargo-1.88.0", +harnessRevision: "dddddddddddddddddddddddddddddddddddddddd", browserEngine: "chromium", browserRevision: "1.61.0", +browserFlags: ["--disable-background-networking"], fontEnvironment: "system-default", buildMode: "production", +fixtureRevision: "dockermap-v1/time-to-answer-fixtures-1", sourceRevision: "aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa", +methodologyVersion: "dockermap-v1/time-to-answer-methodology-8" +}; + +function artifact(value = 10, overrides = {}) { +return { +baseline: "dockermap-v1/time-to-answer-baseline-4", +environment: { ...environment, ...(overrides.environment ?? {}) }, +records: stages.flatMap(([stage, names]) => names.map((fixture) => ({ +fixture, stage, measurementProtocol: stage === "publicationToNodeObservationMs" ? "controlled-poll-phase" : "end-to-end", +sourceEvidenceFile: "synthetic.raw.json", checkpointSha: "aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa", +runs: Array.from({ length: 3 }, () => Array.from({ length: 15 }, () => value)) +}))) +}; +} + +function run(baseline, candidate) { +const directory = mkdtempSync(join(tmpdir(), "dockermap-promote-")); +try { +const baselinePath = join(directory, "baseline.json"); +const candidatePath = join(directory, "candidate.json"); +writeFileSync(baselinePath, JSON.stringify(baseline)); +writeFileSync(candidatePath, JSON.stringify(candidate)); +return spawnSync("npx", ["tsx", "tests/perf/promote.ts", "--baseline", baselinePath, "--candidate", candidatePath], { +cwd: new URL("../..", import.meta.url), encoding: "utf8", timeout: 30_000 +}); +} finally { +rmSync(directory, { recursive: true, force: true }); +} +} + +test("promotion CLI accepts a compatible candidate within the reviewed limit", () => { +const result = run(artifact(10), artifact(12)); +assert.equal(result.status, 0, result.stderr); +assert.match(result.stdout, /PROMOTION: PASS/); +}); + +test("promotion CLI names a stage beyond the reviewed limit", () => { +const candidate = artifact(10); +candidate.records[0].runs = Array.from({ length: 3 }, () => Array.from({ length: 15 }, () => 13)); +const result = run(artifact(10), candidate); +assert.notEqual(result.status, 0); +assert.match(result.stdout, /reference-25\u0000daemonStartToListenerMs/); +assert.match(result.stdout, /PROMOTION: FAIL:Time-to-answer candidate exceeds/); +}); + +test("promotion CLI rejects an incompatible environment before a timing limit", () => { +const result = run(artifact(10), artifact(99, { environment: { osImage: "different-image" } })); +assert.notEqual(result.status, 0); +assert.match(result.stdout, /candidate does not match the pinned baseline environment/); +assert.doesNotMatch(result.stdout, /candidate exceeds the reviewed promotion limit/); +}); diff --git a/tests/perf/promote.ts b/tests/perf/promote.ts new file mode 100644 index 00000000..57765b72 --- /dev/null +++ b/tests/perf/promote.ts @@ -0,0 +1,81 @@ +#!/usr/bin/env node +/** Read-only promotion comparison for two closed composite evidence artifacts. */ +import { readFileSync } from "node:fs"; +import { +assertTimeToAnswerPromotion, +derivedTimeToAnswerSummaries, +timeToAnswerLimit, +validateTimeToAnswerEvidence +} from "../../apps/web/src/lib/performance/timeToAnswerEvidence"; + +type Paths = { baseline: string; candidate: string }; + +function paths(arguments_: readonly string[]): Paths { +const values: Partial = {}; +for (let index = 0; index < arguments_.length; index += 1) { +const flag = arguments_[index]!; +if (flag !== "--baseline" && flag !== "--candidate") throw new Error(`Unknown flag: ${flag}`); +const value = arguments_[index + 1]; +if (!value || value.startsWith("--")) throw new Error(`Missing value for ${flag}`); +const key = flag.slice(2) as keyof Paths; +if (values[key]) throw new Error(`Duplicate flag: ${flag}`); +values[key] = value; +index += 1; +} +if (!values.baseline || !values.candidate) { +throw new Error("Usage: npm run perf:promote -- --baseline --candidate "); +} +return values as Paths; +} + +function readEvidence(path: string): unknown { +try { +return JSON.parse(readFileSync(path, "utf8")); +} catch (error) { +throw new Error(`Cannot read ${path}: ${error instanceof Error ? error.message : String(error)}`); +} +} + +function format(value: number): string { return value.toFixed(2); } + +function main(): void { +const input = paths(process.argv.slice(2)); +const baselineRaw = readEvidence(input.baseline); +const candidateRaw = readEvidence(input.candidate); +const baseline = validateTimeToAnswerEvidence(baselineRaw); +const candidate = validateTimeToAnswerEvidence(candidateRaw); +const baselineSummaries = derivedTimeToAnswerSummaries(baseline); +const candidateSummaries = derivedTimeToAnswerSummaries(candidate); +const regressions: string[] = []; +const violations: string[] = []; +process.stdout.write("| fixture | stage | protocol | baseline reviewed (ms) | limit (ms) | candidate reviewed (ms) | delta (ms) | verdict |\n"); +process.stdout.write("| --- | --- | --- | ---: | ---: | ---: | ---: | --- |\n"); +for (const record of candidate.records) { +const key = `${record.fixture}\u0000${record.stage}`; +const before = baselineSummaries.get(key)!; +const after = candidateSummaries.get(key)!; +const limit = timeToAnswerLimit(before.reviewedMs); +const delta = after.reviewedMs - before.reviewedMs; +if (delta > 0) regressions.push(`${record.fixture}/${record.stage}`); +if (after.reviewedMs > limit) violations.push(key); +process.stdout.write(`| ${record.fixture} | ${record.stage} | ${record.measurementProtocol} | ${format(before.reviewedMs)} | ${format(limit)} | ${format(after.reviewedMs)} | ${format(delta)} | ${after.reviewedMs <= limit ? "PASS" : "FAIL"} |\n`); +} +process.stdout.write(`Slower than baseline: ${regressions.join(", ") || "none"}\n`); +try { +assertTimeToAnswerPromotion(baselineRaw, candidateRaw); +process.stdout.write("PROMOTION: PASS\n"); +} catch (error) { +if (violations.length > 0) process.stdout.write(`Offending keys: ${violations.join(", ")}\n`); +const reason = error instanceof Error ? error.message : String(error); +process.stdout.write(`PROMOTION: FAIL:${reason}\n`); +process.exitCode = 1; +} +} + +try { +main(); +} catch (error) { +const reason = error instanceof Error ? error.message : String(error); +process.stderr.write(`PROMOTION: FAIL:${reason}\n`); +process.exitCode = 1; +} diff --git a/tests/perf/tsconfig.json b/tests/perf/tsconfig.json index 1b47df4c..e72b93c5 100644 --- a/tests/perf/tsconfig.json +++ b/tests/perf/tsconfig.json @@ -4,5 +4,5 @@ "types": ["node"], "allowJs": true }, - "include": ["capture.ts", "captureIndependence.ts", "summarize.ts", "preconditioning.ts", "phaseControl.ts"] + "include": ["capture.ts", "captureIndependence.ts", "summarize.ts", "preconditioning.ts", "phaseControl.ts", "promote.ts"] } From e275c1160252107d4dba16bf31abd8c5059a32d2 Mon Sep 17 00:00:00 2001 From: Jonathan <64296013+Joncallim@users.noreply.github.com> Date: Sun, 27 Sep 2026 18:45:14 +0800 Subject: [PATCH 2/2] docs(perf): make the documented capture-to-promotion sequence runnable What: document checkpoints and complete candidate capture steps.\n\nWhy: make the performance evidence sequence executable from capture through promotion.\n\nHow checked: inspected the updated command block and ran git diff --check. --- docs/testing/TIME_TO_ANSWER_EVIDENCE.md | 19 +++++++++++++++++-- 1 file changed, 17 insertions(+), 2 deletions(-) diff --git a/docs/testing/TIME_TO_ANSWER_EVIDENCE.md b/docs/testing/TIME_TO_ANSWER_EVIDENCE.md index 10226538..050a8ae6 100644 --- a/docs/testing/TIME_TO_ANSWER_EVIDENCE.md +++ b/docs/testing/TIME_TO_ANSWER_EVIDENCE.md @@ -358,6 +358,7 @@ npm run perf:time-to-answer -- \ --checkpoint # 3. dedicated controlled Stage-5 capture; the only owner of the poll-phase protocol npm run perf:stage-five -- \ + --checkpoint \ --metadata /tmp/time-to-answer-metadata.json \ --output /tmp/time-to-answer-stage5.json \ --raw-dir /tmp/time-to-answer-stage5-raw @@ -375,12 +376,26 @@ npx tsx tests/perf/assembleCompositeEvidence.ts \ --output /tmp/time-to-answer-baseline.json # 6. recompute summaries from the raw samples (never trust supplied aggregates) npm run perf:summarize -- --artifact /tmp/time-to-answer-baseline.json -# 7. assemble the candidate's end-to-end and Stage-5 captures into its composite +# 7. pin the candidate environment from the candidate checkout +npm run perf:metadata -- --output /tmp/time-to-answer-candidate-metadata.json +# 8. capture the candidate's end-to-end section at +npm run perf:time-to-answer -- \ + --metadata /tmp/time-to-answer-candidate-metadata.json \ + --output /tmp/time-to-answer-candidate-general.json \ + --raw-dir /tmp/time-to-answer-candidate-raw \ + --checkpoint +# 9. capture the candidate's dedicated Stage-5 section at +npm run perf:stage-five -- \ + --metadata /tmp/time-to-answer-candidate-metadata.json \ + --output /tmp/time-to-answer-candidate-stage5.json \ + --raw-dir /tmp/time-to-answer-candidate-stage5-raw \ + --checkpoint +# 10. assemble the candidate's captures into its composite npx tsx tests/perf/assembleCompositeEvidence.ts \ --general /tmp/time-to-answer-candidate-general.json \ --stageFive /tmp/time-to-answer-candidate-stage5.json \ --output /tmp/time-to-answer-candidate.json -# 8. compare the two composite artifacts (fails closed) +# 11. compare the two composite artifacts (fails closed) npm run perf:promote -- \ --baseline /tmp/time-to-answer-baseline.json \ --candidate /tmp/time-to-answer-candidate.json