diff --git a/.github/workflows/validate.yml b/.github/workflows/validate.yml index 6b148e67..91911c3f 100644 --- a/.github/workflows/validate.yml +++ b/.github/workflows/validate.yml @@ -1,6 +1,10 @@ name: Validate on: push: + # The branch commit is the release-gate authority. An annotated tag points + # to the same SHA and must not duplicate every expensive browser/perf job. + tags-ignore: + - '**' pull_request: schedule: - cron: "0 4 * * 1" diff --git a/demo/performance/README.md b/demo/performance/README.md index 358b390f..7fcd6e64 100644 --- a/demo/performance/README.md +++ b/demo/performance/README.md @@ -34,7 +34,8 @@ not normal-performance targets; the base-relative comparison catches smaller regressions. Small fast operations receive an absolute noise allowance so normal scheduler jitter does not become a false regression. Heap, Long Tasks, warmed-cache growth and the expected rendered-device count are -gated separately. Both raw reports and the comparison are always uploaded as +gated separately. Long-Task maximum/count/total checks use the same +relative-plus-absolute policy as timings. Both raw reports and the comparison are always uploaded as the `large-house-performance` artifact, and the table is written to the GitHub job summary. @@ -71,3 +72,10 @@ changing the meaning of `large-house-v1`. The `cleanFloor` entry ceiling is 160: the reviewed fixture currently warms 120 deterministic room/physical-body entries, and the extra 40 slots allow a legitimate fixture extension without weakening the separate zero-growth gate. + +The absolute switch-cycle/Long-Task ceilings include roughly 20–30% headroom +over the paired 2026-08-09 Ubuntu run where the unchanged base and candidate +both reached about 5.3 s / 2.45 s / 22 tasks / 9.9 s total under runner load. +The same-runner relative checks remain tighter for an actual candidate-only +regression; this prevents an overloaded but symmetric runner from turning an +absolute safety ceiling into a flaky code-regression signal. diff --git a/demo/performance/budgets.json b/demo/performance/budgets.json index 316a2375..c9979696 100644 --- a/demo/performance/budgets.json +++ b/demo/performance/budgets.json @@ -49,13 +49,17 @@ "stat": "median", "maxRegressionRatio": 0.35, "noiseAllowanceMs": 250, - "hardMaxMs": 5000 + "hardMaxMs": 7000 } }, "longTasks": { - "maxSingleMs": 2000, - "maxCountP95": 20, - "maxTotalP95Ms": 8000, + "maxSingleMs": 3000, + "maxSingleRegressionRatio": 0.3, + "maxSingleNoiseAllowanceMs": 250, + "maxCountP95": 30, + "maxCountRegressionRatio": 0.35, + "countNoiseAllowance": 3, + "maxTotalP95Ms": 12000, "maxTotalRegressionRatio": 0.3, "noiseAllowanceMs": 150 }, diff --git a/demo/performance/evaluate.mjs b/demo/performance/evaluate.mjs index df331a19..cd12ae8a 100644 --- a/demo/performance/evaluate.mjs +++ b/demo/performance/evaluate.mjs @@ -95,8 +95,28 @@ export const evaluatePerformanceBudget = ({ candidate, baseline, budgets }) => { return windows.length > 0 && windows.every((item) => item?.supported === true); }); checks.push({ id: 'longTask.available', actual: longTasksAvailable ? 1 : 0, limit: 1, pass: longTasksAvailable }); - checks.push(makeCheck('longTask.maxSingleMs', candidateLong.maxSingleMs, budgets.longTasks.maxSingleMs)); - checks.push(makeCheck('longTask.countP95', candidateLong.countP95, budgets.longTasks.maxCountP95)); + const singleRegressionLimit = relativeLimit( + baselineLong.maxSingleMs, + budgets.longTasks.maxSingleRegressionRatio, + budgets.longTasks.maxSingleNoiseAllowanceMs, + ); + checks.push(makeCheck( + 'longTask.maxSingleMs', + candidateLong.maxSingleMs, + Math.min(budgets.longTasks.maxSingleMs, singleRegressionLimit), + { baseline: baselineLong.maxSingleMs, hardLimit: budgets.longTasks.maxSingleMs }, + )); + const countRegressionLimit = relativeLimit( + baselineLong.countP95, + budgets.longTasks.maxCountRegressionRatio, + budgets.longTasks.countNoiseAllowance, + ); + checks.push(makeCheck( + 'longTask.countP95', + candidateLong.countP95, + Math.min(budgets.longTasks.maxCountP95, countRegressionLimit), + { baseline: baselineLong.countP95, hardLimit: budgets.longTasks.maxCountP95 }, + )); const longRegressionLimit = relativeLimit( baselineLong.totalP95Ms, budgets.longTasks.maxTotalRegressionRatio, diff --git a/test/performance-budget.test.mjs b/test/performance-budget.test.mjs index 91211ddf..74598368 100644 --- a/test/performance-budget.test.mjs +++ b/test/performance-budget.test.mjs @@ -13,6 +13,8 @@ const budgets = { }, longTasks: { maxSingleMs: 200, maxCountP95: 10, maxTotalP95Ms: 500, + maxSingleRegressionRatio: 0.5, maxSingleNoiseAllowanceMs: 20, + maxCountRegressionRatio: 0.5, countNoiseAllowance: 2, maxTotalRegressionRatio: 0.5, noiseAllowanceMs: 20, }, heap: { @@ -75,6 +77,17 @@ test('performance budget rejects long tasks, heap growth, cache growth and missi ); }); +test('performance budget compares single/count Long Tasks to the same-runner baseline', () => { + const baseline = report({ long: 100 }); + baseline.longTasks.countP95 = 4; + const candidate = report({ long: 130 }); + candidate.longTasks.countP95 = 6; + const result = evaluatePerformanceBudget({ baseline, candidate, budgets }); + assert.equal(result.pass, true); + assert.equal(result.checks.find((check) => check.id === 'longTask.maxSingleMs')?.baseline, 100); + assert.equal(result.checks.find((check) => check.id === 'longTask.countP95')?.baseline, 4); +}); + test('performance budget refuses incomparable runtime profiles', () => { const candidate = report(); candidate.runtime.chromium = 'different';