ci: make performance gate runner-noise aware

This commit is contained in:
Matysh
2026-08-09 08:50:26 +03:00
parent d5e6c5cff0
commit 08b19363fd
5 changed files with 56 additions and 7 deletions
+9 -1
View File
@@ -34,7 +34,8 @@ not normal-performance targets; the base-relative comparison catches smaller
regressions. Small fast operations receive an absolute noise
allowance so normal scheduler jitter does not become a false regression. Heap,
Long Tasks, warmed-cache growth and the expected rendered-device count are
gated separately. Both raw reports and the comparison are always uploaded as
gated separately. Long-Task maximum/count/total checks use the same
relative-plus-absolute policy as timings. Both raw reports and the comparison are always uploaded as
the `large-house-performance` artifact, and the table is written to the GitHub
job summary.
@@ -71,3 +72,10 @@ changing the meaning of `large-house-v1`.
The `cleanFloor` entry ceiling is 160: the reviewed fixture currently warms
120 deterministic room/physical-body entries, and the extra 40 slots allow a
legitimate fixture extension without weakening the separate zero-growth gate.
The absolute switch-cycle/Long-Task ceilings include roughly 20–30% headroom
over the paired 2026-08-09 Ubuntu run where the unchanged base and candidate
both reached about 5.3 s / 2.45 s / 22 tasks / 9.9 s total under runner load.
The same-runner relative checks remain tighter for an actual candidate-only
regression; this prevents an overloaded but symmetric runner from turning an
absolute safety ceiling into a flaky code-regression signal.
+8 -4
View File
@@ -49,13 +49,17 @@
"stat": "median",
"maxRegressionRatio": 0.35,
"noiseAllowanceMs": 250,
"hardMaxMs": 5000
"hardMaxMs": 7000
}
},
"longTasks": {
"maxSingleMs": 2000,
"maxCountP95": 20,
"maxTotalP95Ms": 8000,
"maxSingleMs": 3000,
"maxSingleRegressionRatio": 0.3,
"maxSingleNoiseAllowanceMs": 250,
"maxCountP95": 30,
"maxCountRegressionRatio": 0.35,
"countNoiseAllowance": 3,
"maxTotalP95Ms": 12000,
"maxTotalRegressionRatio": 0.3,
"noiseAllowanceMs": 150
},
+22 -2
View File
@@ -95,8 +95,28 @@ export const evaluatePerformanceBudget = ({ candidate, baseline, budgets }) => {
return windows.length > 0 && windows.every((item) => item?.supported === true);
});
checks.push({ id: 'longTask.available', actual: longTasksAvailable ? 1 : 0, limit: 1, pass: longTasksAvailable });
checks.push(makeCheck('longTask.maxSingleMs', candidateLong.maxSingleMs, budgets.longTasks.maxSingleMs));
checks.push(makeCheck('longTask.countP95', candidateLong.countP95, budgets.longTasks.maxCountP95));
const singleRegressionLimit = relativeLimit(
baselineLong.maxSingleMs,
budgets.longTasks.maxSingleRegressionRatio,
budgets.longTasks.maxSingleNoiseAllowanceMs,
);
checks.push(makeCheck(
'longTask.maxSingleMs',
candidateLong.maxSingleMs,
Math.min(budgets.longTasks.maxSingleMs, singleRegressionLimit),
{ baseline: baselineLong.maxSingleMs, hardLimit: budgets.longTasks.maxSingleMs },
));
const countRegressionLimit = relativeLimit(
baselineLong.countP95,
budgets.longTasks.maxCountRegressionRatio,
budgets.longTasks.countNoiseAllowance,
);
checks.push(makeCheck(
'longTask.countP95',
candidateLong.countP95,
Math.min(budgets.longTasks.maxCountP95, countRegressionLimit),
{ baseline: baselineLong.countP95, hardLimit: budgets.longTasks.maxCountP95 },
));
const longRegressionLimit = relativeLimit(
baselineLong.totalP95Ms,
budgets.longTasks.maxTotalRegressionRatio,