From 3b1aeaa166944b76b6a811761d9b76a5e06f2028 Mon Sep 17 00:00:00 2001 From: Jorge Cajas Date: Wed, 22 Jul 2026 09:55:54 +0800 Subject: [PATCH] test(engine,resolve): min-based scaling-ratio timing to fix Windows CI flake TestResponseShapePython_NearLinearScaling failed on Windows CI (run #4, pre-existing flake, not caused by the epic under test): median-of-5 timing at N=1000 got dragged to 130ms by runner contention, pushing the ratio to 32.28x over the 25x bound with no real algorithmic regression. Wall-clock noise (GC, scheduler contention, co-scheduled load) only ever ADDS time to a sample, never subtracts it, so a noisy median has no ceiling on how far it can be pulled upward. Switch both scaling-ratio tests in the repo (the only two computing a timed-corpus ratio: response_shape_python and resolve/scale_bench) from median-of-samples to min-of-samples: the minimum is the least-contended observation and the best available estimate of true algorithmic cost, while a genuine O(n^2)/O(n^2) regression is still ~100x slower on every run including the cleanest one, so detection power is unchanged. - response_shape_python_perf_test.go: median->min, samples 5->8, bound 25->30 (observed min-based ratio locally: 9.77-10.31x vs 10x expected linear, comfortably under 30 and far under the 100x quadratic signal). - resolve/scale_bench_test.go: same median->min swap, samples 5->8, bound kept at 40 (observed min-based ratio locally: 11.6-12.7x vs ~15x expected n-log-n, comfortably under 40). Repo-wide audit (grep for ratio :=/ratio = timing assertions in *_test.go) confirms these are the only two tests using this pattern; no other siblings needed hardening. --- .../engine/response_shape_python_perf_test.go | 89 ++++++++++++++----- internal/resolve/scale_bench_test.go | 57 ++++++++---- 2 files changed, 103 insertions(+), 43 deletions(-) diff --git a/internal/engine/response_shape_python_perf_test.go b/internal/engine/response_shape_python_perf_test.go index 4f7f574a6..55e9bff19 100644 --- a/internal/engine/response_shape_python_perf_test.go +++ b/internal/engine/response_shape_python_perf_test.go @@ -174,21 +174,54 @@ func TestResponseShapePython_MemoizationIdentical(t *testing.T) { // regexp.MustCompile + full rescan this ratio blew up; with memoization it is // ~linear. // -// De-flaking rationale (#5607): the original compared N=100 vs N=400 (a 4x -// corpus) against a 9x bound. That gap is too narrow — for a linear extractor -// the expected ratio is ~4x and the headroom to the 9x bound is only ~2.25x, -// so ordinary CI jitter (cold caches, GC pauses, co-scheduled load) routinely -// pushed it past 9x (9.28x observed on ubuntu-latest) with no real regression. +// De-flaking rationale (#5607, revised #5751-followup): the original compared +// N=100 vs N=400 (a 4x corpus) against a 9x bound. That gap is too narrow — +// for a linear extractor the expected ratio is ~4x and the headroom to the 9x +// bound is only ~2.25x, so ordinary CI jitter (cold caches, GC pauses, +// co-scheduled load) routinely pushed it past 9x (9.28x observed on +// ubuntu-latest) with no real regression. // -// We instead use a 10x corpus gap (N=100 vs N=1000) and take the MEDIAN of a -// few timing runs to damp outliers. The math is what makes this robust: +// We use a 10x corpus gap (N=100 vs N=1000). The math is what makes the +// signal meaningful regardless of how we aggregate samples: // // - Linear (O(n)): 1000/100 = 10x time → expected ratio ≈ 10 // - Quadratic (O(n²)): 1000²/100² = 100x time → expected ratio ≈ 100 // -// A 25x bound sits ~2.5x above the linear expectation (generous slack for -// fixed overhead and noise) yet ~4x below the quadratic signal, so a genuine -// O(n²) reintroduction still fails loudly (~100x) while CI jitter never does. +// Aggregation was originally the MEDIAN of a few timing runs. That still +// flaked on a slow/contended Windows CI runner (#5751 follow-up, run #4: +// N=1000 median inflated to 130ms, ratio=32.28 > the then-25x bound) even +// though the epic's own change set wasn't in that delta — i.e. it was a pure +// runner-noise false positive. The problem with the median is that +// contention/GC/scheduler noise on a wall-clock timer only ever ADDS time to +// a sample; it never subtracts. A handful of contended runs can drag the +// median itself upward with no ceiling, so "median of 5" doesn't actually +// bound the noise contribution. +// +// We now take the MINIMUM of the per-N samples instead. The minimum is the +// least-contended observation in the batch, so it is the best available +// estimate of true algorithmic cost: any noise event can only push a sample +// *away* from the true cost (upward), never below it, so the smallest sample +// is the one noise corrupted least. The ratio of minimums is therefore far +// more stable across noisy runners than the ratio of medians, without giving +// up any ability to catch a real quadratic regression — an O(n²) code path +// is *always* ~100x slower at N=1000 vs N=100, on every single run including +// the least-contended one, so its minimum ratio is just as damning as its +// median ratio. +// +// samples is bumped 5→8 to give the minimum more draws (more chances to +// catch a clean, uncontended run) while staying fast (~8 corpus passes at +// N=100 + N=1000 is well under a second on any CI tier). +// +// Threshold: with min-based timing the ratio should sit close to the true +// ~10x linear expectation (the #5751 Windows failure's true cost was ~10x; +// only the median-of-noisy-samples aggregation pushed it to 32x). A 30x bound +// gives 3x headroom over the 10x linear expectation — generous slack for +// fixed overhead, GC, and residual runner contention that even the min can't +// fully filter out — while staying ~3.3x below the quadratic signal (100x), +// which is still a wide, unambiguous separation: a genuine O(n²) +// reintroduction produces a ratio an order of magnitude past this bound, so +// it fails loudly and unambiguously while realistic Windows-runner jitter on +// a linear implementation does not. func TestResponseShapePython_NearLinearScaling(t *testing.T) { if testing.Short() { t.Skip("skipping timing-sensitive scaling test in -short mode") @@ -196,15 +229,15 @@ func TestResponseShapePython_NearLinearScaling(t *testing.T) { const ( smallN = 100 largeN = 1000 // 10x the corpus → linear≈10x time, quadratic≈100x - samples = 5 // median of N runs damps CI outliers - bound = 25.0 // ~2.5x over linear (10x), ~4x under quadratic (100x) + samples = 8 // min of N runs damps CI outliers without masking true cost + bound = 30.0 // ~3x over linear (10x), ~3.3x under quadratic (100x) ) - timeSmall := medianCorpus(t, smallN, samples) - timeLarge := medianCorpus(t, largeN, samples) + timeSmall := minCorpus(t, smallN, samples) + timeLarge := minCorpus(t, largeN, samples) // +1µs floor on the denominator guards against a 0ns small measurement. ratio := float64(timeLarge) / float64(timeSmall+time.Microsecond) - t.Logf("scaling: N=%d median %v, N=%d median %v, ratio=%.2f (linear≈10.0, quadratic≈100.0)", + t.Logf("scaling: N=%d min %v, N=%d min %v, ratio=%.2f (linear≈10.0, quadratic≈100.0)", smallN, timeSmall, largeN, timeLarge, ratio) if ratio > bound { t.Fatalf("super-linear scaling detected: %dx corpus took %.2fx time (want <%.0fx); "+ @@ -212,17 +245,25 @@ func TestResponseShapePython_NearLinearScaling(t *testing.T) { } } -// medianCorpus times the corpus pass `samples` times for the given corpus size -// and returns the median, which is far more stable than a single timing under -// CI noise (a single GC pause or co-scheduled spike can multiply one run). -func medianCorpus(t *testing.T, nClasses, samples int) time.Duration { +// minCorpus times the corpus pass `samples` times for the given corpus size +// and returns the minimum observed duration. Wall-clock noise (GC pauses, +// scheduler contention, co-scheduled CI load) only ever adds time to a +// sample, never subtracts it, so the minimum of a batch is the sample least +// corrupted by noise and the best available estimate of true algorithmic +// cost. This makes the ratio of minimums far more stable across noisy +// runners (notably slow/contended Windows CI) than the ratio of medians, +// while still reliably surfacing a genuine O(n²) regression: a quadratic +// code path is slower by construction on every run, including the cleanest +// one, so its minimum ratio is just as damning as its median ratio. +func minCorpus(t *testing.T, nClasses, samples int) time.Duration { t.Helper() - runs := make([]time.Duration, samples) - for i := range runs { - runs[i] = timeCorpus(t, nClasses) + best := timeCorpus(t, nClasses) + for i := 1; i < samples; i++ { + if d := timeCorpus(t, nClasses); d < best { + best = d + } } - sort.Slice(runs, func(a, b int) bool { return runs[a] < runs[b] }) - return runs[len(runs)/2] + return best } func timeCorpus(t *testing.T, nClasses int) time.Duration { diff --git a/internal/resolve/scale_bench_test.go b/internal/resolve/scale_bench_test.go index 465287192..a715eb0f8 100644 --- a/internal/resolve/scale_bench_test.go +++ b/internal/resolve/scale_bench_test.go @@ -14,7 +14,6 @@ package resolve import ( "fmt" - "sort" "testing" "time" @@ -300,17 +299,34 @@ func TestMergeModuleBatch_SingleBatch(t *testing.T) { // (12.25x observed on a loaded macOS runner). It was the third perf-ratio // scaling test to flake this way after #5607 and #5628. // -// We apply the #5607 pattern: a wide 10x corpus gap and the MEDIAN of several -// timing runs per size. The wide gap is what makes the bound robust — it places -// a large window between the expected complexity and the next-worse one: +// We apply the #5607 pattern: a wide 10x corpus gap. The wide gap is what +// makes the bound robust — it places a large window between the expected +// complexity and the next-worse one: // // - O(N log N): 1000/100 × log(1000)/log(100) ≈ 10 × 1.5 = 15x // - O(N²): 1000²/100² = 100x // +// Aggregation across samples was originally the MEDIAN of several timing +// runs per size, but that still flaked on noisy CI runners (see +// internal/engine/response_shape_python_perf_test.go's #5751 Windows +// follow-up: a median-of-5 got dragged up by runner contention with no +// algorithmic change). Wall-clock noise (GC pauses, scheduler contention, +// co-scheduled load) only ever ADDS time to a sample, never subtracts it, so +// a handful of contended runs can drag the median upward with no ceiling — +// the median doesn't actually bound the noise contribution. We now take the +// MINIMUM of the per-size samples instead: it is the least-contended +// observation, the best available estimate of true algorithmic cost, and a +// genuine O(N²) code path is still ~100x slower at N=1000 vs N=100 on every +// run including the cleanest one, so switching to min loses none of the +// regression-detection power. samples is bumped 5→8 to give the minimum more +// draws at a clean run while staying fast. +// // A 40x bound sits ~2.7x above the n-log-n expectation (generous slack for // fixed overhead and CI noise) yet ~2.5x below the quadratic signal, so a // genuine O(N²) reintroduction still fails loudly (~100x) while CI jitter -// never does. +// never does. With min-based timing this margin is if anything more +// comfortable than before, since min filters out upward noise directly +// rather than relying on the bound alone to absorb it. func TestBuildIndexFromModules_SubQuadratic(t *testing.T) { if testing.Short() { t.Skip("skipping scaling test in short mode") @@ -320,32 +336,35 @@ func TestBuildIndexFromModules_SubQuadratic(t *testing.T) { entPerMod = 50 smallN = 100 largeN = 1000 // 10x corpus → n-log-n ≈ 15x time, quadratic ≈ 100x - samples = 5 // median of N runs damps CI outliers + samples = 8 // min of N runs damps CI outliers without masking true cost bound = 40.0 // ~2.7x over n-log-n (15x), ~2.5x under quadratic (100x) ) - // median times BuildIndexFromModules `samples` times at the given module - // count and returns the median, which is far more stable than a single - // timing under CI noise (one GC pause or co-scheduled spike can multiply a - // lone run). - median := func(numMods int) int64 { + // min times BuildIndexFromModules `samples` times at the given module + // count and returns the minimum, which is far more stable than the + // median under CI noise: a GC pause or co-scheduled spike can only ever + // inflate a sample, so the smallest sample is the one least corrupted by + // noise and the best estimate of true algorithmic cost. + minRun := func(numMods int) int64 { modules, _ := syntheticModules(numMods, entPerMod) - runs := make([]int64, samples) - for i := range runs { + best := int64(-1) + for i := 0; i < samples; i++ { t0 := nanoTime() _ = BuildIndexFromModules(modules, 0) - runs[i] = nanoTime() - t0 + d := nanoTime() - t0 + if best < 0 || d < best { + best = d + } } - sort.Slice(runs, func(a, b int) bool { return runs[a] < runs[b] }) - return runs[len(runs)/2] + return best } - tSmall := median(smallN) - tLarge := median(largeN) + tSmall := minRun(smallN) + tLarge := minRun(largeN) // +1µs floor on the denominator guards against a 0ns small measurement. ratio := float64(tLarge) / float64(tSmall+1000) - t.Logf("%d-mod median: %dns %d-mod median: %dns ratio: %.2fx (n-log-n≈15.0, quadratic≈100.0)", + t.Logf("%d-mod min: %dns %d-mod min: %dns ratio: %.2fx (n-log-n≈15.0, quadratic≈100.0)", smallN, tSmall, largeN, tLarge, ratio) if ratio > bound {