diff --git a/docs.json b/docs.json index 37e3fce6..04444595 100644 --- a/docs.json +++ b/docs.json @@ -258,7 +258,151 @@ "pages": [ "flaky-tests/detection/new-test-monitor", "flaky-tests/detection/skipped-test-monitor", - "flaky-tests/detection/slow-test-monitor" + "flaky-tests/detection/slow-test-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor", + "flaky-tests/detection/timeout-inflation-monitor" ] }, "flaky-tests/detection/flag-as-flaky", diff --git a/flaky-tests/detection/index.mdx b/flaky-tests/detection/index.mdx index 05e07e37..a6fa7f38 100644 --- a/flaky-tests/detection/index.mdx +++ b/flaky-tests/detection/index.mdx @@ -12,11 +12,11 @@ Each monitor independently observes your test runs and tracks two states per tes For monitors whose action is **Classify test status** (referred to below as _health classification monitors_), the test's overall status is determined by combining all such monitors, with the most severe status winning: -| Priority | Status | Condition | -| -------- | ----------- | ---------------------------------------------------------------------------------------- | -| Highest | **Broken** | Any enabled broken-type monitor (failure rate or failure count) is active for this test | -| Middle | **Flaky** | Any enabled flaky-type monitor (failure rate, failure count, or pass-on-retry) is active | -| Lowest | **Healthy** | No active health classification monitor | +| Priority | Status | Condition | +| --- | --- | --- | +| Highest | **Broken** | Any enabled broken-type monitor (failure rate or failure count) is active for this test | +| Middle | **Flaky** | Any enabled flaky-type monitor (failure rate, failure count, or pass-on-retry) is active | +| Lowest | **Healthy** | No active health classification monitor | If a test triggers both a broken monitor and a flaky monitor simultaneously, it shows as **Broken**. When the broken monitor resolves (e.g., you fix the regression and the failure rate drops), the test transitions to **Flaky** if a flaky monitor is still active, or to **Healthy** if no health classification monitors remain active. @@ -41,21 +41,22 @@ Trunk groups monitors into two categories based on what they do when they activa ### Health Classification Monitors -| Monitor | What it detects | Available actions | Default state | -| -------------------------------------------- | ---------------------------------------------------------------------- | -------------------------------------------------------------------------------------------------------- | ------------- | -| [**Pass-on-Retry**](./pass-on-retry-monitor) | A test fails then passes on the same commit (retry after failure) | Classify (flaky) or [apply labels](../management/test-labels#automatic-labeling-from-monitors) | Enabled | -| [**Failure Rate**](./failure-rate-monitor) | Failure rate exceeds a configured percentage over a time window | Classify (flaky or broken) or [apply labels](../management/test-labels#automatic-labeling-from-monitors) | Disabled | -| [**Failure Count**](./failure-count-monitor) | A test accumulates a configured number of failures in a rolling window | Classify (flaky or broken) or [apply labels](../management/test-labels#automatic-labeling-from-monitors) | Disabled | +| Monitor | What it detects | Available actions | Default state | +| --- | --- | --- | --- | +| [**Pass-on-Retry**](./pass-on-retry-monitor) | A test fails then passes on the same commit (retry after failure) | Classify (flaky) or [apply labels](../management/test-labels#automatic-labeling-from-monitors) | Enabled | +| [**Failure Rate**](./failure-rate-monitor) | Failure rate exceeds a configured percentage over a time window | Classify (flaky or broken) or [apply labels](../management/test-labels#automatic-labeling-from-monitors) | Disabled | +| [**Failure Count**](./failure-count-monitor) | A test accumulates a configured number of failures in a rolling window | Classify (flaky or broken) or [apply labels](../management/test-labels#automatic-labeling-from-monitors) | Disabled | ### Lifecycle and Performance Monitors These monitors apply labels based on lifecycle events or performance characteristics. They do not classify tests as flaky or broken, and they do not contribute to the test's overall health status. -| Monitor | What it detects | Available actions | Default state | -| ------------------------------------------ | ---------------------------------------------------------------------------- | -------------------------------------------------------------------------- | ------------- | -| [**New Test**](./new-test-monitor) | A test case seen for the first time, tracked for a configurable grace period | [Apply labels](../management/test-labels#automatic-labeling-from-monitors) | Disabled | -| [**Skipped Test**](./skipped-test-monitor) | A test is consistently skipped across runs within a time window | [Apply labels](../management/test-labels#automatic-labeling-from-monitors) | Disabled | -| [**Slow Test**](./slow-test-monitor) | A test's average duration exceeds a configured threshold | [Apply labels](../management/test-labels#automatic-labeling-from-monitors) | Disabled | +| Monitor | What it detects | Available actions | Default state | +| --- | --- | --- | --- | +| [**New Test**](./new-test-monitor) | A test case seen for the first time, tracked for a configurable grace period | [Apply labels](../management/test-labels#automatic-labeling-from-monitors) | Disabled | +| [**Skipped Test**](./skipped-test-monitor) | A test is consistently skipped across runs within a time window | [Apply labels](../management/test-labels#automatic-labeling-from-monitors) | Disabled | +| [**Slow Test**](./slow-test-monitor) | A test's average duration exceeds a configured threshold | [Apply labels](../management/test-labels#automatic-labeling-from-monitors) | Disabled | +| [**Timeout Inflation**](./timeout-inflation-monitor) | A test's typical failure duration is much larger than its passing duration, exposing an inflated timeout | [Apply labels](../management/test-labels#automatic-labeling-from-monitors) | Disabled | You can run multiple monitors simultaneously. For example, you might use pass-on-retry to catch classic retry-based flakiness while also running failure rate monitors scoped to different branches. A common pattern is to pair a broken-type failure rate monitor (catching consistently failing tests) with a flaky-type failure rate monitor (catching intermittently failing tests). See [Failure Rate Monitor: Recommended Configurations](./failure-rate-monitor#recommended-configurations) for details. @@ -95,35 +96,33 @@ This is useful when you know a test is flaky but want to suppress the signal tem You can mute a monitor from the test case view in the Trunk app. When muting, you choose a duration: | Duration | -| -------- | -| 1 hour | -| 4 hours | +| --- | +| 1 hour | +| 4 hours | | 24 hours | -| 7 days | -| 30 days | +| 7 days | +| 30 days | While muted, the monitor is excluded from the test's status calculation. If the muted monitor was the only active health classification monitor, the test transitions from flaky to healthy for the duration of the mute. When the mute expires, the monitor is automatically included in the next status evaluation. If it's still active, the test will be flagged again. You can also unmute a monitor early from the test case view. {/* SCREENSHOT: Mute button and duration picker on the test case monitor list. -Show the test case detail page with a monitor's mute button visible, -and ideally the duration picker dropdown open. */} + Show the test case detail page with a monitor's mute button visible, + and ideally the duration picker dropdown open. */} - You can only mute a monitor that has already detected flaky behavior for a - test. If a monitor has never been active for a test, the mute option is - disabled. + You can only mute a monitor that has already detected flaky behavior for a test. If a monitor has never been active for a test, the mute option is disabled. ### When to Mute vs. Other Options -| Situation | Recommended action | -| ------------------------------------------------------------- | ---------------------------------------------------------------- | -| Fix is in progress and you want to suppress noise temporarily | **Mute** the monitor for a few days | -| Test is flaky but no automated monitor has caught it | Use [**Flag as Flaky**](./flag-as-flaky) to mark it as flaky | -| You want to stop a monitor from evaluating a test permanently | Adjust the monitor's branch scope or thresholds instead | -| You want to suppress all flaky signals for a test | Mute each active monitor individually, or address the root cause | +| Situation | Recommended action | +| --- | --- | +| Fix is in progress and you want to suppress noise temporarily | **Mute** the monitor for a few days | +| Test is flaky but no automated monitor has caught it | Use [**Flag as Flaky**](./flag-as-flaky) to mark it as flaky | +| You want to stop a monitor from evaluating a test permanently | Adjust the monitor's branch scope or thresholds instead | +| You want to suppress all flaky signals for a test | Mute each active monitor individually, or address the root cause | ## Variants @@ -131,4 +130,4 @@ If you run the same tests across different environments or architectures, you ca ## Detection Time -Detection of flaky tests is run automatically when test uploads are processed. From the time that a test with configured flake detection is uploaded, it will take at most 20 minutes for the flakiness to be detected. +Detection of flaky tests is run automatically when test uploads are processed. From the time that a test with configured flake detection is uploaded, it will take at most 20 minutes for the flakiness to be detected. \ No newline at end of file diff --git a/flaky-tests/detection/slow-test-monitor.mdx b/flaky-tests/detection/slow-test-monitor.mdx index 7654bbf7..59078a64 100644 --- a/flaky-tests/detection/slow-test-monitor.mdx +++ b/flaky-tests/detection/slow-test-monitor.mdx @@ -23,7 +23,7 @@ Once the monitor activates, detection evidence (the specific runs that triggered ## Configuration | Setting | Description | Default | -|---|---|---| +| --- | --- | --- | | Duration threshold | Duration (milliseconds) at the configured percentile that triggers detection | Required | | Percentile | Which duration percentile to evaluate, as a value between 0 and 1 (for example, 0.5 for p50) | Required | | Window | Time window (minutes) over which duration is measured | Required | @@ -60,9 +60,10 @@ Scope the monitor to branches where test duration matters most, such as `main` o ## Choosing Between Monitors | Goal | Recommended monitor | -|---|---| +| --- | --- | | Flag tests that are taking too long | Slow test monitor | +| Flag tests whose timeouts are far larger than they need to be | [Timeout inflation monitor](./timeout-inflation-monitor) | | Track recently added tests | New test monitor | | Detect tests consistently being skipped | Skipped test monitor | | Detect tests that fail then pass on retry | Pass-on-retry monitor | -| Alert on tests failing at a sustained rate | Failure rate monitor | +| Alert on tests failing at a sustained rate | Failure rate monitor | \ No newline at end of file diff --git a/flaky-tests/detection/timeout-inflation-monitor.mdx b/flaky-tests/detection/timeout-inflation-monitor.mdx new file mode 100644 index 00000000..d514f7a8 --- /dev/null +++ b/flaky-tests/detection/timeout-inflation-monitor.mdx @@ -0,0 +1,71 @@ +--- +title: "Timeout Inflation Monitor" +description: "Flag tests whose failure times sit far above their passing times, exposing timeouts that have ratcheted up beyond what the test actually needs." +og:title: "The Timeout Inflation Monitor in Trunk Flaky Tests" +--- + +The timeout inflation monitor detects tests whose failure durations are much larger than their passing durations. When a test consistently passes in a few seconds at p95 but takes ten times longer to fail, its failures aren't slow runs of a working test - they're a broken test sitting on an inflated timeout. The monitor surfaces these tests so you can bring the timeout down to something anchored in reality and get fast failure signal back. + +## When to Use This Monitor + +- **Tighten inflated timeouts:** Find tests where the configured timeout is far larger than the test ever needs when healthy, so you can ratchet the timeout back down. +- **Speed up broken-build feedback:** Cut the wall-clock cost of a broken test that fails, retries, and hangs against its timeout on every attempt. +- **Audit Playwright and other UI automation suites:** These suites tend to accumulate the largest timeouts and benefit the most from surfacing inflation. + +## How It Works + +For each test case, the monitor looks at recent runs within a configurable window and computes two durations: + +- The **passing duration** at a configured percentile (default p95) — a conservative upper bound on how long a healthy run takes. +- The **failing duration** at a configured percentile (default p50) — the typical case when the test goes red. + +If the failing duration is at least X times larger than the passing duration, and the absolute gap between the two exceeds the minimum absolute gap, the monitor activates and applies the configured labels. Stacking the slowest reasonable success against the typical failure means a single slow outlier can't fool the detector - the failures have to consistently land above the passes. + +## Configuration + +| Setting | Description | Default | +| --- | --- | --- | +| Pass percentile | Percentile of passing-run durations used as the healthy upper bound, as a value between 0 and 1 | 0.95 | +| Fail percentile | Percentile of failing-run durations compared against the passing baseline | 0.5 | +| Activation ratio | How many times larger the failing percentile must be than the passing percentile to activate | 2 | +| Min absolute gap | Minimum absolute gap (milliseconds) between the failing and passing percentiles before the monitor can activate | 3000 | +| Min pass samples | Minimum number of passing runs required inside the window before the monitor can activate | 5 | +| Min fail samples | Minimum number of failing runs required inside the window before the monitor can activate | 3 | +| Window | Time window (hours) over which passing and failing durations are collected | 168 | +| Resolution ratio | Ratio at or below which an active test automatically resolves | 1.5 | +| Branch scope | Branch names or glob patterns to monitor | All branches | +| Action | Apply labels (the only available action — this monitor does not classify) | Apply labels | + +### Pass and Fail Percentiles + +The pass percentile controls what counts as a realistic upper bound on a healthy run. p95 means 95% of a test's passing runs finish at or below the measured duration, so it captures the slowest reasonable success without being thrown off by a single outlier. + +The fail percentile controls what counts as a typical failure. p50 means half of a test's failing runs finish at or below the measured duration, so a small number of unusually fast failures can't hide a broken test that usually hangs on its timeout. + +Using p95 for passes and p50 for failures is deliberately conservative: it compares the slowest reasonable success against the typical failure and requires a clean gap between the two. + +### Activation Ratio and Minimum Absolute Gap + +The activation ratio is how much larger the fail percentile has to be than the pass percentile before the monitor flags the test. The default of 2 means failures have to typically take at least twice as long as the slowest reasonable pass. Raise it to only surface the worst offenders; lower it to catch milder inflation earlier. + +The minimum absolute gap is a floor in milliseconds so tiny durations don't trip the monitor. A test that passes in 10ms and fails in 30ms is technically a 3x ratio but not worth flagging. The default of 3000ms keeps sub-second tests quiet while still catching the multi-second inflation that dominates real CI cost. + +### Sample Sizes and Window + +The window is how far back the monitor looks for passing and failing runs. The default of 168 hours (one week) keeps the signal tied to recent behavior rather than ancient history. + +The minimum pass and fail sample sizes prevent the monitor from acting on thin data. Defaults are 5 passes and 3 failures inside the window, so a test needs enough of both outcomes for the percentile comparison to be meaningful. + +### Resolution Ratio + +Once a test is active, the monitor watches the fail/pass ratio on subsequent runs. When the ratio drops to or below this ratio, the monitor resolves the test and removes the labels it applied. The default of 1.5 means the test has to come back well under the activation threshold - not just barely - before it clears, which prevents flapping around the activation line. + +### Branch Scope + +Scope the monitor to branches where timeout inflation matters most, such as `main` or merge queue branches. Feature branches often have intentionally partial test runs or unusual infrastructure and are less useful for measuring inflation. + +## What to Do About an Inflated Timeout + +When the monitor flags a test, the fix is usually to bring the timeout down to something anchored in the test's real behavior. If the test passes in 2 seconds at p95, it does not need a 60 second timeout — a 3 second timeout gives it 50% more time than it ever uses when healthy, and future breakages will fail fast instead of hanging against the old ceiling. + +Bringing the timeout down turns a three-minute retry-and-hang cycle into a few seconds of actual signal, which is the point of the test suite in the first place. \ No newline at end of file