diff --git a/runner/docs/adr/0041-observability-stack.md b/runner/docs/adr/0041-observability-stack.md index 39220e41f..9cb185697 100644 --- a/runner/docs/adr/0041-observability-stack.md +++ b/runner/docs/adr/0041-observability-stack.md @@ -563,7 +563,8 @@ starting values, tuned after launch: preview-ready below 97 % (Tier-1) or 95 % ( over 1 h, evaluated per tier only with at least 10 non-abandoned previews in that hour; session start p95 above 20 s, evaluated only with at least 20 `ready` starts in the hour (below either floor the rule is not firing, so a firing alert resolves through the normal path); `at_capacity` above 5/h; 5xx above 1 % over -15 min; LiteLLM errors above 5 %; compile errors on one `ht_major` doubling day over +15 min, evaluated only with at least 100 requests after the exclusions (at 1 % one error exceeds the threshold only below 100 requests, so a single 500 cannot page); +LiteLLM errors above 5 % over 1 h, evaluated only with at least 20 non-denied calls (same reasoning at 5 %); compile errors on one `ht_major` doubling day over day; snapshot builds failing above 50 % per framework over 30 min with at least 10 failed; an embed above 20 % errors with more than 50 views in 24 h; backlog older than 2 h. diff --git a/runner/pipeline/o11y-alerts.test.mjs b/runner/pipeline/o11y-alerts.test.mjs index 313537d51..de847ffc0 100644 --- a/runner/pipeline/o11y-alerts.test.mjs +++ b/runner/pipeline/o11y-alerts.test.mjs @@ -18,6 +18,7 @@ const { evaluateAndNotify, slackPoster, escapeSlackMrkdwn, writeNewFingerprintPo const { atCapacityRule, fiveXxRateRule, + FIVE_XX_MIN_REQUESTS, previewReadyRateRule, PREVIEW_READY_MIN_SAMPLES, sessionStartP95Rule, @@ -25,6 +26,7 @@ const { embedErrorRateRule, compileErrorDoublingRule, litellmErrorRateRule, + LITELLM_MIN_CALLS, snapshotBuildFailedRateRule, backlogAgeRule, rejectedKeyRule, @@ -988,6 +990,58 @@ test("fiveXxRateRule: points written before the status column existed (no reason assert.equal(result.firing, false, result.detail); }); +function apiRows(ok, bad) { + return [ + { metric: "api.request", route_class: "api/demos", outcome: "2xx", count: ok }, + { metric: "api.request", route_class: "api/demos", outcome: "5xx", count: bad }, + ]; +} + +// The fake engine only proves the SQL passes the guard and the arithmetic is right, not that Analytics Engine accepts it. +test("fiveXxRateRule sample floor: below 100 requests even 100% 5xx does not fire, at 100 it fires above 1%", async () => { + assert.equal(FIVE_XX_MIN_REQUESTS, 100); + const single = await fiveXxRateRule({}, makeFakeAeQuery(apiRows(0, 1)).queryFn); + assert.equal(single.firing, false, "one 500 out of one request is 100% but says nothing"); + assert.match(single.detail, /only 1 request\(s\), below the 100 needed/); + + const allBad = await fiveXxRateRule({}, makeFakeAeQuery(apiRows(0, 99)).queryFn); + assert.equal(allBad.firing, false, allBad.detail); + + const oneInNinetyNine = await fiveXxRateRule({}, makeFakeAeQuery(apiRows(98, 1)).queryFn); + assert.equal(oneInNinetyNine.firing, false, "1/99 is 1.01% but below the floor"); + + const atFloorOne = await fiveXxRateRule({}, makeFakeAeQuery(apiRows(99, 1)).queryFn); + assert.equal(atFloorOne.firing, false, "1/100 is exactly 1%, not above it: a single error never fires at the floor"); + + const atFloor = await fiveXxRateRule({}, makeFakeAeQuery(apiRows(98, 2)).queryFn); + assert.equal(atFloor.firing, true, atFloor.detail); + assert.match(atFloor.detail, /^2\.00% 5xx over the last 15 min \(2\/100,/); + + const healthy = await fiveXxRateRule({}, makeFakeAeQuery(apiRows(500, 0)).queryFn); + assert.equal(healthy.firing, false); +}); + +test("fiveXxRateRule sample floor: the floor counts the ADJUSTED total, not the raw one", async () => { + const rows = [ + ...apiRows(0, 5), + { metric: "api.request", route_class: "api/session", outcome: "5xx", count: 200 }, + { metric: "session.start", outcome: "at_capacity", count: 200 }, + ]; + const result = await fiveXxRateRule({}, makeFakeAeQuery(rows).queryFn); + assert.equal(result.firing, false, "205 raw requests but only 5 after exclusions"); +}); + +test("fiveXxRateRule sample floor: a firing alert resolves once traffic drops below the floor", async () => { + const writer = fakeInboxWriter(); + const slackCalls = []; + const deps = { inboxWriter: writer, postSlack: async (t) => slackCalls.push(t), aeSink: fakeAeSink(), commonAttrs: COMMON_ATTRS, nowMs: 1000 }; + const firing = await fiveXxRateRule({}, makeFakeAeQuery(apiRows(95, 5)).queryFn); + assert.equal(await evaluateAndNotify(firing, deps), "fired"); + const quiet = await fiveXxRateRule({}, makeFakeAeQuery(apiRows(0, 1)).queryFn); + assert.equal(await evaluateAndNotify(quiet, { ...deps, nowMs: 2000 }), "resolved"); + assert.match(slackCalls[1], /resolved/); +}); + test("previewReadyRateRule: tier 1 below 97% fires, tier 2 within threshold does not (mixed)", async () => { const fake = makeFakeAeQuery([ { metric: "preview.ready_ms", tier: "1", outcome: "ready", count: 90 }, @@ -1308,6 +1362,51 @@ test("litellmErrorRateRule: chat.answer + theme.ai combined over 5% fires; under assert.equal(underResult.firing, false, underResult.detail); // 2/200 = 1% }); +function gatewayRows(chatOk, chatErr, extra = []) { + return [ + { metric: "chat.answer", outcome: "answered", count: chatOk }, + { metric: "chat.answer", outcome: "error", count: chatErr }, + ...extra, + ]; +} + +test("litellmErrorRateRule sample floor: below 20 calls even 100% errors does not fire, at 20 it fires above 5%", async () => { + assert.equal(LITELLM_MIN_CALLS, 20); + const single = await litellmErrorRateRule({}, makeFakeAeQuery(gatewayRows(0, 1)).queryFn); + assert.equal(single.firing, false); + assert.match(single.detail, /only 1 call\(s\), below the 20 needed/); + + const oneInNineteen = await litellmErrorRateRule({}, makeFakeAeQuery(gatewayRows(18, 1)).queryFn); + assert.equal(oneInNineteen.firing, false, "1/19 is 5.26% but below the floor"); + + const oneAtFloor = await litellmErrorRateRule({}, makeFakeAeQuery(gatewayRows(19, 1)).queryFn); + assert.equal(oneAtFloor.firing, false, "1/20 is exactly 5%: a single error never fires at the floor"); + + const atFloor = await litellmErrorRateRule({}, makeFakeAeQuery(gatewayRows(18, 2)).queryFn); + assert.equal(atFloor.firing, true, atFloor.detail); + assert.match(atFloor.detail, /^10\.00% gateway errors over the last hour \(2\/20,/); + + const healthy = await litellmErrorRateRule({}, makeFakeAeQuery(gatewayRows(100, 0)).queryFn); + assert.equal(healthy.firing, false); +}); + +test("litellmErrorRateRule sample floor: denied requests do not count toward the floor", async () => { + const rows = gatewayRows(0, 2, [{ metric: "chat.answer", outcome: "denied", count: 500 }]); + const result = await litellmErrorRateRule({}, makeFakeAeQuery(rows).queryFn); + assert.equal(result.firing, false, "2 real calls plus 500 denied is still below the floor"); +}); + +test("litellmErrorRateRule sample floor: a firing alert resolves once calls drop below the floor", async () => { + const writer = fakeInboxWriter(); + const slackCalls = []; + const deps = { inboxWriter: writer, postSlack: async (t) => slackCalls.push(t), aeSink: fakeAeSink(), commonAttrs: COMMON_ATTRS, nowMs: 1000 }; + const firing = await litellmErrorRateRule({}, makeFakeAeQuery(gatewayRows(18, 2)).queryFn); + assert.equal(await evaluateAndNotify(firing, deps), "fired"); + const quiet = await litellmErrorRateRule({}, makeFakeAeQuery(gatewayRows(0, 1)).queryFn); + assert.equal(await evaluateAndNotify(quiet, { ...deps, nowMs: 2000 }), "resolved"); + assert.match(slackCalls[1], /resolved/); +}); + // `denied` (a rate-limit/budget refusal at `index.ts`'s own gate — never // reaches the LiteLLM gateway) must be excluded from the denominator. // Including it only ever drags the computed error% down, which can mask a diff --git a/runner/workers/o11y/src/alerts/rules.ts b/runner/workers/o11y/src/alerts/rules.ts index 24b79c825..a9a2d8ad5 100644 --- a/runner/workers/o11y/src/alerts/rules.ts +++ b/runner/workers/o11y/src/alerts/rules.ts @@ -176,6 +176,9 @@ export async function atCapacityRule(env: Env, queryFn: AeQueryFn = runAeQuery): * `reason`, and only the 503 is dropped. A build-failed 500 there still counts. */ const FIVE_XX_STILL_BUILDING_ROUTE_CLASSES = ["d/:id", "embed/:id"]; +/** Requests (after the exclusions below) in the 15-minute window under which the rule is not evaluated: at a 1 % threshold one error only exceeds it when the total is under 100, so a floor of 100 means a single 500 can never page. */ +export const FIVE_XX_MIN_REQUESTS = 100; + export async function fiveXxRateRule(env: Env, queryFn: AeQueryFn = runAeQuery): Promise { const windowMs = 15 * 60 * 1000; const routeClassCol = col("route_class"); @@ -208,12 +211,14 @@ export async function fiveXxRateRule(env: Env, queryFn: AeQueryFn = runAeQuery): const adjustedFiveXx = Math.max(0, fiveXx - deliberate); const adjustedTotal = Math.max(0, total - deliberate); const pct = ratio(adjustedFiveXx, adjustedTotal) * 100; + const enoughSamples = adjustedTotal >= FIVE_XX_MIN_REQUESTS; return { rule: "api-5xx-rate", - firing: adjustedTotal > 0 && pct > 1, + firing: enoughSamples && pct > 1, detail: `${pct.toFixed(2)}% 5xx over the last 15 min (${adjustedFiveXx}/${adjustedTotal}, threshold 1%, ` + - `excludes at-capacity/container-starting/chat-theme-gateway refusals and still-building 503s)`, + `excludes at-capacity/container-starting/chat-theme-gateway refusals and still-building 503s)` + + (enoughSamples ? "" : `; only ${adjustedTotal} request(s), below the ${FIVE_XX_MIN_REQUESTS} needed to evaluate the rate`), }; } @@ -376,6 +381,9 @@ export async function snapshotBuildFailedRateRule(env: Env, queryFn: AeQueryFn = // ---- LiteLLM errors: above 5% (chat.answer + theme.ai, both gateway ------- // call sites; window: 1h, same reasoning as session-start) ------------------ +/** Gateway calls (denied excluded) in the hour under which the rule is not evaluated: at a 5 % threshold one error only exceeds it when the total is under 20, so a floor of 20 means a single gateway error can never page. */ +export const LITELLM_MIN_CALLS = 20; + export async function litellmErrorRateRule(env: Env, queryFn: AeQueryFn = runAeQuery): Promise { const windowMs = HOUR_MS; const [chat, theme] = await Promise.all([ @@ -390,10 +398,13 @@ export async function litellmErrorRateRule(env: Env, queryFn: AeQueryFn = runAeQ .reduce((sum, [, c]) => sum + c, 0); const errors = (chat.get("error") ?? 0) + (theme.get("error") ?? 0); const pct = ratio(errors, total) * 100; + const enoughSamples = total >= LITELLM_MIN_CALLS; return { rule: "litellm-error-rate", - firing: total > 0 && pct > 5, - detail: `${pct.toFixed(2)}% gateway errors over the last hour (${errors}/${total}, threshold 5%)`, + firing: enoughSamples && pct > 5, + detail: + `${pct.toFixed(2)}% gateway errors over the last hour (${errors}/${total}, threshold 5%)` + + (enoughSamples ? "" : `; only ${total} call(s), below the ${LITELLM_MIN_CALLS} needed to evaluate the rate`), }; }