Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 2 additions & 1 deletion runner/docs/adr/0041-observability-stack.md
Original file line number Diff line number Diff line change
Expand Up @@ -563,7 +563,8 @@ starting values, tuned after launch: preview-ready below 97 % (Tier-1) or 95 % (
over 1 h, evaluated per tier only with at least 10 non-abandoned previews in that hour;
session start p95 above 20 s, evaluated only with at least 20 `ready` starts in the hour
(below either floor the rule is not firing, so a firing alert resolves through the normal path); `at_capacity` above 5/h; 5xx above 1 % over
15 min; LiteLLM errors above 5 %; compile errors on one `ht_major` doubling day over
15 min, evaluated only with at least 100 requests after the exclusions (at 1 % one error exceeds the threshold only below 100 requests, so a single 500 cannot page);
LiteLLM errors above 5 % over 1 h, evaluated only with at least 20 non-denied calls (same reasoning at 5 %); compile errors on one `ht_major` doubling day over
day; snapshot builds failing above 50 % per framework over 30 min with at least 10
failed; an embed above 20 % errors with more than 50 views in 24 h; backlog older
than 2 h.
Expand Down
99 changes: 99 additions & 0 deletions runner/pipeline/o11y-alerts.test.mjs
Original file line number Diff line number Diff line change
Expand Up @@ -18,13 +18,15 @@ const { evaluateAndNotify, slackPoster, escapeSlackMrkdwn, writeNewFingerprintPo
const {
atCapacityRule,
fiveXxRateRule,
FIVE_XX_MIN_REQUESTS,
previewReadyRateRule,
PREVIEW_READY_MIN_SAMPLES,
sessionStartP95Rule,
SESSION_START_P95_MIN_SAMPLES,
embedErrorRateRule,
compileErrorDoublingRule,
litellmErrorRateRule,
LITELLM_MIN_CALLS,
snapshotBuildFailedRateRule,
backlogAgeRule,
rejectedKeyRule,
Expand Down Expand Up @@ -988,6 +990,58 @@ test("fiveXxRateRule: points written before the status column existed (no reason
assert.equal(result.firing, false, result.detail);
});

function apiRows(ok, bad) {
return [
{ metric: "api.request", route_class: "api/demos", outcome: "2xx", count: ok },
{ metric: "api.request", route_class: "api/demos", outcome: "5xx", count: bad },
];
}

// The fake engine only proves the SQL passes the guard and the arithmetic is right, not that Analytics Engine accepts it.
test("fiveXxRateRule sample floor: below 100 requests even 100% 5xx does not fire, at 100 it fires above 1%", async () => {
assert.equal(FIVE_XX_MIN_REQUESTS, 100);
const single = await fiveXxRateRule({}, makeFakeAeQuery(apiRows(0, 1)).queryFn);
assert.equal(single.firing, false, "one 500 out of one request is 100% but says nothing");
assert.match(single.detail, /only 1 request\(s\), below the 100 needed/);

const allBad = await fiveXxRateRule({}, makeFakeAeQuery(apiRows(0, 99)).queryFn);
assert.equal(allBad.firing, false, allBad.detail);

const oneInNinetyNine = await fiveXxRateRule({}, makeFakeAeQuery(apiRows(98, 1)).queryFn);
assert.equal(oneInNinetyNine.firing, false, "1/99 is 1.01% but below the floor");

const atFloorOne = await fiveXxRateRule({}, makeFakeAeQuery(apiRows(99, 1)).queryFn);
assert.equal(atFloorOne.firing, false, "1/100 is exactly 1%, not above it: a single error never fires at the floor");

const atFloor = await fiveXxRateRule({}, makeFakeAeQuery(apiRows(98, 2)).queryFn);
assert.equal(atFloor.firing, true, atFloor.detail);
assert.match(atFloor.detail, /^2\.00% 5xx over the last 15 min \(2\/100,/);

const healthy = await fiveXxRateRule({}, makeFakeAeQuery(apiRows(500, 0)).queryFn);
assert.equal(healthy.firing, false);
});

test("fiveXxRateRule sample floor: the floor counts the ADJUSTED total, not the raw one", async () => {
const rows = [
...apiRows(0, 5),
{ metric: "api.request", route_class: "api/session", outcome: "5xx", count: 200 },
{ metric: "session.start", outcome: "at_capacity", count: 200 },
];
const result = await fiveXxRateRule({}, makeFakeAeQuery(rows).queryFn);
assert.equal(result.firing, false, "205 raw requests but only 5 after exclusions");
});

test("fiveXxRateRule sample floor: a firing alert resolves once traffic drops below the floor", async () => {
const writer = fakeInboxWriter();
const slackCalls = [];
const deps = { inboxWriter: writer, postSlack: async (t) => slackCalls.push(t), aeSink: fakeAeSink(), commonAttrs: COMMON_ATTRS, nowMs: 1000 };
const firing = await fiveXxRateRule({}, makeFakeAeQuery(apiRows(95, 5)).queryFn);
assert.equal(await evaluateAndNotify(firing, deps), "fired");
const quiet = await fiveXxRateRule({}, makeFakeAeQuery(apiRows(0, 1)).queryFn);
assert.equal(await evaluateAndNotify(quiet, { ...deps, nowMs: 2000 }), "resolved");
assert.match(slackCalls[1], /resolved/);
});

test("previewReadyRateRule: tier 1 below 97% fires, tier 2 within threshold does not (mixed)", async () => {
const fake = makeFakeAeQuery([
{ metric: "preview.ready_ms", tier: "1", outcome: "ready", count: 90 },
Expand Down Expand Up @@ -1308,6 +1362,51 @@ test("litellmErrorRateRule: chat.answer + theme.ai combined over 5% fires; under
assert.equal(underResult.firing, false, underResult.detail); // 2/200 = 1%
});

function gatewayRows(chatOk, chatErr, extra = []) {
return [
{ metric: "chat.answer", outcome: "answered", count: chatOk },
{ metric: "chat.answer", outcome: "error", count: chatErr },
...extra,
];
}

test("litellmErrorRateRule sample floor: below 20 calls even 100% errors does not fire, at 20 it fires above 5%", async () => {
assert.equal(LITELLM_MIN_CALLS, 20);
const single = await litellmErrorRateRule({}, makeFakeAeQuery(gatewayRows(0, 1)).queryFn);
assert.equal(single.firing, false);
assert.match(single.detail, /only 1 call\(s\), below the 20 needed/);

const oneInNineteen = await litellmErrorRateRule({}, makeFakeAeQuery(gatewayRows(18, 1)).queryFn);
assert.equal(oneInNineteen.firing, false, "1/19 is 5.26% but below the floor");

const oneAtFloor = await litellmErrorRateRule({}, makeFakeAeQuery(gatewayRows(19, 1)).queryFn);
assert.equal(oneAtFloor.firing, false, "1/20 is exactly 5%: a single error never fires at the floor");

const atFloor = await litellmErrorRateRule({}, makeFakeAeQuery(gatewayRows(18, 2)).queryFn);
assert.equal(atFloor.firing, true, atFloor.detail);
assert.match(atFloor.detail, /^10\.00% gateway errors over the last hour \(2\/20,/);

const healthy = await litellmErrorRateRule({}, makeFakeAeQuery(gatewayRows(100, 0)).queryFn);
assert.equal(healthy.firing, false);
});

test("litellmErrorRateRule sample floor: denied requests do not count toward the floor", async () => {
const rows = gatewayRows(0, 2, [{ metric: "chat.answer", outcome: "denied", count: 500 }]);
const result = await litellmErrorRateRule({}, makeFakeAeQuery(rows).queryFn);
assert.equal(result.firing, false, "2 real calls plus 500 denied is still below the floor");
});

test("litellmErrorRateRule sample floor: a firing alert resolves once calls drop below the floor", async () => {
const writer = fakeInboxWriter();
const slackCalls = [];
const deps = { inboxWriter: writer, postSlack: async (t) => slackCalls.push(t), aeSink: fakeAeSink(), commonAttrs: COMMON_ATTRS, nowMs: 1000 };
const firing = await litellmErrorRateRule({}, makeFakeAeQuery(gatewayRows(18, 2)).queryFn);
assert.equal(await evaluateAndNotify(firing, deps), "fired");
const quiet = await litellmErrorRateRule({}, makeFakeAeQuery(gatewayRows(0, 1)).queryFn);
assert.equal(await evaluateAndNotify(quiet, { ...deps, nowMs: 2000 }), "resolved");
assert.match(slackCalls[1], /resolved/);
});

// `denied` (a rate-limit/budget refusal at `index.ts`'s own gate — never
// reaches the LiteLLM gateway) must be excluded from the denominator.
// Including it only ever drags the computed error% down, which can mask a
Expand Down
19 changes: 15 additions & 4 deletions runner/workers/o11y/src/alerts/rules.ts
Original file line number Diff line number Diff line change
Expand Up @@ -176,6 +176,9 @@ export async function atCapacityRule(env: Env, queryFn: AeQueryFn = runAeQuery):
* `reason`, and only the 503 is dropped. A build-failed 500 there still counts. */
const FIVE_XX_STILL_BUILDING_ROUTE_CLASSES = ["d/:id", "embed/:id"];

/** Requests (after the exclusions below) in the 15-minute window under which the rule is not evaluated: at a 1 % threshold one error only exceeds it when the total is under 100, so a floor of 100 means a single 500 can never page. */
export const FIVE_XX_MIN_REQUESTS = 100;

export async function fiveXxRateRule(env: Env, queryFn: AeQueryFn = runAeQuery): Promise<RuleResult> {
const windowMs = 15 * 60 * 1000;
const routeClassCol = col("route_class");
Expand Down Expand Up @@ -208,12 +211,14 @@ export async function fiveXxRateRule(env: Env, queryFn: AeQueryFn = runAeQuery):
const adjustedFiveXx = Math.max(0, fiveXx - deliberate);
const adjustedTotal = Math.max(0, total - deliberate);
const pct = ratio(adjustedFiveXx, adjustedTotal) * 100;
const enoughSamples = adjustedTotal >= FIVE_XX_MIN_REQUESTS;
return {
rule: "api-5xx-rate",
firing: adjustedTotal > 0 && pct > 1,
firing: enoughSamples && pct > 1,
detail:
`${pct.toFixed(2)}% 5xx over the last 15 min (${adjustedFiveXx}/${adjustedTotal}, threshold 1%, ` +
`excludes at-capacity/container-starting/chat-theme-gateway refusals and still-building 503s)`,
`excludes at-capacity/container-starting/chat-theme-gateway refusals and still-building 503s)` +
(enoughSamples ? "" : `; only ${adjustedTotal} request(s), below the ${FIVE_XX_MIN_REQUESTS} needed to evaluate the rate`),
};
}

Expand Down Expand Up @@ -376,6 +381,9 @@ export async function snapshotBuildFailedRateRule(env: Env, queryFn: AeQueryFn =
// ---- LiteLLM errors: above 5% (chat.answer + theme.ai, both gateway -------
// call sites; window: 1h, same reasoning as session-start) ------------------

/** Gateway calls (denied excluded) in the hour under which the rule is not evaluated: at a 5 % threshold one error only exceeds it when the total is under 20, so a floor of 20 means a single gateway error can never page. */
export const LITELLM_MIN_CALLS = 20;

export async function litellmErrorRateRule(env: Env, queryFn: AeQueryFn = runAeQuery): Promise<RuleResult> {
const windowMs = HOUR_MS;
const [chat, theme] = await Promise.all([
Expand All @@ -390,10 +398,13 @@ export async function litellmErrorRateRule(env: Env, queryFn: AeQueryFn = runAeQ
.reduce((sum, [, c]) => sum + c, 0);
const errors = (chat.get("error") ?? 0) + (theme.get("error") ?? 0);
const pct = ratio(errors, total) * 100;
const enoughSamples = total >= LITELLM_MIN_CALLS;
return {
rule: "litellm-error-rate",
firing: total > 0 && pct > 5,
detail: `${pct.toFixed(2)}% gateway errors over the last hour (${errors}/${total}, threshold 5%)`,
firing: enoughSamples && pct > 5,
detail:
`${pct.toFixed(2)}% gateway errors over the last hour (${errors}/${total}, threshold 5%)` +
(enoughSamples ? "" : `; only ${total} call(s), below the ${LITELLM_MIN_CALLS} needed to evaluate the rate`),
};
}

Expand Down
Loading