// @title Failure and retry waste // @summary Consumption spent on requests that did not succeed, and the throttling and server errors that drive client retries. // @posture count // @posture-note Aggregates consumption by request outcome. Reports loss, never its recoverable value. // // Measured consumption spent on work that did not succeed. // Read-only. Runs in your Log Analytics workspace; sends nothing anywhere. // // WHAT THIS MEASURES, AND WHAT IT DOES NOT. // // It measures the outcome of every gateway request and the model tokens // consumed under each outcome. Tokens consumed by a request that returned // 5xx were paid for and thrown away; that is waste, and it is exact. // // It does NOT claim to have identified retries. Proving that request B is a // retry of request A needs client-side correlation that no APIM log // carries. What the log does support is the waste retries produce, and the // conditions that drive them: 429 throttling and backend 5xx. Those are // reported as what they are — pressure and loss — not as a retry count // inferred from timing. // // The distinction matters because a retry count is the kind of number a // reader would act on. An inferred one would be a guess wearing a figure's // clothing. // // Anchored on the GATEWAY side: every request belongs in the denominator, // including those that never reached a model. Both sides are collapsed to // one row per CorrelationId before joining, so the join cannot multiply. let _startTime = ago(30d); let _endTime = now(); let llm = ApiManagementGatewayLlmLog | where TimeGenerated between (_startTime .. _endTime) | summarize TotalTokensSet = make_set(TotalTokens, 2) by CorrelationId | extend TotalTokens = iff(array_length(TotalTokensSet) == 1, tolong(TotalTokensSet[0]), long(null)) | project CorrelationId, TotalTokens; let gateway = ApiManagementGatewayLogs | where TimeGenerated between (_startTime .. _endTime) | summarize ResponseCodeSet = make_set(ResponseCode, 2), ApiIdSet = make_set(ApiId, 2) by CorrelationId | extend // A request logged as both 200 and 500 is a fact about the log, not // something to resolve by choosing. It reports as indeterminate. ResponseCode = iff(array_length(ResponseCodeSet) == 1, toint(ResponseCodeSet[0]), int(null)), ApiId = iff(array_length(ApiIdSet) == 1, tostring(ApiIdSet[0]), '') | project CorrelationId, ResponseCode, ApiId; let joined = gateway | join kind=leftouter (llm) on CorrelationId | extend Outcome = case( isnull(ResponseCode), 'INDETERMINATE — gateway records disagree on status', ResponseCode >= 500, 'SERVER ERROR (5xx) — consumption lost', ResponseCode == 429, 'THROTTLED (429) — drives client retry', ResponseCode >= 400, 'CLIENT ERROR (4xx)', 'SUCCEEDED'); let totalRequests = toscalar(joined | count); let totalTokens = toscalar(joined | summarize sum(TotalTokens)); joined | summarize Requests = count(), RequestsWithUsage = countif(isnotnull(TotalTokens)), TokensConsumed = sum(TotalTokens) by Outcome // // KUSTO'S sum() RETURNS 0 FOR A GROUP WHERE EVERY VALUE IS NULL, not null. // That turns "no request in this group reported usage" into the confident // claim "this group consumed nothing" — the exact substitution the rest of // the kit refuses to make. A bar at zero reads as measured absence of spend; // a gap reads as absence of evidence, which is what it is. So a token total // resting on zero reporting requests is returned absent. // // It matters most here. A 429 is throttled before it reaches a model, so its // bucket legitimately has no usage — but so does any outcome whose requests // predate the LLM diagnostic category being switched on. Reporting 0 for the // second case would read as "these failures cost nothing", which is the // opposite of the finding. | extend TokensConsumed = iff(RequestsWithUsage > 0, TokensConsumed, long(null)) | extend ShareOfRequestsPct = round(100.0 * Requests / totalRequests, 1), // Share of all measured consumption sitting under this outcome. For // every row except SUCCEEDED, this is consumption with nothing to show // for it. Absent where the outcome reported no usage at all — a share of // an unknown is not zero. ShareOfTokensPct = iff(RequestsWithUsage > 0 and totalTokens > 0, round(100.0 * TokensConsumed / totalTokens, 1), real(null)) | project Outcome, Requests, ShareOfRequestsPct, RequestsWithUsage, TokensConsumed, ShareOfTokensPct | order by TokensConsumed desc, Requests desc