// @title Where consumption concentrates // @summary Requests and tokens by model and deployment, with each row's share of total request volume. // @posture count // @posture-note Aggregates by model and deployment. Share of volume is arithmetic, not a verdict. // // Where the consumption concentrates, by model and deployment. // Read-only. Runs in your Log Analytics workspace; sends nothing anywhere. // // Collapsed to one row per CorrelationId before aggregation, for the reason // given in 01-observed-spend.kql: the LLM table emits several records per // logical request, so record-level counts and sums are not request-level // economics. Model and deployment are taken only where the records agree; // where they disagree, or were never reported, the row is grouped under // '(not reported)' rather than being dropped. Dropping it would shrink the // denominator and quietly overstate the share held by everything else. let _startTime = ago(30d); let _endTime = now(); let requests = ApiManagementGatewayLlmLog | where TimeGenerated between (_startTime .. _endTime) | summarize ModelNameSet = make_set(ModelName, 2), DeploymentNameSet = make_set(DeploymentName, 2), PromptTokensSet = make_set(PromptTokens, 2), CompletionTokensSet = make_set(CompletionTokens, 2), TotalTokensSet = make_set(TotalTokens, 2) by CorrelationId | extend ModelName = iff(array_length(ModelNameSet) == 1, tostring(ModelNameSet[0]), ''), DeploymentName = iff(array_length(DeploymentNameSet) == 1, tostring(DeploymentNameSet[0]), ''), PromptTokens = iff(array_length(PromptTokensSet) == 1, tolong(PromptTokensSet[0]), long(null)), CompletionTokens = iff(array_length(CompletionTokensSet) == 1, tolong(CompletionTokensSet[0]), long(null)), TotalTokens = iff(array_length(TotalTokensSet) == 1, tolong(TotalTokensSet[0]), long(null)); let totalRequests = toscalar(requests | count); requests | extend ModelName = iff(isempty(ModelName), '(not reported)', ModelName), DeploymentName = iff(isempty(DeploymentName), '(not reported)', DeploymentName) | summarize Requests = count(), RequestsWithUsage = countif(isnotnull(TotalTokens)), InputTokens = sum(PromptTokens), OutputTokens = sum(CompletionTokens), TotalTokens = sum(TotalTokens) by ModelName, DeploymentName | extend ShareOfRequestsPct = round(100.0 * Requests / totalRequests, 1), UsageCoveragePct = round(100.0 * RequestsWithUsage / Requests, 1) // // KUSTO'S sum() RETURNS 0 FOR A GROUP WHERE EVERY VALUE IS NULL, not null. // That turns "no request in this group reported usage" into the confident // claim "this group consumed nothing" — the exact substitution the rest of // the kit refuses to make. A bar at zero reads as measured absence of spend; // a gap reads as absence of evidence, which is what it is. So a token total // resting on zero reporting requests is returned absent. | extend InputTokens = iff(RequestsWithUsage > 0, InputTokens, long(null)), OutputTokens = iff(RequestsWithUsage > 0, OutputTokens, long(null)), TotalTokens = iff(RequestsWithUsage > 0, TotalTokens, long(null)) | project ModelName, DeploymentName, Requests, ShareOfRequestsPct, UsageCoveragePct, InputTokens, OutputTokens, TotalTokens | order by TotalTokens desc, Requests desc