// @title Validation candidates // @summary Operations ranked by input context paid for relative to output produced. A candidate worth testing — never a conclusion. // @posture rank // @posture-note Orders candidates by a measured quantity. Applies no threshold and reaches no verdict. // // Economic candidates worth testing before anything changes. // Read-only. Runs in your Log Analytics workspace; sends nothing anywhere. // // THIS QUERY RANKS. IT DOES NOT CONCLUDE. // // It reports, per operation and model, how much input context is being paid // for relative to the output produced. A high input-to-output ratio is the // signature of context that may not be earning its cost — retrieved // documents that go unused, prompts that accumulated, history replayed on // every turn. // // It is a CANDIDATE and nothing more. Whether that context is waste or // whether it is the reason the output is correct cannot be established from // a log, and no threshold is applied here: Metergrade does not know this // estate's quality, latency or reliability requirements, and a query that // invented one would be recommending a change it had not tested. Validation // against those requirements is what turns a candidate into a verdict, and // that step happens deliberately, not in a workbook tile. // // Ranked by total input tokens, because the largest consumer of context is // where testing is worth the effort first — not by ratio alone, which would // promote a trivial operation with an extreme ratio above a dominant one. let _startTime = ago(30d); let _endTime = now(); let llm = ApiManagementGatewayLlmLog | where TimeGenerated between (_startTime .. _endTime) | summarize ModelNameSet = make_set(ModelName, 2), PromptTokensSet = make_set(PromptTokens, 2), CompletionTokensSet = make_set(CompletionTokens, 2) by CorrelationId | extend ModelName = iff(array_length(ModelNameSet) == 1, tostring(ModelNameSet[0]), ''), PromptTokens = iff(array_length(PromptTokensSet) == 1, tolong(PromptTokensSet[0]), long(null)), CompletionTokens = iff(array_length(CompletionTokensSet) == 1, tolong(CompletionTokensSet[0]), long(null)) // Only requests with usable economics can be ranked on economics. | where isnotnull(PromptTokens) and isnotnull(CompletionTokens) | project CorrelationId, ModelName, PromptTokens, CompletionTokens; let gateway = ApiManagementGatewayLogs | where TimeGenerated between (_startTime .. _endTime) | summarize ApiIdSet = make_set(ApiId, 2), OperationIdSet = make_set(OperationId, 2) by CorrelationId | extend ApiId = iff(array_length(ApiIdSet) == 1, tostring(ApiIdSet[0]), ''), OperationId = iff(array_length(OperationIdSet) == 1, tostring(OperationIdSet[0]), '') | project CorrelationId, ApiId, OperationId; llm | join kind=leftouter (gateway) on CorrelationId | extend ApiId = iff(isempty(coalesce(ApiId, '')), '(unattributed)', ApiId), OperationId = iff(isempty(coalesce(OperationId, '')), '(unattributed)', OperationId), ModelName = iff(isempty(ModelName), '(not reported)', ModelName) | summarize Requests = count(), InputTokens = sum(PromptTokens), OutputTokens = sum(CompletionTokens), MedianInput = percentile(PromptTokens, 50), MedianOutput = percentile(CompletionTokens, 50) by ApiId, OperationId, ModelName // Guarded against divide-by-zero: an operation that produced no output at // all has no ratio, and reporting one as infinite would be a fabrication. | extend InputPerOutputToken = iff(OutputTokens > 0, round(1.0 * InputTokens / OutputTokens, 1), real(null)) | project ApiId, OperationId, ModelName, Requests, InputTokens, OutputTokens, InputPerOutputToken, MedianInput, MedianOutput | order by InputTokens desc | take 50