From b9cdf7a57178cdc27166d5b72a83dce751d37a78 Mon Sep 17 00:00:00 2001 From: PJ Date: Wed, 12 Aug 2026 22:56:58 +0530 Subject: [PATCH] feat(llmclient): parse usage and the served model An LLM-in-the-loop evaluation has to report tokens per action and cost per defect, and the client discarded both counters. Served model is recorded separately from the requested one because a router can substitute a differently-priced variant. Claude-Session: https://claude.ai/code/session_01A5KmftdEJ49A9z5mF5ESrX --- internal/llmclient/client.go | 14 ++++++++ internal/llmclient/client_test.go | 55 +++++++++++++++++++++++++++++++ 2 files changed, 69 insertions(+) diff --git a/internal/llmclient/client.go b/internal/llmclient/client.go index 7b21b84..4d46295 100644 --- a/internal/llmclient/client.go +++ b/internal/llmclient/client.go @@ -119,7 +119,21 @@ type JSONSchema struct { // Response is the slice of a chat-completions response we read. type Response struct { + // Model is the model the provider actually served. A router can satisfy one + // requested id with a differently-priced variant, so cost accounting reads + // this rather than the requested id. + Model string `json:"model"` Choices []Choice `json:"choices"` + Usage Usage `json:"usage"` +} + +// Usage is the provider's token accounting for one call, present on every +// non-streaming OpenAI-compatible chat completion. Zero values mean the +// provider omitted the object. +type Usage struct { + PromptTokens int `json:"prompt_tokens"` + CompletionTokens int `json:"completion_tokens"` + TotalTokens int `json:"total_tokens"` } // Choice is one completion choice. diff --git a/internal/llmclient/client_test.go b/internal/llmclient/client_test.go index b4bac25..bb781d3 100644 --- a/internal/llmclient/client_test.go +++ b/internal/llmclient/client_test.go @@ -106,6 +106,61 @@ func TestChatCompletionRequestShapeAndParse(t *testing.T) { } } +// TestChatCompletionParsesUsageAndServedModel pins the accounting fields: cost +// per defect and tokens per action are computed from them, and a router can +// serve a request with a differently-priced model than the one asked for. +func TestChatCompletionParsesUsageAndServedModel(t *testing.T) { + server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + _, _ = w.Write([]byte(`{ + "model": "vendor/model-2026-05", + "choices": [{"message": {"content": "{}"}}], + "usage": {"prompt_tokens": 1200, "completion_tokens": 34, "total_tokens": 1234} + }`)) + })) + defer server.Close() + + t.Setenv("OPENROUTER_API_KEY", "test-key") + t.Setenv("OPENROUTER_BASE_URL", server.URL) + client, err := New() + if err != nil { + t.Fatalf("New: %v", err) + } + response, err := client.ChatCompletion(context.Background(), Request{Model: "vendor/model"}) + if err != nil { + t.Fatalf("ChatCompletion: %v", err) + } + if response.Model != "vendor/model-2026-05" { + t.Errorf("served model = %q, want vendor/model-2026-05", response.Model) + } + want := Usage{PromptTokens: 1200, CompletionTokens: 34, TotalTokens: 1234} + if response.Usage != want { + t.Errorf("usage = %+v, want %+v", response.Usage, want) + } +} + +// TestChatCompletionToleratesMissingUsage keeps a provider that omits the usage +// object from failing the call; the record simply carries zero tokens. +func TestChatCompletionToleratesMissingUsage(t *testing.T) { + server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + _, _ = w.Write([]byte(`{"choices":[{"message":{"content":"{}"}}]}`)) + })) + defer server.Close() + + t.Setenv("OPENROUTER_API_KEY", "test-key") + t.Setenv("OPENROUTER_BASE_URL", server.URL) + client, err := New() + if err != nil { + t.Fatalf("New: %v", err) + } + response, err := client.ChatCompletion(context.Background(), Request{Model: "m"}) + if err != nil { + t.Fatalf("ChatCompletion: %v", err) + } + if (response.Usage != Usage{}) { + t.Errorf("usage = %+v, want the zero value when the provider omits it", response.Usage) + } +} + func TestNewRequiresAPIKey(t *testing.T) { t.Setenv("OPENROUTER_API_KEY", "") t.Setenv("OPENAI_API_KEY", "")