feat(llmclient): parse usage and the served model

An LLM-in-the-loop evaluation has to report tokens per action and cost per
defect, and the client discarded both counters. Served model is recorded
separately from the requested one because a router can substitute a
differently-priced variant.

Claude-Session: https://claude.ai/code/session_01A5KmftdEJ49A9z5mF5ESrX
This commit is contained in:
pj committed 2026-08-12 22:56:58 +05:30
1 parent 76dce1a75e
commit b9cdf7a571
2 files changed
+69

No files matched your search

+14
View File
@@ -119,7 +119,21 @@ type JSONSchema struct {
// Response is the slice of a chat-completions response we read. // Response is the slice of a chat-completions response we read.
type Response struct { type Response struct {
// Model is the model the provider actually served. A router can satisfy one
// requested id with a differently-priced variant, so cost accounting reads
// this rather than the requested id.
Model string `json:"model"`
Choices []Choice `json:"choices"` Choices []Choice `json:"choices"`
Usage Usage `json:"usage"`
}
// Usage is the provider's token accounting for one call, present on every
// non-streaming OpenAI-compatible chat completion. Zero values mean the
// provider omitted the object.
type Usage struct {
PromptTokens int `json:"prompt_tokens"`
CompletionTokens int `json:"completion_tokens"`
TotalTokens int `json:"total_tokens"`
} }
// Choice is one completion choice. // Choice is one completion choice.
+55
View File
@@ -106,6 +106,61 @@ func TestChatCompletionRequestShapeAndParse(t *testing.T) {
} }
} }
// TestChatCompletionParsesUsageAndServedModel pins the accounting fields: cost
// per defect and tokens per action are computed from them, and a router can
// serve a request with a differently-priced model than the one asked for.
func TestChatCompletionParsesUsageAndServedModel(t *testing.T) {
server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
_, _ = w.Write([]byte(`{
"model": "vendor/model-2026-05",
"choices": [{"message": {"content": "{}"}}],
"usage": {"prompt_tokens": 1200, "completion_tokens": 34, "total_tokens": 1234}
}`))
}))
defer server.Close()
t.Setenv("OPENROUTER_API_KEY", "test-key")
t.Setenv("OPENROUTER_BASE_URL", server.URL)
client, err := New()
if err != nil {
t.Fatalf("New: %v", err)
}
response, err := client.ChatCompletion(context.Background(), Request{Model: "vendor/model"})
if err != nil {
t.Fatalf("ChatCompletion: %v", err)
}
if response.Model != "vendor/model-2026-05" {
t.Errorf("served model = %q, want vendor/model-2026-05", response.Model)
}
want := Usage{PromptTokens: 1200, CompletionTokens: 34, TotalTokens: 1234}
if response.Usage != want {
t.Errorf("usage = %+v, want %+v", response.Usage, want)
}
}
// TestChatCompletionToleratesMissingUsage keeps a provider that omits the usage
// object from failing the call; the record simply carries zero tokens.
func TestChatCompletionToleratesMissingUsage(t *testing.T) {
server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
_, _ = w.Write([]byte(`{"choices":[{"message":{"content":"{}"}}]}`))
}))
defer server.Close()
t.Setenv("OPENROUTER_API_KEY", "test-key")
t.Setenv("OPENROUTER_BASE_URL", server.URL)
client, err := New()
if err != nil {
t.Fatalf("New: %v", err)
}
response, err := client.ChatCompletion(context.Background(), Request{Model: "m"})
if err != nil {
t.Fatalf("ChatCompletion: %v", err)
}
if (response.Usage != Usage{}) {
t.Errorf("usage = %+v, want the zero value when the provider omits it", response.Usage)
}
}
func TestNewRequiresAPIKey(t *testing.T) { func TestNewRequiresAPIKey(t *testing.T) {
t.Setenv("OPENROUTER_API_KEY", "") t.Setenv("OPENROUTER_API_KEY", "")
t.Setenv("OPENAI_API_KEY", "") t.Setenv("OPENAI_API_KEY", "")