mirror of
https://github.com/priyanshujain/sanderling.git
synced 2026-10-02 19:17:10 +00:00
feat(llmclient): parse usage and the served model
An LLM-in-the-loop evaluation has to report tokens per action and cost per defect, and the client discarded both counters. Served model is recorded separately from the requested one because a router can substitute a differently-priced variant. Claude-Session: https://claude.ai/code/session_01A5KmftdEJ49A9z5mF5ESrX
This commit is contained in:
1 parent
76dce1a75e
commit
b9cdf7a571
2 files changed
+69
No files matched your search
@@ -119,7 +119,21 @@ type JSONSchema struct {
|
|||||||
|
|
||||||
// Response is the slice of a chat-completions response we read.
|
// Response is the slice of a chat-completions response we read.
|
||||||
type Response struct {
|
type Response struct {
|
||||||
|
// Model is the model the provider actually served. A router can satisfy one
|
||||||
|
// requested id with a differently-priced variant, so cost accounting reads
|
||||||
|
// this rather than the requested id.
|
||||||
|
Model string `json:"model"`
|
||||||
Choices []Choice `json:"choices"`
|
Choices []Choice `json:"choices"`
|
||||||
|
Usage Usage `json:"usage"`
|
||||||
|
}
|
||||||
|
|
||||||
|
// Usage is the provider's token accounting for one call, present on every
|
||||||
|
// non-streaming OpenAI-compatible chat completion. Zero values mean the
|
||||||
|
// provider omitted the object.
|
||||||
|
type Usage struct {
|
||||||
|
PromptTokens int `json:"prompt_tokens"`
|
||||||
|
CompletionTokens int `json:"completion_tokens"`
|
||||||
|
TotalTokens int `json:"total_tokens"`
|
||||||
}
|
}
|
||||||
|
|
||||||
// Choice is one completion choice.
|
// Choice is one completion choice.
|
||||||
|
|||||||
@@ -106,6 +106,61 @@ func TestChatCompletionRequestShapeAndParse(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestChatCompletionParsesUsageAndServedModel pins the accounting fields: cost
|
||||||
|
// per defect and tokens per action are computed from them, and a router can
|
||||||
|
// serve a request with a differently-priced model than the one asked for.
|
||||||
|
func TestChatCompletionParsesUsageAndServedModel(t *testing.T) {
|
||||||
|
server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||||
|
_, _ = w.Write([]byte(`{
|
||||||
|
"model": "vendor/model-2026-05",
|
||||||
|
"choices": [{"message": {"content": "{}"}}],
|
||||||
|
"usage": {"prompt_tokens": 1200, "completion_tokens": 34, "total_tokens": 1234}
|
||||||
|
}`))
|
||||||
|
}))
|
||||||
|
defer server.Close()
|
||||||
|
|
||||||
|
t.Setenv("OPENROUTER_API_KEY", "test-key")
|
||||||
|
t.Setenv("OPENROUTER_BASE_URL", server.URL)
|
||||||
|
client, err := New()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("New: %v", err)
|
||||||
|
}
|
||||||
|
response, err := client.ChatCompletion(context.Background(), Request{Model: "vendor/model"})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("ChatCompletion: %v", err)
|
||||||
|
}
|
||||||
|
if response.Model != "vendor/model-2026-05" {
|
||||||
|
t.Errorf("served model = %q, want vendor/model-2026-05", response.Model)
|
||||||
|
}
|
||||||
|
want := Usage{PromptTokens: 1200, CompletionTokens: 34, TotalTokens: 1234}
|
||||||
|
if response.Usage != want {
|
||||||
|
t.Errorf("usage = %+v, want %+v", response.Usage, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestChatCompletionToleratesMissingUsage keeps a provider that omits the usage
|
||||||
|
// object from failing the call; the record simply carries zero tokens.
|
||||||
|
func TestChatCompletionToleratesMissingUsage(t *testing.T) {
|
||||||
|
server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||||
|
_, _ = w.Write([]byte(`{"choices":[{"message":{"content":"{}"}}]}`))
|
||||||
|
}))
|
||||||
|
defer server.Close()
|
||||||
|
|
||||||
|
t.Setenv("OPENROUTER_API_KEY", "test-key")
|
||||||
|
t.Setenv("OPENROUTER_BASE_URL", server.URL)
|
||||||
|
client, err := New()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("New: %v", err)
|
||||||
|
}
|
||||||
|
response, err := client.ChatCompletion(context.Background(), Request{Model: "m"})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("ChatCompletion: %v", err)
|
||||||
|
}
|
||||||
|
if (response.Usage != Usage{}) {
|
||||||
|
t.Errorf("usage = %+v, want the zero value when the provider omits it", response.Usage)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
func TestNewRequiresAPIKey(t *testing.T) {
|
func TestNewRequiresAPIKey(t *testing.T) {
|
||||||
t.Setenv("OPENROUTER_API_KEY", "")
|
t.Setenv("OPENROUTER_API_KEY", "")
|
||||||
t.Setenv("OPENAI_API_KEY", "")
|
t.Setenv("OPENAI_API_KEY", "")
|
||||||
|
|||||||
Reference in new issue
Block a user