mirror of
https://github.com/priyanshujain/sanderling.git
synced 2026-10-02 19:17:10 +00:00
feat(llmclient): parse usage and the served model
An LLM-in-the-loop evaluation has to report tokens per action and cost per defect, and the client discarded both counters. Served model is recorded separately from the requested one because a router can substitute a differently-priced variant. Claude-Session: https://claude.ai/code/session_01A5KmftdEJ49A9z5mF5ESrX
This commit is contained in:
1 parent
76dce1a75e
commit
b9cdf7a571
2 files changed
+69
No files matched your search
@@ -119,7 +119,21 @@ type JSONSchema struct {
|
||||
|
||||
// Response is the slice of a chat-completions response we read.
|
||||
type Response struct {
|
||||
// Model is the model the provider actually served. A router can satisfy one
|
||||
// requested id with a differently-priced variant, so cost accounting reads
|
||||
// this rather than the requested id.
|
||||
Model string `json:"model"`
|
||||
Choices []Choice `json:"choices"`
|
||||
Usage Usage `json:"usage"`
|
||||
}
|
||||
|
||||
// Usage is the provider's token accounting for one call, present on every
|
||||
// non-streaming OpenAI-compatible chat completion. Zero values mean the
|
||||
// provider omitted the object.
|
||||
type Usage struct {
|
||||
PromptTokens int `json:"prompt_tokens"`
|
||||
CompletionTokens int `json:"completion_tokens"`
|
||||
TotalTokens int `json:"total_tokens"`
|
||||
}
|
||||
|
||||
// Choice is one completion choice.
|
||||
|
||||
Reference in new issue
Block a user