!1 merge main into main

feat(gateway): pass through exact upstream token usage in streams

Created-by: valuelesser
Commit-by: valuelesser
Merged-by: JianFeeeee
Description: fix(gateway): emit OpenAI-standard token usage in responses
feat(gateway): pass through exact upstream token usage in streams

See merge request: JianFeeeee/ModelRouter!1
This commit is contained in:
2026-08-18 19:14:46 +08:00
5 changed files with 163 additions and 11 deletions

View File

@ -53,11 +53,14 @@ type RespMessage struct {
}
type ChatChunk struct {
ID string `json:"id"`
Object string `json:"object"`
Created int64 `json:"created"`
Model string `json:"model"`
Choices []ChunkChoice `json:"choices"`
ID string `json:"id"`
Object string `json:"object"`
Created int64 `json:"created"`
Model string `json:"model"`
Choices []ChunkChoice `json:"choices"`
// Usage is sent in the final chunk of a stream (empty choices) so
// OpenAI-compatible clients can read token usage.
Usage *types.TokenUsage `json:"usage,omitempty"`
}
type ChunkChoice struct {
@ -626,7 +629,11 @@ func (g *Gateway) streamChat(w http.ResponseWriter, ctx context.Context, cands [
}) {
return
}
var lastUsage *types.TokenUsage
for ck := range chunks {
if ck.Usage != nil {
lastUsage = ck.Usage
}
chunk := ChatChunk{
ID: id, Object: "chat.completion.chunk", Created: created, Model: effective,
}
@ -653,6 +660,27 @@ func (g *Gateway) streamChat(w http.ResponseWriter, ctx context.Context, cands [
ID: id, Object: "chat.completion.chunk", Created: created, Model: effective,
Choices: []ChunkChoice{{Index: 0, Delta: RespMessage{}, FinishReason: &stop}},
})
// Final usage chunk (OpenAI standard: empty choices + usage before [DONE]).
// Prefer the upstream's exact usage if the stream carried it; fall back to
// the gateway's estimate otherwise.
var tut *types.TokenUsage
if lastUsage != nil {
tut = lastUsage
} else if rec.Prompt+rec.Compl > 0 {
u := types.TokenUsage{
Prompt: int(rec.Prompt),
Completion: int(rec.Compl),
Total: int(rec.Prompt + rec.Compl),
}
tut = &u
}
if tut != nil {
send(ChatChunk{
ID: id, Object: "chat.completion.chunk", Created: created, Model: effective,
Choices: []ChunkChoice{},
Usage: tut,
})
}
fmt.Fprintf(w, "data: [DONE]\n\n")
if flusher != nil {
flusher.Flush()
@ -771,7 +799,11 @@ func (g *Gateway) streamChatAuto(w http.ResponseWriter, ctx context.Context, cha
}) {
return
}
var lastUsage *types.TokenUsage
for ck := range chunks {
if ck.Usage != nil {
lastUsage = ck.Usage
}
chunk := ChatChunk{
ID: id, Object: "chat.completion.chunk", Created: created, Model: rec.Model,
}
@ -798,6 +830,27 @@ func (g *Gateway) streamChatAuto(w http.ResponseWriter, ctx context.Context, cha
ID: id, Object: "chat.completion.chunk", Created: created, Model: rec.Model,
Choices: []ChunkChoice{{Index: 0, Delta: RespMessage{}, FinishReason: &stop}},
})
// Final usage chunk (OpenAI standard: empty choices + usage before [DONE]).
// Prefer the upstream's exact usage if the stream carried it; fall back to
// the gateway's estimate otherwise.
var tut *types.TokenUsage
if lastUsage != nil {
tut = lastUsage
} else if rec.Prompt+rec.Compl > 0 {
u := types.TokenUsage{
Prompt: int(rec.Prompt),
Completion: int(rec.Compl),
Total: int(rec.Prompt + rec.Compl),
}
tut = &u
}
if tut != nil {
send(ChatChunk{
ID: id, Object: "chat.completion.chunk", Created: created, Model: rec.Model,
Choices: []ChunkChoice{},
Usage: tut,
})
}
fmt.Fprintf(w, "data: [DONE]\n\n")
if flusher != nil {
flusher.Flush()

View File

@ -67,7 +67,24 @@ function adapter.transform_stream_chunk(raw_chunk)
local ok, chunk = pcall(json.decode, raw_chunk)
if not ok then return "" end
if not chunk.choices or #chunk.choices == 0 then return "" end
-- OpenAI-style streams may attach usage to a chunk with empty choices
-- (the final usage chunk). Keys must match Go's TokenUsage json tags
-- (prompt/completion/total); the gateway re-emits standard *_tokens.
local uses = nil
if type(chunk.usage) == "table" then
uses = {
prompt = chunk.usage.prompt_tokens or chunk.usage.prompt or 0,
completion = chunk.usage.completion_tokens or chunk.usage.completion or 0,
total = chunk.usage.total_tokens or chunk.usage.total or 0,
}
end
if not chunk.choices or #chunk.choices == 0 then
if uses ~= nil then
return json.encode({ usage = uses, done = (chunk.usage ~= nil) })
end
return ""
end
local delta = chunk.choices[1].delta or {}
local fr = chunk.choices[1].finish_reason
@ -75,11 +92,13 @@ function adapter.transform_stream_chunk(raw_chunk)
content = delta.content or "",
done = (fr ~= nil)
}
if uses ~= nil then
unified.usage = uses
end
if delta.reasoning_content then
unified.reasoning_content = delta.reasoning_content
end
if delta.tool_calls then
-- pass raw streaming fragments through; OpenAI clients accumulate index+id+name+arguments
unified.tool_calls = delta.tool_calls
end
return json.encode(unified)

View File

@ -112,7 +112,25 @@ function adapter.transform_stream_chunk(raw_chunk)
local ok, chunk = pcall(json.decode, raw_chunk)
if not ok then return "" end
if not chunk.choices or #chunk.choices == 0 then return "" end
-- OpenAI-style streams may attach usage to a chunk with empty choices
-- (the final usage chunk). Preserve it; the gateway emits it as the
-- terminal usage chunk. Note: keys must match Go's TokenUsage json tags
-- (prompt/completion/total); the gateway re-emits standard *_tokens.
local uses = nil
if type(chunk.usage) == "table" then
uses = {
prompt = chunk.usage.prompt_tokens or chunk.usage.prompt or 0,
completion = chunk.usage.completion_tokens or chunk.usage.completion or 0,
total = chunk.usage.total_tokens or chunk.usage.total or 0,
}
end
if not chunk.choices or #chunk.choices == 0 then
if uses ~= nil then
return json.encode({ usage = uses, done = (chunk.usage ~= nil) })
end
return ""
end
local delta = chunk.choices[1].delta or {}
local fr = chunk.choices[1].finish_reason
@ -120,6 +138,9 @@ function adapter.transform_stream_chunk(raw_chunk)
content = delta.content or "",
done = (fr ~= nil)
}
if uses ~= nil then
unified.usage = uses
end
if delta.reasoning_content then
unified.reasoning_content = delta.reasoning_content
end

View File

@ -901,20 +901,52 @@ func standardSSEChunk(data string) string {
} `json:"delta"`
FinishReason *string `json:"finish_reason"`
} `json:"choices"`
UpstreamUsage struct {
PromptTokens int `json:"prompt_tokens"`
CompletionTokens int `json:"completion_tokens"`
TotalTokens int `json:"total_tokens"`
Prompt int `json:"prompt"`
Completion int `json:"completion"`
Total int `json:"total"`
} `json:"usage"`
}
if err := json.Unmarshal([]byte(data), &raw); err != nil {
return ""
}
if len(raw.Choices) == 0 {
if len(raw.Choices) == 0 && raw.UpstreamUsage.Total == 0 && raw.UpstreamUsage.TotalTokens == 0 {
return ""
}
var usage *types.TokenUsage
pu := raw.UpstreamUsage
if pu.Total > 0 || pu.TotalTokens > 0 {
usage = &types.TokenUsage{
Prompt: pickFirst(pu.PromptTokens, pu.Prompt),
Completion: pickFirst(pu.CompletionTokens, pu.Completion),
Total: pickFirst(pu.TotalTokens, pu.Total),
}
}
out, _ := json.Marshal(types.UnifiedChunk{
Content: raw.Choices[0].Delta.Content,
Done: raw.Choices[0].FinishReason != nil,
Content: func() string {
if len(raw.Choices) > 0 {
return raw.Choices[0].Delta.Content
}
return ""
}(),
Done: len(raw.Choices) > 0 && raw.Choices[0].FinishReason != nil,
Usage: usage,
})
return string(out)
}
// pickFirst returns a if non-zero, else b (for usage keys that may appear in
// either standard *_tokens or legacy short form).
func pickFirst(a, b int) int {
if a != 0 {
return a
}
return b
}
func truncate(s string, n int) string {
if len(s) <= n {
return s

View File

@ -85,6 +85,28 @@ type TokenUsage struct {
Total int `json:"total"`
}
// MarshalJSON emits both the legacy short keys (prompt/completion/total, used
// by the internal unified representation and older clients) and the OpenAI
// standard keys (prompt_tokens/completion_tokens/total_tokens). Standard
// clients such as DSH and DevEco Code read the *_tokens fields.
func (t TokenUsage) MarshalJSON() ([]byte, error) {
return json.Marshal(struct {
PromptTokens int `json:"prompt_tokens"`
CompletionTokens int `json:"completion_tokens"`
TotalTokens int `json:"total_tokens"`
Prompt int `json:"prompt"`
Completion int `json:"completion"`
Total int `json:"total"`
}{
PromptTokens: t.Prompt,
CompletionTokens: t.Completion,
TotalTokens: t.Total,
Prompt: t.Prompt,
Completion: t.Completion,
Total: t.Total,
})
}
type ImageData struct {
B64JSON string `json:"b64_json,omitempty"`
URL string `json:"url,omitempty"`
@ -116,6 +138,11 @@ type UnifiedChunk struct {
Done bool `json:"done"`
ToolCalls json.RawMessage `json:"tool_calls,omitempty"`
ReasoningContent string `json:"reasoning_content,omitempty"`
// Usage carries the upstream token usage when the stream chunk provides
// it (OpenAI-style streams attach usage to some chunks; the final one
// often has empty choices). Gateway uses it to emit exact usage in the
// final stream chunk instead of estimates.
Usage *TokenUsage `json:"usage,omitempty"`
}
// Meta passed to Lua build_headers hook