mirror of
https://gitcode.com/JianFeeeee/ModelRouter.git
synced 2026-09-20 08:57:57 +00:00
fix(gateway): pass through upstream token usage in streams for all adapters
The prior usage-passthrough fix only covered openai/opencode; the same empty-choices+usage drop bug remained in the 5 sibling OpenAI-compatible adapters, and non-OpenAI providers (anthropic/gemini/ollama) never surfaced streaming usage at all. - deepseek/github/groq/kimicode/mistral: preserve usage on empty-choices chunks and attach it to normal chunks (same pattern as openai.lua) - anthropic: emit usage from message_start (prompt) and message_delta (completion); gateway merges split usage additively - gemini: read usageMetadata in the stream path - ollama: fix non-streaming key (usage -> token_usage, matches UnifiedResponse json tag) and read prompt_eval_count/eval_count; surface counts from the done stream chunk - gateway: mergeUsage combines usage across chunks (non-zero fields win, total recomputed from prompt+completion) so split usage doesn't lose the prompt half; single-chunk case (OpenAI) preserved exactly - usage-only chunks: done=false (no redundant terminal stop), matching the Go fallback standardSSEChunk
This commit is contained in:
@ -53,11 +53,14 @@ function adapter.transform_response(raw_body)
|
||||
local ok, resp = pcall(json.decode, raw_body)
|
||||
if not ok then return raw_body end
|
||||
|
||||
local p = resp.prompt_eval_count or 0
|
||||
local c = resp.eval_count or 0
|
||||
local unified = {
|
||||
content = "",
|
||||
finish_reason = resp.done_reason or "",
|
||||
tool_calls = {},
|
||||
usage = { prompt = 0, completion = 0, total = 0 }
|
||||
-- key must be token_usage to match Go's UnifiedResponse json tag
|
||||
token_usage = { prompt = p, completion = c, total = p + c }
|
||||
}
|
||||
|
||||
if resp.message then
|
||||
@ -70,12 +73,32 @@ end
|
||||
function adapter.transform_stream_chunk(raw_chunk)
|
||||
local ok, chunk = pcall(json.decode, raw_chunk)
|
||||
if not ok then return "" end
|
||||
if not chunk.message then return "" end
|
||||
|
||||
-- Ollama's terminal chunk (done=true) carries token counts but may omit
|
||||
-- message; pass them through so the gateway emits real usage.
|
||||
local uses = nil
|
||||
if chunk.done then
|
||||
local p = chunk.prompt_eval_count or 0
|
||||
local c = chunk.eval_count or 0
|
||||
if p > 0 or c > 0 then
|
||||
uses = { prompt = p, completion = c, total = p + c }
|
||||
end
|
||||
end
|
||||
|
||||
if not chunk.message then
|
||||
if uses ~= nil then
|
||||
return json.encode({ content = "", done = true, usage = uses })
|
||||
end
|
||||
return ""
|
||||
end
|
||||
|
||||
local unified = {
|
||||
content = chunk.message.content or "",
|
||||
done = chunk.done or false
|
||||
}
|
||||
if uses ~= nil then
|
||||
unified.usage = uses
|
||||
end
|
||||
if chunk.message.reasoning_content then
|
||||
unified.reasoning_content = chunk.message.reasoning_content
|
||||
end
|
||||
|
||||
Reference in New Issue
Block a user