mirror of
https://gitcode.com/JianFeeeee/ModelRouter.git
synced 2026-09-20 00:48:00 +00:00
The prior usage-passthrough fix only covered openai/opencode; the same empty-choices+usage drop bug remained in the 5 sibling OpenAI-compatible adapters, and non-OpenAI providers (anthropic/gemini/ollama) never surfaced streaming usage at all. - deepseek/github/groq/kimicode/mistral: preserve usage on empty-choices chunks and attach it to normal chunks (same pattern as openai.lua) - anthropic: emit usage from message_start (prompt) and message_delta (completion); gateway merges split usage additively - gemini: read usageMetadata in the stream path - ollama: fix non-streaming key (usage -> token_usage, matches UnifiedResponse json tag) and read prompt_eval_count/eval_count; surface counts from the done stream chunk - gateway: mergeUsage combines usage across chunks (non-zero fields win, total recomputed from prompt+completion) so split usage doesn't lose the prompt half; single-chunk case (OpenAI) preserved exactly - usage-only chunks: done=false (no redundant terminal stop), matching the Go fallback standardSSEChunk
126 lines
4.3 KiB
Lua
126 lines
4.3 KiB
Lua
local adapter = {}
|
||
|
||
adapter.name = "deepseek"
|
||
adapter.version = "2.1.0"
|
||
adapter.endpoint = "/chat/completions"
|
||
adapter.headers = {}
|
||
|
||
function adapter.transform_request(raw_body)
|
||
local ok, req = pcall(json.decode, raw_body)
|
||
if not ok then return raw_body end
|
||
|
||
-- V4 已替换 legacy 模型名(deepseek-chat/reasoner 2026-07-24 停用)
|
||
req.model = req.model or "deepseek-v4-flash"
|
||
if req.model == "deepseek-chat" or req.model == "deepseek-reasoner" then
|
||
req.model = "deepseek-v4-flash"
|
||
end
|
||
req.stream = req.stream or false
|
||
if req.disable_thinking then
|
||
req.extra_body = req.extra_body or {}
|
||
req.extra_body.thinking = { type = "disabled" }
|
||
end
|
||
req.disable_thinking = nil
|
||
|
||
-- V4 thinking 模式要求:带 tool_calls 的 assistant 消息必须回传 reasoning_content。
|
||
-- OpenAI 兼容客户端不会发该字段,补空串即可通过校验。
|
||
if req.messages then
|
||
for _, msg in ipairs(req.messages) do
|
||
if msg.role == "assistant" and msg.tool_calls and msg.tool_calls[1] then
|
||
if msg.reasoning_content == nil then
|
||
msg.reasoning_content = ""
|
||
end
|
||
end
|
||
end
|
||
end
|
||
return json.encode(req)
|
||
end
|
||
|
||
function adapter.transform_response(raw_body)
|
||
local ok, resp = pcall(json.decode, raw_body)
|
||
if not ok or resp == nil then return raw_body end
|
||
|
||
local unified = {
|
||
content = "",
|
||
finish_reason = "",
|
||
token_usage = { prompt = 0, completion = 0, total = 0 }
|
||
}
|
||
|
||
if type(resp.usage) == "table" then
|
||
unified.token_usage.prompt = resp.usage.prompt_tokens or 0
|
||
unified.token_usage.completion = resp.usage.completion_tokens or 0
|
||
unified.token_usage.total = resp.usage.total_tokens or 0
|
||
end
|
||
|
||
if type(resp.choices) == "table" and #resp.choices > 0 then
|
||
local ch = resp.choices[1]
|
||
if type(ch.message) == "table" then
|
||
unified.content = ch.message.content or ""
|
||
if ch.message.reasoning_content then
|
||
unified.reasoning_content = ch.message.reasoning_content
|
||
end
|
||
if type(ch.message.tool_calls) == "table" then
|
||
local tcs = {}
|
||
for _, tc in ipairs(ch.message.tool_calls) do
|
||
local args_ok, args = pcall(json.decode, tc["function"].arguments)
|
||
if not args_ok then args = {} end
|
||
table.insert(tcs, {
|
||
id = tc.id,
|
||
type = tc.type or "function",
|
||
name = tc["function"].name,
|
||
arguments = args
|
||
})
|
||
end
|
||
unified.tool_calls = tcs
|
||
end
|
||
end
|
||
unified.finish_reason = ch.finish_reason or ""
|
||
end
|
||
|
||
return json.encode(unified)
|
||
end
|
||
|
||
function adapter.transform_stream_chunk(raw_chunk)
|
||
local ok, chunk = pcall(json.decode, raw_chunk)
|
||
if not ok then return "" end
|
||
|
||
-- OpenAI-style streams may attach usage to a chunk with empty choices
|
||
-- (the final usage chunk). Keys must match Go's TokenUsage json tags
|
||
-- (prompt/completion/total); the gateway re-emits standard *_tokens.
|
||
local uses = nil
|
||
if type(chunk.usage) == "table" then
|
||
uses = {
|
||
prompt = chunk.usage.prompt_tokens or chunk.usage.prompt or 0,
|
||
completion = chunk.usage.completion_tokens or chunk.usage.completion or 0,
|
||
total = chunk.usage.total_tokens or chunk.usage.total or 0,
|
||
}
|
||
end
|
||
|
||
if not chunk.choices or #chunk.choices == 0 then
|
||
if uses ~= nil then
|
||
-- usage-only chunk is not a content/finish signal; the gateway
|
||
-- emits its own terminal stop chunk and merges this usage.
|
||
return json.encode({ usage = uses, done = false })
|
||
end
|
||
return ""
|
||
end
|
||
local delta = chunk.choices[1].delta or {}
|
||
local fr = chunk.choices[1].finish_reason
|
||
|
||
local unified = {
|
||
content = delta.content or "",
|
||
done = (fr ~= nil)
|
||
}
|
||
if uses ~= nil then
|
||
unified.usage = uses
|
||
end
|
||
if delta.reasoning_content then
|
||
unified.reasoning_content = delta.reasoning_content
|
||
end
|
||
if delta.tool_calls then
|
||
unified.tool_calls = delta.tool_calls
|
||
end
|
||
return json.encode(unified)
|
||
end
|
||
|
||
return adapter
|