Files
ModelRouter/internal/lua/adapters/deepseek.lua
dev 21ec8f59d8 feat: surface zero cache hits — distinguish 'missed' from 'not reported'
Live testing across the zen pool showed models report
prompt_tokens_details.cached_tokens even when the hit count is 0 (e.g.
nemotron-3-ultra-free returns cached_tokens:0, audio_tokens:0,
cache_write_tokens:0). The previous >0 guard dropped those objects, so a
cache-enabled upstream looked identical to one without cache support.

- types: PromptTokensDetails.CachedTokens always emitted (drop inner
  omitempty) so clients see cached_tokens:0 explicitly; dsh reads it as
  a 0% hit instead of 'no data'
- adapters (9): forward prompt_tokens_details whenever the upstream
  provides it (presence check instead of >0)
- Req: add cache_reported flag set when usage carried cache accounting;
  WebUI shows an amber 0% tag for reported-but-missed rows and keeps
  the em-dash only for sources that never report cache data
2026-08-25 10:04:44 +08:00

152 lines
5.6 KiB
Lua
Raw Permalink Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

local adapter = {}
adapter.name = "deepseek"
adapter.version = "2.1.0"
adapter.endpoint = "/chat/completions"
adapter.headers = {}
function adapter.transform_request(raw_body)
local ok, req = pcall(json.decode, raw_body)
if not ok then return raw_body end
-- V4 已替换 legacy 模型名deepseek-chat/reasoner 2026-07-24 停用)
req.model = req.model or "deepseek-v4-flash"
if req.model == "deepseek-chat" or req.model == "deepseek-reasoner" then
req.model = "deepseek-v4-flash"
end
req.stream = req.stream or false
if req.disable_thinking then
req.extra_body = req.extra_body or {}
req.extra_body.thinking = { type = "disabled" }
end
req.disable_thinking = nil
-- V4 thinking 模式要求:带 tool_calls 的 assistant 消息必须回传 reasoning_content。
-- OpenAI 兼容客户端不会发该字段,补空串即可通过校验。
if req.messages then
for _, msg in ipairs(req.messages) do
if msg.role == "assistant" and msg.tool_calls and msg.tool_calls[1] then
if msg.reasoning_content == nil then
msg.reasoning_content = ""
end
end
end
end
return json.encode(req)
end
function adapter.transform_response(raw_body)
local ok, resp = pcall(json.decode, raw_body)
if not ok or resp == nil then return raw_body end
local unified = {
content = "",
finish_reason = "",
token_usage = { prompt = 0, completion = 0, total = 0 }
}
if type(resp.usage) == "table" then
unified.token_usage.prompt = resp.usage.prompt_tokens or 0
unified.token_usage.completion = resp.usage.completion_tokens or 0
unified.token_usage.total = resp.usage.total_tokens or 0
unified.token_usage.prompt_cache_hit_tokens = resp.usage.prompt_cache_hit_tokens or 0
unified.token_usage.prompt_cache_miss_tokens = resp.usage.prompt_cache_miss_tokens or 0
if resp.usage.prompt_tokens_details and resp.usage.prompt_tokens_details.cached_tokens ~= nil then
unified.token_usage.prompt_tokens_details = { cached_tokens = resp.usage.prompt_tokens_details.cached_tokens }
end
end
if type(resp.choices) == "table" and #resp.choices > 0 then
local ch = resp.choices[1]
if type(ch.message) == "table" then
unified.content = ch.message.content or ""
if ch.message.reasoning_content then
unified.reasoning_content = ch.message.reasoning_content
end
if type(ch.message.tool_calls) == "table" then
local tcs = {}
for _, tc in ipairs(ch.message.tool_calls) do
local args_ok, args = pcall(json.decode, tc["function"].arguments)
if not args_ok then args = {} end
table.insert(tcs, {
id = tc.id,
type = tc.type or "function",
name = tc["function"].name,
arguments = args
})
end
unified.tool_calls = tcs
end
end
unified.finish_reason = ch.finish_reason or ""
end
return json.encode(unified)
end
function adapter.transform_stream_chunk(raw_chunk)
local ok, chunk = pcall(json.decode, raw_chunk)
if not ok then return "" end
-- OpenAI-style streams may attach usage to a chunk with empty choices
-- (the final usage chunk). Keys must match Go's TokenUsage json tags
-- (prompt/completion/total); the gateway re-emits standard *_tokens.
local uses = nil
if type(chunk.usage) == "table" then
uses = {
prompt = chunk.usage.prompt_tokens or chunk.usage.prompt or 0,
completion = chunk.usage.completion_tokens or chunk.usage.completion or 0,
total = chunk.usage.total_tokens or chunk.usage.total or 0,
prompt_cache_hit_tokens = chunk.usage.prompt_cache_hit_tokens or 0,
prompt_cache_miss_tokens = chunk.usage.prompt_cache_miss_tokens or 0,
}
if chunk.usage.prompt_tokens_details and chunk.usage.prompt_tokens_details.cached_tokens ~= nil then
uses.prompt_tokens_details = { cached_tokens = chunk.usage.prompt_tokens_details.cached_tokens }
end
end
if not chunk.choices or #chunk.choices == 0 then
if uses ~= nil then
-- usage-only chunk is not a content/finish signal; the gateway
-- emits its own terminal stop chunk and merges this usage.
return json.encode({ usage = uses, done = false })
end
return ""
end
local delta = chunk.choices[1].delta or {}
local fr = chunk.choices[1].finish_reason
local finish = (type(fr) == "string" and fr ~= "") and fr or nil
local unified = {
content = delta.content or "",
done = (finish ~= nil)
}
if finish then
unified.finish_reason = finish
end
if uses ~= nil then
unified.usage = uses
end
if delta.reasoning_content then
unified.reasoning_content = delta.reasoning_content
end
if delta.tool_calls then
unified.tool_calls = delta.tool_calls
end
return json.encode(unified)
end
-- 错误收敛DeepSeek 走标准 OpenAI 信封 {error:{message,...}}
function adapter.transform_error(status, body)
local ok, resp = pcall(json.decode, body)
if not ok or type(resp) ~= "table" then return nil end
local e = resp.error
if type(e) == "table" and type(e.message) == "string" then
return e.message
end
return nil
end
return adapter