mirror of
https://gitcode.com/JianFeeeee/ModelRouter.git
synced 2026-09-20 08:57:57 +00:00
Live testing across the zen pool showed models report prompt_tokens_details.cached_tokens even when the hit count is 0 (e.g. nemotron-3-ultra-free returns cached_tokens:0, audio_tokens:0, cache_write_tokens:0). The previous >0 guard dropped those objects, so a cache-enabled upstream looked identical to one without cache support. - types: PromptTokensDetails.CachedTokens always emitted (drop inner omitempty) so clients see cached_tokens:0 explicitly; dsh reads it as a 0% hit instead of 'no data' - adapters (9): forward prompt_tokens_details whenever the upstream provides it (presence check instead of >0) - Req: add cache_reported flag set when usage carried cache accounting; WebUI shows an amber 0% tag for reported-but-missed rows and keeps the em-dash only for sources that never report cache data
143 lines
5.2 KiB
Lua
143 lines
5.2 KiB
Lua
-- sensenova API format adapter
|
||
-- streaming: delta only has reasoning_content, no content field
|
||
-- non-streaming: has both content and reasoning_content
|
||
|
||
local adapter = {}
|
||
adapter.name = "sensenova"
|
||
adapter.version = "1.0.0"
|
||
adapter.endpoint = "/chat/completions"
|
||
adapter.headers = {}
|
||
|
||
-- Same as openai - strip provider-specific fields
|
||
function adapter.transform_request(raw_body)
|
||
local ok, req = pcall(json.decode, raw_body)
|
||
if not ok then return raw_body end
|
||
req.disable_thinking = nil
|
||
req.extra_body = nil
|
||
if req.messages then
|
||
for _, msg in ipairs(req.messages) do
|
||
msg.reasoning_content = nil
|
||
end
|
||
end
|
||
return json.encode(req)
|
||
end
|
||
|
||
-- Same as openai - extract content from response
|
||
function adapter.transform_response(raw_body)
|
||
local ok, resp = pcall(json.decode, raw_body)
|
||
if not ok or resp == nil then return raw_body end
|
||
|
||
local unified = {
|
||
content = "",
|
||
finish_reason = "",
|
||
token_usage = { prompt = 0, completion = 0, total = 0 }
|
||
}
|
||
|
||
if type(resp.usage) == "table" then
|
||
unified.token_usage.prompt = resp.usage.prompt_tokens or 0
|
||
unified.token_usage.completion = resp.usage.completion_tokens or 0
|
||
unified.token_usage.total = resp.usage.total_tokens or 0
|
||
local hit = 0
|
||
if type(resp.usage.prompt_tokens_details) == "table" and resp.usage.prompt_tokens_details.cached_tokens ~= nil then
|
||
hit = resp.usage.prompt_tokens_details.cached_tokens
|
||
unified.token_usage.prompt_tokens_details = { cached_tokens = hit }
|
||
end
|
||
if (resp.usage.prompt_cache_hit_tokens or 0) > 0 then
|
||
unified.token_usage.prompt_cache_hit_tokens = resp.usage.prompt_cache_hit_tokens
|
||
unified.token_usage.prompt_cache_miss_tokens = resp.usage.prompt_cache_miss_tokens or 0
|
||
if hit == 0 then
|
||
unified.token_usage.prompt_tokens_details = { cached_tokens = resp.usage.prompt_cache_hit_tokens }
|
||
end
|
||
end
|
||
end
|
||
|
||
if type(resp.choices) == "table" and #resp.choices > 0 then
|
||
local ch = resp.choices[1]
|
||
if type(ch.message) == "table" then
|
||
unified.content = ch.message.content or ""
|
||
if ch.message.reasoning_content then
|
||
unified.reasoning_content = ch.message.reasoning_content
|
||
end
|
||
end
|
||
unified.finish_reason = ch.finish_reason or ""
|
||
end
|
||
|
||
return json.encode(unified)
|
||
end
|
||
|
||
-- Sensenova-specific stream handling
|
||
-- Upstream puts content in reasoning_content only (no content field)
|
||
-- Also sends finish_reason="" (empty string) on every chunk
|
||
function adapter.transform_stream_chunk(raw_chunk)
|
||
local ok, chunk = pcall(json.decode, raw_chunk)
|
||
if not ok then return "" end
|
||
|
||
local uses = nil
|
||
if type(chunk.usage) == "table" then
|
||
uses = {
|
||
prompt = chunk.usage.prompt_tokens or chunk.usage.prompt or 0,
|
||
completion = chunk.usage.completion_tokens or chunk.usage.completion or 0,
|
||
total = chunk.usage.total_tokens or chunk.usage.total or 0,
|
||
}
|
||
if type(chunk.usage.prompt_tokens_details) == "table" and chunk.usage.prompt_tokens_details.cached_tokens ~= nil then
|
||
uses.prompt_tokens_details = { cached_tokens = chunk.usage.prompt_tokens_details.cached_tokens }
|
||
elseif (chunk.usage.prompt_cache_hit_tokens or 0) > 0 then
|
||
uses.prompt_cache_hit_tokens = chunk.usage.prompt_cache_hit_tokens
|
||
uses.prompt_cache_miss_tokens = chunk.usage.prompt_cache_miss_tokens or 0
|
||
uses.prompt_tokens_details = { cached_tokens = chunk.usage.prompt_cache_hit_tokens }
|
||
end
|
||
end
|
||
|
||
if not chunk.choices or #chunk.choices == 0 then
|
||
if uses ~= nil then
|
||
return json.encode({ usage = uses, done = false })
|
||
end
|
||
return ""
|
||
end
|
||
|
||
local delta = chunk.choices[1].delta or {}
|
||
|
||
-- Sensenova: delta has reasoning_content but no content
|
||
local content = delta.content or ""
|
||
if content == "" and delta.reasoning_content then
|
||
content = delta.reasoning_content
|
||
end
|
||
|
||
-- Sensenova: finish_reason is "" on every chunk, "stop" on last
|
||
local fr = chunk.choices[1].finish_reason
|
||
local done = (fr == "stop" or fr == "length")
|
||
|
||
local unified = {
|
||
content = content,
|
||
done = done,
|
||
}
|
||
if chunk.choices[1].finish_reason and chunk.choices[1].finish_reason ~= "" then
|
||
unified.finish_reason = chunk.choices[1].finish_reason
|
||
end
|
||
if delta.tool_calls then
|
||
unified.tool_calls = delta.tool_calls
|
||
end
|
||
if uses ~= nil then
|
||
unified.usage = uses
|
||
end
|
||
|
||
return json.encode(unified)
|
||
end
|
||
|
||
-- 错误收敛:sensenova 为 OpenAI 风格 {error:{message,...}};
|
||
-- 配额类错误单独点出便于客户端识别重置周期。
|
||
function adapter.transform_error(status, body)
|
||
local ok, resp = pcall(json.decode, body)
|
||
if not ok or type(resp) ~= "table" then return nil end
|
||
local e = resp.error
|
||
if type(e) ~= "table" then return nil end
|
||
if status == 429 and type(e.code) == "string"
|
||
and e.code == "insufficient_quota" then
|
||
return "workspace quota exhausted (resets periodically)"
|
||
end
|
||
if type(e.message) == "string" then return e.message end
|
||
return nil
|
||
end
|
||
|
||
return adapter
|