Files
ModelRouter/internal/lua/adapters/sensenova.lua
dev 21ec8f59d8 feat: surface zero cache hits — distinguish 'missed' from 'not reported'
Live testing across the zen pool showed models report
prompt_tokens_details.cached_tokens even when the hit count is 0 (e.g.
nemotron-3-ultra-free returns cached_tokens:0, audio_tokens:0,
cache_write_tokens:0). The previous >0 guard dropped those objects, so a
cache-enabled upstream looked identical to one without cache support.

- types: PromptTokensDetails.CachedTokens always emitted (drop inner
  omitempty) so clients see cached_tokens:0 explicitly; dsh reads it as
  a 0% hit instead of 'no data'
- adapters (9): forward prompt_tokens_details whenever the upstream
  provides it (presence check instead of >0)
- Req: add cache_reported flag set when usage carried cache accounting;
  WebUI shows an amber 0% tag for reported-but-missed rows and keeps
  the em-dash only for sources that never report cache data
2026-08-25 10:04:44 +08:00

143 lines
5.2 KiB
Lua
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

-- sensenova API format adapter
-- streaming: delta only has reasoning_content, no content field
-- non-streaming: has both content and reasoning_content
local adapter = {}
adapter.name = "sensenova"
adapter.version = "1.0.0"
adapter.endpoint = "/chat/completions"
adapter.headers = {}
-- Same as openai - strip provider-specific fields
function adapter.transform_request(raw_body)
local ok, req = pcall(json.decode, raw_body)
if not ok then return raw_body end
req.disable_thinking = nil
req.extra_body = nil
if req.messages then
for _, msg in ipairs(req.messages) do
msg.reasoning_content = nil
end
end
return json.encode(req)
end
-- Same as openai - extract content from response
function adapter.transform_response(raw_body)
local ok, resp = pcall(json.decode, raw_body)
if not ok or resp == nil then return raw_body end
local unified = {
content = "",
finish_reason = "",
token_usage = { prompt = 0, completion = 0, total = 0 }
}
if type(resp.usage) == "table" then
unified.token_usage.prompt = resp.usage.prompt_tokens or 0
unified.token_usage.completion = resp.usage.completion_tokens or 0
unified.token_usage.total = resp.usage.total_tokens or 0
local hit = 0
if type(resp.usage.prompt_tokens_details) == "table" and resp.usage.prompt_tokens_details.cached_tokens ~= nil then
hit = resp.usage.prompt_tokens_details.cached_tokens
unified.token_usage.prompt_tokens_details = { cached_tokens = hit }
end
if (resp.usage.prompt_cache_hit_tokens or 0) > 0 then
unified.token_usage.prompt_cache_hit_tokens = resp.usage.prompt_cache_hit_tokens
unified.token_usage.prompt_cache_miss_tokens = resp.usage.prompt_cache_miss_tokens or 0
if hit == 0 then
unified.token_usage.prompt_tokens_details = { cached_tokens = resp.usage.prompt_cache_hit_tokens }
end
end
end
if type(resp.choices) == "table" and #resp.choices > 0 then
local ch = resp.choices[1]
if type(ch.message) == "table" then
unified.content = ch.message.content or ""
if ch.message.reasoning_content then
unified.reasoning_content = ch.message.reasoning_content
end
end
unified.finish_reason = ch.finish_reason or ""
end
return json.encode(unified)
end
-- Sensenova-specific stream handling
-- Upstream puts content in reasoning_content only (no content field)
-- Also sends finish_reason="" (empty string) on every chunk
function adapter.transform_stream_chunk(raw_chunk)
local ok, chunk = pcall(json.decode, raw_chunk)
if not ok then return "" end
local uses = nil
if type(chunk.usage) == "table" then
uses = {
prompt = chunk.usage.prompt_tokens or chunk.usage.prompt or 0,
completion = chunk.usage.completion_tokens or chunk.usage.completion or 0,
total = chunk.usage.total_tokens or chunk.usage.total or 0,
}
if type(chunk.usage.prompt_tokens_details) == "table" and chunk.usage.prompt_tokens_details.cached_tokens ~= nil then
uses.prompt_tokens_details = { cached_tokens = chunk.usage.prompt_tokens_details.cached_tokens }
elseif (chunk.usage.prompt_cache_hit_tokens or 0) > 0 then
uses.prompt_cache_hit_tokens = chunk.usage.prompt_cache_hit_tokens
uses.prompt_cache_miss_tokens = chunk.usage.prompt_cache_miss_tokens or 0
uses.prompt_tokens_details = { cached_tokens = chunk.usage.prompt_cache_hit_tokens }
end
end
if not chunk.choices or #chunk.choices == 0 then
if uses ~= nil then
return json.encode({ usage = uses, done = false })
end
return ""
end
local delta = chunk.choices[1].delta or {}
-- Sensenova: delta has reasoning_content but no content
local content = delta.content or ""
if content == "" and delta.reasoning_content then
content = delta.reasoning_content
end
-- Sensenova: finish_reason is "" on every chunk, "stop" on last
local fr = chunk.choices[1].finish_reason
local done = (fr == "stop" or fr == "length")
local unified = {
content = content,
done = done,
}
if chunk.choices[1].finish_reason and chunk.choices[1].finish_reason ~= "" then
unified.finish_reason = chunk.choices[1].finish_reason
end
if delta.tool_calls then
unified.tool_calls = delta.tool_calls
end
if uses ~= nil then
unified.usage = uses
end
return json.encode(unified)
end
-- 错误收敛sensenova 为 OpenAI 风格 {error:{message,...}}
-- 配额类错误单独点出便于客户端识别重置周期。
function adapter.transform_error(status, body)
local ok, resp = pcall(json.decode, body)
if not ok or type(resp) ~= "table" then return nil end
local e = resp.error
if type(e) ~= "table" then return nil end
if status == 429 and type(e.code) == "string"
and e.code == "insufficient_quota" then
return "workspace quota exhausted (resets periodically)"
end
if type(e.message) == "string" then return e.message end
return nil
end
return adapter