Files
ModelRouter/internal/lua/adapters/sensenova.lua
dev ec89daad62 fix(adapters): pass through cache tokens in the remaining 8 adapters
Live testing proved both sensenova and zen DO return cache fields:
- zen laguna-s-2.1-free: usage.prompt_tokens_details.cached_tokens = 32
  (real hit), plus cache_write_tokens/audio_tokens
- sensenova glm-5.2: prompt_tokens_details.cached_tokens present (0 on
  short prompts)

The previous round only patched deepseek/openai/anthropic/gemini.lua;
sensenova/opencode (localzen!) and the other adapters still dropped them.

- sensenova/opencode/groq/mistral/github/kimicode: stream + response
  cache passthrough (same pattern as openai.lua)
- agentrouter: response passthrough + NEW stream usage forwarding (it
  previously dropped the terminal usage-only chunk entirely)
- ollama skipped intentionally: its native API has no cache fields

Verified end-to-end through the gateway: localzen/laguna-s-2.1-free now
returns prompt_tokens_details.cached_tokens=32 to clients, and the request
record carries cache_hit_tokens (both chat and stream paths).
2026-08-25 09:52:55 +08:00

143 lines
5.2 KiB
Lua
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

-- sensenova API format adapter
-- streaming: delta only has reasoning_content, no content field
-- non-streaming: has both content and reasoning_content
local adapter = {}
adapter.name = "sensenova"
adapter.version = "1.0.0"
adapter.endpoint = "/chat/completions"
adapter.headers = {}
-- Same as openai - strip provider-specific fields
function adapter.transform_request(raw_body)
local ok, req = pcall(json.decode, raw_body)
if not ok then return raw_body end
req.disable_thinking = nil
req.extra_body = nil
if req.messages then
for _, msg in ipairs(req.messages) do
msg.reasoning_content = nil
end
end
return json.encode(req)
end
-- Same as openai - extract content from response
function adapter.transform_response(raw_body)
local ok, resp = pcall(json.decode, raw_body)
if not ok or resp == nil then return raw_body end
local unified = {
content = "",
finish_reason = "",
token_usage = { prompt = 0, completion = 0, total = 0 }
}
if type(resp.usage) == "table" then
unified.token_usage.prompt = resp.usage.prompt_tokens or 0
unified.token_usage.completion = resp.usage.completion_tokens or 0
unified.token_usage.total = resp.usage.total_tokens or 0
local hit = 0
if type(resp.usage.prompt_tokens_details) == "table" and (resp.usage.prompt_tokens_details.cached_tokens or 0) > 0 then
hit = resp.usage.prompt_tokens_details.cached_tokens
unified.token_usage.prompt_tokens_details = { cached_tokens = hit }
end
if (resp.usage.prompt_cache_hit_tokens or 0) > 0 then
unified.token_usage.prompt_cache_hit_tokens = resp.usage.prompt_cache_hit_tokens
unified.token_usage.prompt_cache_miss_tokens = resp.usage.prompt_cache_miss_tokens or 0
if hit == 0 then
unified.token_usage.prompt_tokens_details = { cached_tokens = resp.usage.prompt_cache_hit_tokens }
end
end
end
if type(resp.choices) == "table" and #resp.choices > 0 then
local ch = resp.choices[1]
if type(ch.message) == "table" then
unified.content = ch.message.content or ""
if ch.message.reasoning_content then
unified.reasoning_content = ch.message.reasoning_content
end
end
unified.finish_reason = ch.finish_reason or ""
end
return json.encode(unified)
end
-- Sensenova-specific stream handling
-- Upstream puts content in reasoning_content only (no content field)
-- Also sends finish_reason="" (empty string) on every chunk
function adapter.transform_stream_chunk(raw_chunk)
local ok, chunk = pcall(json.decode, raw_chunk)
if not ok then return "" end
local uses = nil
if type(chunk.usage) == "table" then
uses = {
prompt = chunk.usage.prompt_tokens or chunk.usage.prompt or 0,
completion = chunk.usage.completion_tokens or chunk.usage.completion or 0,
total = chunk.usage.total_tokens or chunk.usage.total or 0,
}
if type(chunk.usage.prompt_tokens_details) == "table" and (chunk.usage.prompt_tokens_details.cached_tokens or 0) > 0 then
uses.prompt_tokens_details = { cached_tokens = chunk.usage.prompt_tokens_details.cached_tokens }
elseif (chunk.usage.prompt_cache_hit_tokens or 0) > 0 then
uses.prompt_cache_hit_tokens = chunk.usage.prompt_cache_hit_tokens
uses.prompt_cache_miss_tokens = chunk.usage.prompt_cache_miss_tokens or 0
uses.prompt_tokens_details = { cached_tokens = chunk.usage.prompt_cache_hit_tokens }
end
end
if not chunk.choices or #chunk.choices == 0 then
if uses ~= nil then
return json.encode({ usage = uses, done = false })
end
return ""
end
local delta = chunk.choices[1].delta or {}
-- Sensenova: delta has reasoning_content but no content
local content = delta.content or ""
if content == "" and delta.reasoning_content then
content = delta.reasoning_content
end
-- Sensenova: finish_reason is "" on every chunk, "stop" on last
local fr = chunk.choices[1].finish_reason
local done = (fr == "stop" or fr == "length")
local unified = {
content = content,
done = done,
}
if chunk.choices[1].finish_reason and chunk.choices[1].finish_reason ~= "" then
unified.finish_reason = chunk.choices[1].finish_reason
end
if delta.tool_calls then
unified.tool_calls = delta.tool_calls
end
if uses ~= nil then
unified.usage = uses
end
return json.encode(unified)
end
-- 错误收敛sensenova 为 OpenAI 风格 {error:{message,...}}
-- 配额类错误单独点出便于客户端识别重置周期。
function adapter.transform_error(status, body)
local ok, resp = pcall(json.decode, body)
if not ok or type(resp) ~= "table" then return nil end
local e = resp.error
if type(e) ~= "table" then return nil end
if status == 429 and type(e.code) == "string"
and e.code == "insufficient_quota" then
return "workspace quota exhausted (resets periodically)"
end
if type(e.message) == "string" then return e.message end
return nil
end
return adapter