mirror of
https://gitcode.com/JianFeeeee/ModelRouter.git
synced 2026-09-20 00:48:00 +00:00
Live testing proved both sensenova and zen DO return cache fields: - zen laguna-s-2.1-free: usage.prompt_tokens_details.cached_tokens = 32 (real hit), plus cache_write_tokens/audio_tokens - sensenova glm-5.2: prompt_tokens_details.cached_tokens present (0 on short prompts) The previous round only patched deepseek/openai/anthropic/gemini.lua; sensenova/opencode (localzen!) and the other adapters still dropped them. - sensenova/opencode/groq/mistral/github/kimicode: stream + response cache passthrough (same pattern as openai.lua) - agentrouter: response passthrough + NEW stream usage forwarding (it previously dropped the terminal usage-only chunk entirely) - ollama skipped intentionally: its native API has no cache fields Verified end-to-end through the gateway: localzen/laguna-s-2.1-free now returns prompt_tokens_details.cached_tokens=32 to clients, and the request record carries cache_hit_tokens (both chat and stream paths).
174 lines
6.8 KiB
Lua
174 lines
6.8 KiB
Lua
local adapter = {}
|
||
|
||
adapter.name = "kimicode"
|
||
adapter.version = "1.0.0"
|
||
adapter.endpoint = "/v1/chat/completions"
|
||
adapter.headers = {}
|
||
|
||
-- KimiCode / Kimi K2 属于 OpenAI 兼容协议;但部分云端 API 会校验调用方
|
||
-- "app"(只放行特定 agent),要求每次请求带上按 secret 计算的应用签名。
|
||
-- 这里演示 build_headers 钩子:基于 timestamp + 请求体哈希生成签名头。
|
||
--
|
||
-- 配置要求(source.meta):
|
||
-- meta:
|
||
-- app_id: <申请到的 app id>
|
||
-- app_key: <你的 key(由网关的 base_url 复用 api_key 亦可)>
|
||
-- app_secret: <签名密钥>
|
||
-- app_agent: code-agent # 若云端要求声明 agent 身份
|
||
function adapter.transform_request(raw_body)
|
||
local ok, req = pcall(json.decode, raw_body)
|
||
if not ok then return raw_body end
|
||
req.model = req.model or "kimi-k2"
|
||
req.disable_thinking = nil
|
||
req.extra_body = nil
|
||
return json.encode(req)
|
||
end
|
||
|
||
-- 可选的动态签名钩子。meta 由 Go 注入:
|
||
-- meta.url / meta.method / meta.body / meta.api_key / meta.timestamp / meta.source.meta
|
||
function adapter.build_headers(meta)
|
||
local h = {
|
||
["Content-Type"] = "application/json",
|
||
["X-App-Id"] = tostring((meta.source.meta or {}).app_id or ""),
|
||
["X-Timestamp"] = tostring(meta.timestamp),
|
||
}
|
||
local agent = (meta.source.meta or {}).app_agent
|
||
if agent and agent ~= "" then
|
||
h["X-Agent"] = agent
|
||
end
|
||
-- 校验 app:通常要求 Authorization 用 app secret 派生签名
|
||
local secret = (meta.source.meta or {}).app_secret
|
||
local api_key = meta.source.meta and meta.source.meta.api_key or meta.api_key
|
||
if secret and secret ~= "" then
|
||
local body_hash = sha256_hex(meta.body)
|
||
local sign_string = tostring(meta.timestamp) .. meta.method .. meta.url .. body_hash
|
||
local sign = hmac_sha256_hex(secret, sign_string)
|
||
h["Authorization"] = "Bearer " .. api_key
|
||
h["X-App-Sign"] = sign
|
||
else
|
||
h["Authorization"] = "Bearer " .. api_key
|
||
end
|
||
return h
|
||
end
|
||
|
||
function adapter.transform_response(raw_body)
|
||
local ok, resp = pcall(json.decode, raw_body)
|
||
if not ok or resp == nil then return raw_body end
|
||
|
||
local unified = {
|
||
content = "",
|
||
finish_reason = "",
|
||
token_usage = { prompt = 0, completion = 0, total = 0 }
|
||
}
|
||
if type(resp.usage) == "table" then
|
||
unified.token_usage.prompt = resp.usage.prompt_tokens or 0
|
||
unified.token_usage.completion = resp.usage.completion_tokens or 0
|
||
unified.token_usage.total = resp.usage.total_tokens or 0
|
||
local hit = 0
|
||
if type(resp.usage.prompt_tokens_details) == "table" and (resp.usage.prompt_tokens_details.cached_tokens or 0) > 0 then
|
||
hit = resp.usage.prompt_tokens_details.cached_tokens
|
||
unified.token_usage.prompt_tokens_details = { cached_tokens = hit }
|
||
end
|
||
if (resp.usage.prompt_cache_hit_tokens or 0) > 0 then
|
||
unified.token_usage.prompt_cache_hit_tokens = resp.usage.prompt_cache_hit_tokens
|
||
unified.token_usage.prompt_cache_miss_tokens = resp.usage.prompt_cache_miss_tokens or 0
|
||
if hit == 0 then
|
||
unified.token_usage.prompt_tokens_details = { cached_tokens = resp.usage.prompt_cache_hit_tokens }
|
||
end
|
||
end
|
||
end
|
||
if type(resp.choices) == "table" and #resp.choices > 0 then
|
||
local ch = resp.choices[1]
|
||
if type(ch.message) == "table" then
|
||
unified.content = ch.message.content or ""
|
||
if ch.message.reasoning_content then
|
||
unified.reasoning_content = ch.message.reasoning_content
|
||
end
|
||
if type(ch.message.tool_calls) == "table" then
|
||
local tcs = {}
|
||
for _, tc in ipairs(ch.message.tool_calls) do
|
||
local args_ok, args = pcall(json.decode, tc["function"].arguments)
|
||
if not args_ok then args = {} end
|
||
table.insert(tcs, {
|
||
id = tc.id,
|
||
type = tc.type or "function",
|
||
name = tc["function"].name,
|
||
arguments = args
|
||
})
|
||
end
|
||
unified.tool_calls = tcs
|
||
end
|
||
end
|
||
unified.finish_reason = ch.finish_reason or ""
|
||
end
|
||
return json.encode(unified)
|
||
end
|
||
|
||
function adapter.transform_stream_chunk(raw_chunk)
|
||
local ok, chunk = pcall(json.decode, raw_chunk)
|
||
if not ok then return "" end
|
||
|
||
-- OpenAI-style streams may attach usage to a chunk with empty choices
|
||
-- (the final usage chunk). Keys must match Go's TokenUsage json tags
|
||
-- (prompt/completion/total); the gateway re-emits standard *_tokens.
|
||
local uses = nil
|
||
if type(chunk.usage) == "table" then
|
||
uses = {
|
||
prompt = chunk.usage.prompt_tokens or chunk.usage.prompt or 0,
|
||
completion = chunk.usage.completion_tokens or chunk.usage.completion or 0,
|
||
total = chunk.usage.total_tokens or chunk.usage.total or 0,
|
||
}
|
||
if type(chunk.usage.prompt_tokens_details) == "table" and (chunk.usage.prompt_tokens_details.cached_tokens or 0) > 0 then
|
||
uses.prompt_tokens_details = { cached_tokens = chunk.usage.prompt_tokens_details.cached_tokens }
|
||
elseif (chunk.usage.prompt_cache_hit_tokens or 0) > 0 then
|
||
uses.prompt_cache_hit_tokens = chunk.usage.prompt_cache_hit_tokens
|
||
uses.prompt_cache_miss_tokens = chunk.usage.prompt_cache_miss_tokens or 0
|
||
uses.prompt_tokens_details = { cached_tokens = chunk.usage.prompt_cache_hit_tokens }
|
||
end
|
||
end
|
||
|
||
if not chunk.choices or #chunk.choices == 0 then
|
||
if uses ~= nil then
|
||
-- usage-only chunk is not a content/finish signal; the gateway
|
||
-- emits its own terminal stop chunk and merges this usage.
|
||
return json.encode({ usage = uses, done = false })
|
||
end
|
||
return ""
|
||
end
|
||
local delta = chunk.choices[1].delta or {}
|
||
local fr = chunk.choices[1].finish_reason
|
||
|
||
local finish = (type(fr) == "string" and fr ~= "") and fr or nil
|
||
|
||
local unified = {
|
||
content = delta.content or "",
|
||
done = (finish ~= nil)
|
||
}
|
||
if finish then
|
||
unified.finish_reason = finish
|
||
end
|
||
if uses ~= nil then
|
||
unified.usage = uses
|
||
end
|
||
if delta.reasoning_content then
|
||
unified.reasoning_content = delta.reasoning_content
|
||
end
|
||
if delta.tool_calls then
|
||
unified.tool_calls = delta.tool_calls
|
||
end
|
||
return json.encode(unified)
|
||
end
|
||
|
||
-- 错误收敛:kimi 网关为 OpenAI 风格 {error:{message,...}}
|
||
function adapter.transform_error(status, body)
|
||
local ok, resp = pcall(json.decode, body)
|
||
if not ok or type(resp) ~= "table" then return nil end
|
||
local e = resp.error
|
||
if type(e) == "table" and type(e.message) == "string" then
|
||
return e.message
|
||
end
|
||
return nil
|
||
end
|
||
|
||
return adapter
|