mirror of
https://gitcode.com/JianFeeeee/ModelRouter.git
synced 2026-10-03 23:54:06 +00:00
The opencode adapters derived x-opencode-session from meta.timestamp, i.e. a
brand new session on every request. The upstream prefix cache is
session-scoped, so no request could ever hit it, and the cache fields the
endpoint does report (prompt_tokens_details.cached_tokens,
prompt_cache_hit_tokens/prompt_cache_miss_tokens) always came back 0/absent.
Measured against the live endpoint, same 6032-token prompt:
fixed session id -> 2nd call: hit 5888, miss 144
rotating session id -> every call: hit 0, miss 6032
Fix: derive the session from the source name (stable), matching how
x-opencode-project is already derived. x-opencode-request stays unique per
request — it is only a request identifier, not part of the cache key.
Applied to both opencodego and opencodezen.
Through the gateway the same prompt now reports, on the 2nd call:
details={'cached_tokens': 5888} hit=5888 miss=144 (non-streaming)
prompt_tokens_details={'cached_tokens': 5888} (streaming)
Test: TestOpenCodeSessionIsStableForCache asserts the session is stable
across requests for one source while the request id differs.
(cherry picked from commit 791d198f47)
256 lines
11 KiB
Lua
256 lines
11 KiB
Lua
local adapter = {}
|
||
|
||
adapter.name = "opencodezen"
|
||
adapter.version = "1.0.0"
|
||
adapter.endpoint = "/chat/completions"
|
||
|
||
-- OpenCode Zen(按量付费端点,https://opencode.ai/zen/v1)
|
||
--
|
||
-- 与 OpenCode Go(opencodego.lua,https://opencode.ai/zen/go/v1)是两个不同的
|
||
-- 服务,行为要求并不相同,因此各有专用适配器:
|
||
--
|
||
-- * 本适配器(Zen)服务免费池 / 按量付费池。免费模型必须带 opencode 客户端
|
||
-- 指纹(UA + x-opencode-*),否则上游拒绝:"free tier can only be used in
|
||
-- OpenCode"。Zen 的免费池以非 thinking 模型为主,历史上回传
|
||
-- reasoning_content 会引出上游报错,故这里剥掉(Go 侧相反,见 opencodego.lua)。
|
||
-- * Go 订阅读者请看 opencodego.lua:那边**必须**回传 reasoning_content,
|
||
-- 否则 thinking 模式直接 400。
|
||
--
|
||
-- 缺 x-opencode-client/session/request/project 身份头会被判为匿名客户端,
|
||
-- 部分模型在带 tools 时上游直接失败(流式返回单 chunk
|
||
-- "finish_reason":"network_error" 空 content,非流式 503
|
||
-- "Endpoint is unavailable",表现为"空回复")。
|
||
adapter.headers = {
|
||
["User-Agent"] = "opencode/1.18.21 ai-sdk/provider-utils/4.0.23 runtime/bun/1.3.14",
|
||
}
|
||
|
||
-- 每请求生成身份头。沙箱无 os/math,用 meta.timestamp + 请求体哈希派生:
|
||
-- 同秒内重复请求 id 相同可接受(zen 只校验存在性,不校验格式)。
|
||
local function rand_id(prefix, seed)
|
||
return prefix .. string.sub(sha256_hex(seed), 1, 24)
|
||
end
|
||
|
||
function adapter.build_headers(meta)
|
||
local ts = tostring(meta.timestamp or "")
|
||
local src = (meta.source and meta.source.name) or ""
|
||
return {
|
||
["User-Agent"] = adapter.headers["User-Agent"],
|
||
["x-opencode-client"] = "cli",
|
||
-- project 固定:同网关实例共享一个工作区身份
|
||
["x-opencode-project"] = string.sub(sha256_hex("llmsproxy|" .. src), 1, 32),
|
||
-- session 必须**按源稳定**,不能每请求换:上游的前缀缓存在同一 session
|
||
-- 内才复用。实测(同一段 6032 token 提示词):
|
||
-- 固定 session -> 第 2 次命中 5888/6032,cached_tokens=5888
|
||
-- 每请求换 session -> 永远 0 命中
|
||
-- 原先用 meta.timestamp 派生,等于每请求都是新会话,缓存永远无效,
|
||
-- 上游也无法做会话亲和路由。
|
||
["x-opencode-session"] = rand_id("ses_", "session|llmsproxy|" .. src),
|
||
-- request id 仍每请求唯一(它只是请求标识,不参与缓存键)
|
||
["x-opencode-request"] = rand_id("msg_", "request|" .. ts .. "|" .. tostring(meta.body or "")),
|
||
}
|
||
end
|
||
|
||
-- OpenAI /chat/completions format (pass-through, strip provider-specific fields)
|
||
-- zen 上游 schema 只接受 text content part(无视觉/音频能力):多模态 part
|
||
-- (image_url / input_audio / file 等)一律剥离。剥离后 content 变空的消息
|
||
-- 若不再携带 tool_calls / tool_call_id 才整条丢弃(避免上游
|
||
-- "unknown variant `image_url`, expected `text`");带工具调用的必须保留,
|
||
-- 否则会把紧随其后的 tool 结果变成孤儿,模型会反复重发同一个调用。
|
||
-- zen 上游角色白名单只有 system / user / assistant / tool / latest_reminder:
|
||
-- OpenAI 的 developer(及 function 等)不在其中,直接透传会触发上游
|
||
-- "unknown variant `developer`, expected one of ..." 错误;统一归一化为 system。
|
||
local ROLE_WHITELIST = {
|
||
system = true,
|
||
user = true,
|
||
assistant = true,
|
||
tool = true,
|
||
latest_reminder = true,
|
||
}
|
||
|
||
function adapter.transform_request(raw_body)
|
||
local ok, req = pcall(json.decode, raw_body)
|
||
if not ok then return raw_body end
|
||
req.disable_thinking = nil
|
||
req.extra_body = nil
|
||
-- stream_options is only valid alongside stream:true; sending it on a
|
||
-- non-streaming request is rejected by strict upstreams.
|
||
if req.stream then
|
||
if type(req.stream_options) ~= "table" then req.stream_options = {} end
|
||
req.stream_options.include_usage = true
|
||
end
|
||
if req.messages then
|
||
local kept = {}
|
||
for _, msg in ipairs(req.messages) do
|
||
if type(msg.role) == "string" and not ROLE_WHITELIST[msg.role] then
|
||
msg.role = "system"
|
||
end
|
||
-- Zen 免费池不接受回传 reasoning_content(Go 侧相反)
|
||
msg.reasoning_content = nil
|
||
local drop = false
|
||
if type(msg.content) == "table" then
|
||
local parts = {}
|
||
for _, part in ipairs(msg.content) do
|
||
if type(part) == "table" and part.type ~= nil and part.type ~= "text" then
|
||
-- multimodal part not supported by zen
|
||
else
|
||
table.insert(parts, part)
|
||
end
|
||
end
|
||
if #parts == 0 then
|
||
-- Content collapsed to nothing after stripping unsupported
|
||
-- parts. A message that still carries a tool call must
|
||
-- NEVER be dropped: the very next message is its tool
|
||
-- result, and dropping the call orphans that result. The
|
||
-- model then sees a result for a call it never made and
|
||
-- re-issues the same tool call on every turn (observed as
|
||
-- an infinite "repeated tool call" loop).
|
||
if (type(msg.tool_calls) == "table" and #msg.tool_calls > 0)
|
||
or msg.tool_call_id ~= nil then
|
||
msg.content = ""
|
||
else
|
||
drop = true
|
||
end
|
||
else
|
||
msg.content = parts
|
||
end
|
||
end
|
||
if not drop then
|
||
table.insert(kept, msg)
|
||
end
|
||
end
|
||
req.messages = kept
|
||
end
|
||
return json.encode(req)
|
||
end
|
||
|
||
function adapter.transform_response(raw_body)
|
||
local ok, resp = pcall(json.decode, raw_body)
|
||
if not ok or resp == nil then return raw_body end
|
||
|
||
local unified = {
|
||
content = "",
|
||
finish_reason = "",
|
||
token_usage = { prompt = 0, completion = 0, total = 0 }
|
||
}
|
||
|
||
if type(resp.usage) == "table" then
|
||
unified.token_usage.prompt = resp.usage.prompt_tokens or 0
|
||
unified.token_usage.completion = resp.usage.completion_tokens or 0
|
||
unified.token_usage.total = resp.usage.total_tokens or 0
|
||
local hit = 0
|
||
if type(resp.usage.prompt_tokens_details) == "table" and resp.usage.prompt_tokens_details.cached_tokens ~= nil then
|
||
hit = resp.usage.prompt_tokens_details.cached_tokens
|
||
unified.token_usage.prompt_tokens_details = { cached_tokens = hit }
|
||
end
|
||
if (resp.usage.prompt_cache_hit_tokens or 0) > 0 then
|
||
unified.token_usage.prompt_cache_hit_tokens = resp.usage.prompt_cache_hit_tokens
|
||
unified.token_usage.prompt_cache_miss_tokens = resp.usage.prompt_cache_miss_tokens or 0
|
||
if hit == 0 then
|
||
unified.token_usage.prompt_tokens_details = { cached_tokens = resp.usage.prompt_cache_hit_tokens }
|
||
end
|
||
end
|
||
end
|
||
|
||
if type(resp.choices) == "table" and #resp.choices > 0 then
|
||
local ch = resp.choices[1]
|
||
if type(ch.message) == "table" then
|
||
unified.content = ch.message.content or ""
|
||
local reasoning = ch.message.reasoning_content or ch.message.reasoning
|
||
if reasoning then
|
||
unified.reasoning_content = reasoning
|
||
end
|
||
if type(ch.message.tool_calls) == "table" then
|
||
local tcs = {}
|
||
for _, tc in ipairs(ch.message.tool_calls) do
|
||
local args_ok, args = pcall(json.decode, tc["function"].arguments)
|
||
if not args_ok then args = {} end
|
||
table.insert(tcs, {
|
||
id = tc.id,
|
||
type = tc.type or "function",
|
||
name = tc["function"].name,
|
||
arguments = args
|
||
})
|
||
end
|
||
unified.tool_calls = tcs
|
||
end
|
||
end
|
||
unified.finish_reason = ch.finish_reason or ""
|
||
end
|
||
|
||
return json.encode(unified)
|
||
end
|
||
|
||
function adapter.transform_stream_chunk(raw_chunk)
|
||
local ok, chunk = pcall(json.decode, raw_chunk)
|
||
if not ok then return "" end
|
||
|
||
-- OpenAI-style streams may attach usage to a chunk with empty choices
|
||
-- (the final usage chunk). Preserve it; the gateway emits it as the
|
||
-- terminal usage chunk. Note: keys must match Go's TokenUsage json tags
|
||
-- (prompt/completion/total); the gateway re-emits standard *_tokens.
|
||
local uses = nil
|
||
if type(chunk.usage) == "table" then
|
||
uses = {
|
||
prompt = chunk.usage.prompt_tokens or chunk.usage.prompt or 0,
|
||
completion = chunk.usage.completion_tokens or chunk.usage.completion or 0,
|
||
total = chunk.usage.total_tokens or chunk.usage.total or 0,
|
||
}
|
||
if type(chunk.usage.prompt_tokens_details) == "table" and chunk.usage.prompt_tokens_details.cached_tokens ~= nil then
|
||
uses.prompt_tokens_details = { cached_tokens = chunk.usage.prompt_tokens_details.cached_tokens }
|
||
elseif (chunk.usage.prompt_cache_hit_tokens or 0) > 0 then
|
||
uses.prompt_cache_hit_tokens = chunk.usage.prompt_cache_hit_tokens
|
||
uses.prompt_cache_miss_tokens = chunk.usage.prompt_cache_miss_tokens or 0
|
||
uses.prompt_tokens_details = { cached_tokens = chunk.usage.prompt_cache_hit_tokens }
|
||
end
|
||
end
|
||
|
||
if not chunk.choices or #chunk.choices == 0 then
|
||
if uses ~= nil then
|
||
-- usage-only chunk is not a content/finish signal; the gateway
|
||
-- emits its own terminal stop chunk and merges this usage.
|
||
return json.encode({ usage = uses, done = false })
|
||
end
|
||
return ""
|
||
end
|
||
local delta = chunk.choices[1].delta or {}
|
||
local fr = chunk.choices[1].finish_reason
|
||
|
||
local finish = (type(fr) == "string" and fr ~= "") and fr or nil
|
||
|
||
local unified = {
|
||
content = delta.content or "",
|
||
done = (finish ~= nil)
|
||
}
|
||
if finish then
|
||
unified.finish_reason = finish
|
||
end
|
||
if uses ~= nil then
|
||
unified.usage = uses
|
||
end
|
||
-- zen 用 reasoning 字段承载推理文本(OpenAI 惯例是 reasoning_content)
|
||
local reasoning = delta.reasoning_content or delta.reasoning
|
||
if reasoning then
|
||
unified.reasoning_content = reasoning
|
||
end
|
||
if delta.tool_calls then
|
||
-- pass raw streaming fragments through; OpenAI clients accumulate index+id+name+arguments
|
||
unified.tool_calls = delta.tool_calls
|
||
end
|
||
return json.encode(unified)
|
||
end
|
||
|
||
-- 错误收敛(可选钩子):zen 错误信封固定为 {error={type,message}};
|
||
-- 免费池限流 FreeUsageLimitError 单独标注。返回 nil 走通用兜底。
|
||
function adapter.transform_error(status, body)
|
||
local ok, resp = pcall(json.decode, body)
|
||
if not ok or type(resp) ~= "table" then return nil end
|
||
local e = resp.error
|
||
if type(e) ~= "table" then return nil end
|
||
if e.type == "FreeUsageLimitError" then
|
||
return "zen free pool quota exhausted"
|
||
end
|
||
return e.message
|
||
end
|
||
|
||
return adapter
|