mirror of
https://gitcode.com/JianFeeeee/ModelRouter.git
synced 2026-09-20 17:07:59 +00:00
回答「opencodego 的用量与费用透传呢」时逐字段核对上游产出,发现 usage 漏了
一项、费用整项丢失。
## 上游实际发什么(实测 opencode.ai/zen/go/v1)
{
"choices": [...],
"usage": { "prompt_tokens": 37, "completion_tokens": 40, "total_tokens": 77,
"prompt_cache_hit_tokens": 0, "prompt_cache_miss_tokens": 37,
"prompt_tokens_details": {"cached_tokens": 0},
"completion_tokens_details": {"reasoning_tokens": 40} },
"cost": "0"
}
cost 在**顶层**且是**字符串**。流式时还会单独发一帧:
{"choices":[],"cost":"0"}
## 此前丢了两样
1. completion_tokens_details.reasoning_tokens —— 输出里有多少是思考 token。
没有它,客户端无法判断 completion_tokens 里多少是可见回答、多少是思考,
而两者都按输出计费。
2. cost —— 唯一的费用信号,网关整个丢弃。Go 订阅是包月制恒为 "0",
但 Zen 按量付费模型(以及未来的其它源)有信息量。
顺带修掉一处流式/非流式不一致:命中缓存时上游同时给
prompt_tokens_details.cached_tokens 和独立的 hit/miss,流式路径写成了 elseif,
只留 details,与非流式产出不同(只认独立字段的老客户端会看不到缓存)。
## 实现
- types.TokenUsage += CompletionTokensDetails;UnifiedResponse / UnifiedChunk += Cost
- opencodego/opencodezen 适配器映射两个字段;空 choices 帧改成 usage 与 cost
都可带(早退只带 usage 会把同帧的 cost 丢干净 —— 新测试先抓到的就是这个)
- Gateway ChatCompletion / ChatChunk += cost,随终帧发(对齐上游的
{"choices":[],"cost":"0"} 形态)
- Go 兜底 standardSSEChunk 同步支持(openai 系适配器不再漏 reasoning_tokens;
纯 cost 帧不再被整体丢弃),新增 rawCostString 兼容字符串/数字两种形态
费用只做**搬运**:不解析、不换算、不汇总 —— 它是上游事实,且只有部分上游提供。
## 验证
经网关实测 gozen:deepseek-v4.1-flash,流式与非流式产出逐字段一致:
prompt_tokens_details.cached_tokens=6784
prompt_cache_hit_tokens=6784 / miss=148
completion_tokens_details.reasoning_tokens=16
cost="0"
测试:TestOpenCodeCostAndReasoningPassthrough(含「无数据不得凭空造字段」反例)、
TestOpenCodeStreamCacheFieldsMatchNonStream、TestTokenUsageMarshalsCompletionTokensDetails。
343 lines
16 KiB
Lua
343 lines
16 KiB
Lua
local adapter = {}
|
||
|
||
adapter.name = "opencodezen"
|
||
adapter.version = "1.0.0"
|
||
adapter.endpoint = "/chat/completions"
|
||
|
||
-- OpenCode Zen(按量付费端点,https://opencode.ai/zen/v1)
|
||
--
|
||
-- 与 OpenCode Go(opencodego.lua,https://opencode.ai/zen/go/v1)是两个不同的
|
||
-- 服务,行为要求并不相同,因此各有专用适配器:
|
||
--
|
||
-- * 本适配器(Zen)服务免费池 / 按量付费池。免费模型必须带 opencode 客户端
|
||
-- 指纹(UA + x-opencode-*),否则上游拒绝:"free tier can only be used in
|
||
-- OpenCode"。Zen 的免费池以非 thinking 模型为主,历史上回传
|
||
-- reasoning_content 会引出上游报错,故这里剥掉(Go 侧相反,见 opencodego.lua)。
|
||
-- * Go 订阅读者请看 opencodego.lua:那边**必须**回传 reasoning_content,
|
||
-- 否则 thinking 模式直接 400。
|
||
--
|
||
-- 缺 x-opencode-client/session/request/project 身份头会被判为匿名客户端,
|
||
-- 部分模型在带 tools 时上游直接失败(流式返回单 chunk
|
||
-- "finish_reason":"network_error" 空 content,非流式 503
|
||
-- "Endpoint is unavailable",表现为"空回复")。
|
||
adapter.headers = {
|
||
["User-Agent"] = "opencode/1.18.21 ai-sdk/provider-utils/4.0.23 runtime/bun/1.3.14",
|
||
}
|
||
|
||
-- 每请求生成身份头。沙箱无 os/math,用 meta.timestamp + 请求体哈希派生:
|
||
-- 同秒内重复请求 id 相同可接受(zen 只校验存在性,不校验格式)。
|
||
local function rand_id(prefix, seed)
|
||
return prefix .. string.sub(sha256_hex(seed), 1, 24)
|
||
end
|
||
|
||
-- 会话指纹:取历史里**第一条 user 消息**。
|
||
--
|
||
-- 为什么不直接用客户端的会话 id:实测抓包(tcpdump 抓 127.0.0.1:8081 的真实
|
||
-- agent 请求)确认通用 OpenAI 客户端**根本不发**任何会话标识 —— body 里没有
|
||
-- user / session_id / conversation_id / metadata,请求头也只有 X-Stainless-*
|
||
-- (OpenAI JS SDK)与 User-Agent。opencode 原生客户端那套 x-opencode-session
|
||
-- 是它自己的概念,通用客户端无从转发。
|
||
--
|
||
-- 退而求其次但要够用:历史被逐轮重放,**第一条 user 消息在整个会话里恒定**,
|
||
-- 用它做指纹即可得到"每会话一个 session",而不是"每源一个 session"。
|
||
-- 拿不到时(无 user 消息)返回空串,调用方回退到按源稳定。
|
||
local function conversation_fingerprint(body)
|
||
if type(body) ~= "string" or body == "" then return "" end
|
||
local ok, req = pcall(json.decode, body)
|
||
if not ok or type(req) ~= "table" then return "" end
|
||
for _, m in ipairs(req.messages or {}) do
|
||
if type(m) == "table" and m.role == "user" then
|
||
local c = m.content
|
||
if type(c) == "string" then
|
||
if c ~= "" then return c end
|
||
elseif type(c) == "table" then
|
||
for _, part in ipairs(c) do
|
||
if type(part) == "table" and part.type == "text" and part.text then
|
||
return part.text
|
||
end
|
||
end
|
||
end
|
||
return ""
|
||
end
|
||
end
|
||
return ""
|
||
end
|
||
|
||
-- session_seed 决定上游会话号(x-opencode-session 的种子)。
|
||
--
|
||
-- 优先级:
|
||
-- 1) 客户端自带的会话标识(meta.client_session)——pi 等在开启
|
||
-- compat.sendSessionAffinityHeaders 后会发 x-session-affinity,
|
||
-- 这是平台真实的会话 id,整个会话恒定且天然按会话隔离。
|
||
-- 2) 退路:历史里**第一条 user 消息**做会话指纹。
|
||
-- 3) 再退:只按源固定(连 user 消息都没有时)。
|
||
--
|
||
-- 为什么需要 2/3:通用 OpenAI 客户端默认**根本不发**会话标识 —— 实测抓包
|
||
-- (tcpdump 抓 127.0.0.1:8081 真实 agent 请求)确认 body 里没有
|
||
-- user / session_id / conversation_id / metadata,请求头也只有
|
||
-- X-Stainless-*(OpenAI JS SDK)与 User-Agent。opencode 原生客户端那套
|
||
-- x-opencode-session 是它自己的概念,通用客户端无从转发。
|
||
--
|
||
-- 为什么必须稳定:上游前缀缓存是**会话级**的。会话号每请求一变,
|
||
-- 缓存永不命中(实测:固定会号第 2 次命中 5888,每请求换会号则恒为 0)。
|
||
local function session_seed(meta)
|
||
local src = (meta.source and meta.source.name) or ""
|
||
local base = "session|llmsproxy|" .. src
|
||
local cs = meta.client_session
|
||
if type(cs) == "string" and cs ~= "" then
|
||
return base .. "|client|" .. cs
|
||
end
|
||
return base .. "|" .. conversation_fingerprint(meta.body)
|
||
end
|
||
|
||
function adapter.build_headers(meta)
|
||
local ts = tostring(meta.timestamp or "")
|
||
local src = (meta.source and meta.source.name) or ""
|
||
return {
|
||
["User-Agent"] = adapter.headers["User-Agent"],
|
||
["x-opencode-client"] = "cli",
|
||
-- project 固定:同网关实例共享一个工作区身份
|
||
["x-opencode-project"] = string.sub(sha256_hex("llmsproxy|" .. src), 1, 32),
|
||
-- session 必须**按源稳定**,不能每请求换:上游的前缀缓存在同一 session
|
||
-- 内才复用。实测(同一段 6032 token 提示词):
|
||
-- 固定 session -> 第 2 次命中 5888/6032,cached_tokens=5888
|
||
-- 每请求换 session -> 永远 0 命中
|
||
-- 原先用 meta.timestamp 派生,等于每请求都是新会话,缓存永远无效,
|
||
-- 上游也无法做会话亲和路由。
|
||
["x-opencode-session"] = rand_id("ses_", session_seed(meta)),
|
||
-- request id 仍每请求唯一(它只是请求标识,不参与缓存键)
|
||
["x-opencode-request"] = rand_id("msg_", "request|" .. ts .. "|" .. tostring(meta.body or "")),
|
||
}
|
||
end
|
||
|
||
-- OpenAI /chat/completions format (pass-through, strip provider-specific fields)
|
||
-- zen 上游 schema 只接受 text content part(无视觉/音频能力):多模态 part
|
||
-- (image_url / input_audio / file 等)一律剥离。剥离后 content 变空的消息
|
||
-- 若不再携带 tool_calls / tool_call_id 才整条丢弃(避免上游
|
||
-- "unknown variant `image_url`, expected `text`");带工具调用的必须保留,
|
||
-- 否则会把紧随其后的 tool 结果变成孤儿,模型会反复重发同一个调用。
|
||
-- zen 上游角色白名单只有 system / user / assistant / tool / latest_reminder:
|
||
-- OpenAI 的 developer(及 function 等)不在其中,直接透传会触发上游
|
||
-- "unknown variant `developer`, expected one of ..." 错误;统一归一化为 system。
|
||
local ROLE_WHITELIST = {
|
||
system = true,
|
||
user = true,
|
||
assistant = true,
|
||
tool = true,
|
||
latest_reminder = true,
|
||
}
|
||
|
||
function adapter.transform_request(raw_body)
|
||
local ok, req = pcall(json.decode, raw_body)
|
||
if not ok then return raw_body end
|
||
req.disable_thinking = nil
|
||
req.extra_body = nil
|
||
-- stream_options is only valid alongside stream:true; sending it on a
|
||
-- non-streaming request is rejected by strict upstreams.
|
||
if req.stream then
|
||
if type(req.stream_options) ~= "table" then req.stream_options = {} end
|
||
req.stream_options.include_usage = true
|
||
end
|
||
if req.messages then
|
||
local kept = {}
|
||
for _, msg in ipairs(req.messages) do
|
||
if type(msg.role) == "string" and not ROLE_WHITELIST[msg.role] then
|
||
msg.role = "system"
|
||
end
|
||
-- Zen 免费池不接受回传 reasoning_content(Go 侧相反)
|
||
msg.reasoning_content = nil
|
||
local drop = false
|
||
if type(msg.content) == "table" then
|
||
local parts = {}
|
||
for _, part in ipairs(msg.content) do
|
||
if type(part) == "table" and part.type ~= nil and part.type ~= "text" then
|
||
-- multimodal part not supported by zen
|
||
else
|
||
table.insert(parts, part)
|
||
end
|
||
end
|
||
if #parts == 0 then
|
||
-- Content collapsed to nothing after stripping unsupported
|
||
-- parts. A message that still carries a tool call must
|
||
-- NEVER be dropped: the very next message is its tool
|
||
-- result, and dropping the call orphans that result. The
|
||
-- model then sees a result for a call it never made and
|
||
-- re-issues the same tool call on every turn (observed as
|
||
-- an infinite "repeated tool call" loop).
|
||
if (type(msg.tool_calls) == "table" and #msg.tool_calls > 0)
|
||
or msg.tool_call_id ~= nil then
|
||
msg.content = ""
|
||
else
|
||
drop = true
|
||
end
|
||
else
|
||
msg.content = parts
|
||
end
|
||
end
|
||
if not drop then
|
||
table.insert(kept, msg)
|
||
end
|
||
end
|
||
req.messages = kept
|
||
end
|
||
return json.encode(req)
|
||
end
|
||
|
||
function adapter.transform_response(raw_body)
|
||
local ok, resp = pcall(json.decode, raw_body)
|
||
if not ok or resp == nil then return raw_body end
|
||
|
||
local unified = {
|
||
content = "",
|
||
finish_reason = "",
|
||
token_usage = { prompt = 0, completion = 0, total = 0 }
|
||
}
|
||
|
||
if type(resp.usage) == "table" then
|
||
unified.token_usage.prompt = resp.usage.prompt_tokens or 0
|
||
unified.token_usage.completion = resp.usage.completion_tokens or 0
|
||
unified.token_usage.total = resp.usage.total_tokens or 0
|
||
local hit = 0
|
||
if type(resp.usage.prompt_tokens_details) == "table" and resp.usage.prompt_tokens_details.cached_tokens ~= nil then
|
||
hit = resp.usage.prompt_tokens_details.cached_tokens
|
||
unified.token_usage.prompt_tokens_details = { cached_tokens = hit }
|
||
end
|
||
if (resp.usage.prompt_cache_hit_tokens or 0) > 0 then
|
||
unified.token_usage.prompt_cache_hit_tokens = resp.usage.prompt_cache_hit_tokens
|
||
unified.token_usage.prompt_cache_miss_tokens = resp.usage.prompt_cache_miss_tokens or 0
|
||
if hit == 0 then
|
||
unified.token_usage.prompt_tokens_details = { cached_tokens = resp.usage.prompt_cache_hit_tokens }
|
||
end
|
||
end
|
||
-- 上游报告输出里有多少是思考 token。不给客户端的话,无法判断
|
||
-- completion_tokens 里多少是「可见回答」、多少是「思考」(都按输出计费)。
|
||
local ctd = resp.usage.completion_tokens_details
|
||
if type(ctd) == "table" and ctd.reasoning_tokens ~= nil then
|
||
unified.token_usage.completion_tokens_details = { reasoning_tokens = ctd.reasoning_tokens }
|
||
end
|
||
end
|
||
|
||
-- 本次调用的费用,上游放在**顶层**且是**字符串**(如 "0"、"0.0012")。
|
||
-- Go 订阅是包月制,恒为 "0";只有 Zen 按量付费模型才有信息量。
|
||
-- 原样透传:网关不解析、不换算、不汇总 —— 它只是上游事实的搬运者。
|
||
if resp.cost ~= nil then
|
||
unified.cost = tostring(resp.cost)
|
||
end
|
||
|
||
if type(resp.choices) == "table" and #resp.choices > 0 then
|
||
local ch = resp.choices[1]
|
||
if type(ch.message) == "table" then
|
||
unified.content = ch.message.content or ""
|
||
local reasoning = ch.message.reasoning_content or ch.message.reasoning
|
||
if reasoning then
|
||
unified.reasoning_content = reasoning
|
||
end
|
||
if type(ch.message.tool_calls) == "table" then
|
||
local tcs = {}
|
||
for _, tc in ipairs(ch.message.tool_calls) do
|
||
local args_ok, args = pcall(json.decode, tc["function"].arguments)
|
||
if not args_ok then args = {} end
|
||
table.insert(tcs, {
|
||
id = tc.id,
|
||
type = tc.type or "function",
|
||
name = tc["function"].name,
|
||
arguments = args
|
||
})
|
||
end
|
||
unified.tool_calls = tcs
|
||
end
|
||
end
|
||
unified.finish_reason = ch.finish_reason or ""
|
||
end
|
||
|
||
return json.encode(unified)
|
||
end
|
||
|
||
function adapter.transform_stream_chunk(raw_chunk)
|
||
local ok, chunk = pcall(json.decode, raw_chunk)
|
||
if not ok then return "" end
|
||
|
||
-- OpenAI-style streams may attach usage to a chunk with empty choices
|
||
-- (the final usage chunk). Preserve it; the gateway emits it as the
|
||
-- terminal usage chunk. Note: keys must match Go's TokenUsage json tags
|
||
-- (prompt/completion/total); the gateway re-emits standard *_tokens.
|
||
local uses = nil
|
||
if type(chunk.usage) == "table" then
|
||
uses = {
|
||
prompt = chunk.usage.prompt_tokens or chunk.usage.prompt or 0,
|
||
completion = chunk.usage.completion_tokens or chunk.usage.completion or 0,
|
||
total = chunk.usage.total_tokens or chunk.usage.total or 0,
|
||
}
|
||
-- 独立字段与 prompt_tokens_details **并存**透传,不要写成 elseif:
|
||
-- 上游命中缓存时两者都发,非流式路径也是两个都带。写成 elseif 会让
|
||
-- 流式丢掉 prompt_cache_hit_tokens/miss,与同一源的**非流式**产出不一致
|
||
-- (dsh 优先读 details,但只认独立字段的老客户端会看不到缓存)。
|
||
if type(chunk.usage.prompt_tokens_details) == "table" and chunk.usage.prompt_tokens_details.cached_tokens ~= nil then
|
||
uses.prompt_tokens_details = { cached_tokens = chunk.usage.prompt_tokens_details.cached_tokens }
|
||
end
|
||
if (chunk.usage.prompt_cache_hit_tokens or 0) > 0 then
|
||
uses.prompt_cache_hit_tokens = chunk.usage.prompt_cache_hit_tokens
|
||
uses.prompt_cache_miss_tokens = chunk.usage.prompt_cache_miss_tokens or 0
|
||
if uses.prompt_tokens_details == nil then
|
||
uses.prompt_tokens_details = { cached_tokens = chunk.usage.prompt_cache_hit_tokens }
|
||
end
|
||
end
|
||
local ctd = chunk.usage.completion_tokens_details
|
||
if type(ctd) == "table" and ctd.reasoning_tokens ~= nil then
|
||
uses.completion_tokens_details = { reasoning_tokens = ctd.reasoning_tokens }
|
||
end
|
||
end
|
||
|
||
if not chunk.choices or #chunk.choices == 0 then
|
||
-- 空 choices 的帧既不是内容也不是结束信号。上游发两种:
|
||
-- {"choices":[],"usage":{...}} —— 终帧用量
|
||
-- {"choices":[],"cost":"0"} —— 单独的费用帧
|
||
-- 两者可能落在同一帧上,必须都转出去;早退只带 usage 会把费用丢干净。
|
||
local out = { done = false }
|
||
if uses ~= nil then out.usage = uses end
|
||
if chunk.cost ~= nil then out.cost = tostring(chunk.cost) end
|
||
if uses == nil and chunk.cost == nil then return "" end
|
||
return json.encode(out)
|
||
end
|
||
local delta = chunk.choices[1].delta or {}
|
||
local fr = chunk.choices[1].finish_reason
|
||
|
||
local finish = (type(fr) == "string" and fr ~= "") and fr or nil
|
||
|
||
local unified = {
|
||
content = delta.content or "",
|
||
done = (finish ~= nil)
|
||
}
|
||
if finish then
|
||
unified.finish_reason = finish
|
||
end
|
||
if uses ~= nil then
|
||
unified.usage = uses
|
||
end
|
||
-- zen 用 reasoning 字段承载推理文本(OpenAI 惯例是 reasoning_content)
|
||
local reasoning = delta.reasoning_content or delta.reasoning
|
||
if reasoning then
|
||
unified.reasoning_content = reasoning
|
||
end
|
||
if delta.tool_calls then
|
||
-- pass raw streaming fragments through; OpenAI clients accumulate index+id+name+arguments
|
||
unified.tool_calls = delta.tool_calls
|
||
end
|
||
return json.encode(unified)
|
||
end
|
||
|
||
-- 错误收敛(可选钩子):zen 错误信封固定为 {error={type,message}};
|
||
-- 免费池限流 FreeUsageLimitError 单独标注。返回 nil 走通用兜底。
|
||
function adapter.transform_error(status, body)
|
||
local ok, resp = pcall(json.decode, body)
|
||
if not ok or type(resp) ~= "table" then return nil end
|
||
local e = resp.error
|
||
if type(e) ~= "table" then return nil end
|
||
if e.type == "FreeUsageLimitError" then
|
||
return "zen free pool quota exhausted"
|
||
end
|
||
return e.message
|
||
end
|
||
|
||
return adapter
|