Files
ModelRouter/internal/lua/adapters/opencodego.lua
JianFeeeee c744ee151e feat(opencode): 透传 completion_tokens_details.reasoning_tokens 与上游 cost
回答「opencodego 的用量与费用透传呢」时逐字段核对上游产出,发现 usage 漏了
一项、费用整项丢失。

## 上游实际发什么(实测 opencode.ai/zen/go/v1)

  {
    "choices": [...],
    "usage": { "prompt_tokens": 37, "completion_tokens": 40, "total_tokens": 77,
               "prompt_cache_hit_tokens": 0, "prompt_cache_miss_tokens": 37,
               "prompt_tokens_details": {"cached_tokens": 0},
               "completion_tokens_details": {"reasoning_tokens": 40} },
    "cost": "0"
  }

cost 在**顶层**且是**字符串**。流式时还会单独发一帧:
{"choices":[],"cost":"0"}

## 此前丢了两样

1. completion_tokens_details.reasoning_tokens —— 输出里有多少是思考 token。
   没有它,客户端无法判断 completion_tokens 里多少是可见回答、多少是思考,
   而两者都按输出计费。
2. cost —— 唯一的费用信号,网关整个丢弃。Go 订阅是包月制恒为 "0",
   但 Zen 按量付费模型(以及未来的其它源)有信息量。

顺带修掉一处流式/非流式不一致:命中缓存时上游同时给
prompt_tokens_details.cached_tokens 和独立的 hit/miss,流式路径写成了 elseif,
只留 details,与非流式产出不同(只认独立字段的老客户端会看不到缓存)。

## 实现

- types.TokenUsage += CompletionTokensDetails;UnifiedResponse / UnifiedChunk += Cost
- opencodego/opencodezen 适配器映射两个字段;空 choices 帧改成 usage 与 cost
  都可带(早退只带 usage 会把同帧的 cost 丢干净 —— 新测试先抓到的就是这个)
- Gateway ChatCompletion / ChatChunk += cost,随终帧发(对齐上游的
  {"choices":[],"cost":"0"} 形态)
- Go 兜底 standardSSEChunk 同步支持(openai 系适配器不再漏 reasoning_tokens;
  纯 cost 帧不再被整体丢弃),新增 rawCostString 兼容字符串/数字两种形态

费用只做**搬运**:不解析、不换算、不汇总 —— 它是上游事实,且只有部分上游提供。

## 验证

经网关实测 gozen:deepseek-v4.1-flash,流式与非流式产出逐字段一致:
  prompt_tokens_details.cached_tokens=6784
  prompt_cache_hit_tokens=6784 / miss=148
  completion_tokens_details.reasoning_tokens=16
  cost="0"

测试:TestOpenCodeCostAndReasoningPassthrough(含「无数据不得凭空造字段」反例)、
TestOpenCodeStreamCacheFieldsMatchNonStream、TestTokenUsageMarshalsCompletionTokensDetails。
2026-09-11 18:19:40 +08:00

355 lines
17 KiB
Lua
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

local adapter = {}
adapter.name = "opencodego"
adapter.version = "1.0.0"
adapter.endpoint = "/chat/completions"
-- OpenCode Go包月订阅端点https://opencode.ai/zen/go/v1
--
-- 与 OpenCode Zenopencodezen.luahttps://opencode.ai/zen/v1是两个不同的
-- 服务,行为要求并不相同,因此各有专用适配器:
--
-- * 本适配器Go服务包月订阅池。上游**强制要求**
-- `x-opencode-session` 头,缺失直接 400 "MissingSessionID"。
-- * Go 的 thinking 模型deepseek-v4.1-flash 等)**必须回传** assistant 轮的
-- `reasoning_content`,否则 400
-- The `reasoning_content` in the thinking mode must be passed back to the API.
-- 这是本适配器与 Zen 适配器最关键的差异 —— Zen 会剥掉它Go 必须原样保留。
-- * Go 上游对协议更严格:非流式请求带 `stream_options` 会被拒
-- "stream_options should be set along with stream")。
--
-- 同样需要 opencode 客户端指纹UA + x-opencode-*)才能被正确识别与路由。
adapter.headers = {
["User-Agent"] = "opencode/1.18.21 ai-sdk/provider-utils/4.0.23 runtime/bun/1.3.14",
}
-- 每请求生成身份头。沙箱无 os/math用 meta.timestamp + 请求体哈希派生:
-- 同秒内重复请求 id 相同可接受zen 只校验存在性,不校验格式)。
local function rand_id(prefix, seed)
return prefix .. string.sub(sha256_hex(seed), 1, 24)
end
-- 会话指纹:取历史里**第一条 user 消息**。
--
-- 为什么不直接用客户端的会话 id实测抓包tcpdump 抓 127.0.0.1:8081 的真实
-- agent 请求)确认通用 OpenAI 客户端**根本不发**任何会话标识 —— body 里没有
-- user / session_id / conversation_id / metadata请求头也只有 X-Stainless-*
-- OpenAI JS SDK与 User-Agent。opencode 原生客户端那套 x-opencode-session
-- 是它自己的概念,通用客户端无从转发。
--
-- 退而求其次但要够用:历史被逐轮重放,**第一条 user 消息在整个会话里恒定**
-- 用它做指纹即可得到"每会话一个 session",而不是"每源一个 session"。
-- 拿不到时(无 user 消息)返回空串,调用方回退到按源稳定。
local function conversation_fingerprint(body)
if type(body) ~= "string" or body == "" then return "" end
local ok, req = pcall(json.decode, body)
if not ok or type(req) ~= "table" then return "" end
for _, m in ipairs(req.messages or {}) do
if type(m) == "table" and m.role == "user" then
local c = m.content
if type(c) == "string" then
if c ~= "" then return c end
elseif type(c) == "table" then
for _, part in ipairs(c) do
if type(part) == "table" and part.type == "text" and part.text then
return part.text
end
end
end
return ""
end
end
return ""
end
-- session_seed 决定上游会话号x-opencode-session 的种子)。
--
-- 优先级:
-- 1) 客户端自带的会话标识meta.client_session——pi 等在开启
-- compat.sendSessionAffinityHeaders 后会发 x-session-affinity
-- 这是平台真实的会话 id整个会话恒定且天然按会话隔离。
-- 2) 退路:历史里**第一条 user 消息**做会话指纹。
-- 3) 再退:只按源固定(连 user 消息都没有时)。
--
-- 为什么需要 2/3通用 OpenAI 客户端默认**根本不发**会话标识 —— 实测抓包
-- tcpdump 抓 127.0.0.1:8081 真实 agent 请求)确认 body 里没有
-- user / session_id / conversation_id / metadata请求头也只有
-- X-Stainless-*OpenAI JS SDK与 User-Agent。opencode 原生客户端那套
-- x-opencode-session 是它自己的概念,通用客户端无从转发。
--
-- 为什么必须稳定:上游前缀缓存是**会话级**的。会话号每请求一变,
-- 缓存永不命中(实测:固定会号第 2 次命中 5888每请求换会号则恒为 0
local function session_seed(meta)
local src = (meta.source and meta.source.name) or ""
local base = "session|llmsproxy|" .. src
local cs = meta.client_session
if type(cs) == "string" and cs ~= "" then
return base .. "|client|" .. cs
end
return base .. "|" .. conversation_fingerprint(meta.body)
end
function adapter.build_headers(meta)
local ts = tostring(meta.timestamp or "")
local src = (meta.source and meta.source.name) or ""
return {
["User-Agent"] = adapter.headers["User-Agent"],
["x-opencode-client"] = "cli",
-- project 固定:同网关实例共享一个工作区身份
["x-opencode-project"] = string.sub(sha256_hex("llmsproxy|" .. src), 1, 32),
-- session 必须**按源稳定**,不能每请求换:上游的前缀缓存在同一 session
-- 内才复用。实测(同一段 6032 token 提示词):
-- 固定 session -> 第 2 次命中 5888/6032cached_tokens=5888
-- 每请求换 session -> 永远 0 命中
-- 原先用 meta.timestamp 派生,等于每请求都是新会话,缓存永远无效,
-- 上游也无法做会话亲和路由。
["x-opencode-session"] = rand_id("ses_", session_seed(meta)),
-- request id 仍每请求唯一(它只是请求标识,不参与缓存键)
["x-opencode-request"] = rand_id("msg_", "request|" .. ts .. "|" .. tostring(meta.body or "")),
}
end
-- OpenAI /chat/completions format (pass-through, strip provider-specific fields)
-- zen 上游 schema 只接受 text content part无视觉/音频能力):多模态 part
-- image_url / input_audio / file 等)一律剥离。剥离后 content 变空的消息
-- 若不再携带 tool_calls / tool_call_id 才整条丢弃(避免上游
-- "unknown variant `image_url`, expected `text`");带工具调用的必须保留,
-- 否则会把紧随其后的 tool 结果变成孤儿,模型会反复重发同一个调用。
-- zen 上游角色白名单只有 system / user / assistant / tool / latest_reminder
-- OpenAI 的 developer及 function 等)不在其中,直接透传会触发上游
-- "unknown variant `developer`, expected one of ..." 错误;统一归一化为 system。
local ROLE_WHITELIST = {
system = true,
user = true,
assistant = true,
tool = true,
latest_reminder = true,
}
function adapter.transform_request(raw_body)
local ok, req = pcall(json.decode, raw_body)
if not ok then return raw_body end
req.disable_thinking = nil
req.extra_body = nil
-- stream_options is only valid alongside stream:true; sending it on a
-- non-streaming request is rejected by strict upstreams.
if req.stream then
if type(req.stream_options) ~= "table" then req.stream_options = {} end
req.stream_options.include_usage = true
end
if req.messages then
local kept = {}
for _, msg in ipairs(req.messages) do
if type(msg.role) == "string" and not ROLE_WHITELIST[msg.role] then
msg.role = "system"
end
-- Go 的 thinking 模式**必须**回传 reasoning_content剥掉它会让
-- 上游直接 400"The `reasoning_content` in the thinking mode must
-- be passed back to the API"agent 的每一轮都会失败。
-- 注意:与 opencodezen.lua 的行为**相反**,不要在这里剥。
--
-- 但绝大多数客户端pi 等)根本不保存也不回传 reasoning只保留
-- 工具调用本身。上游只在“带 tool_calls 的助手轮”上校验这个字段,
-- 实测**空串即可通过校验**,所以缺省时补空串:既满足上游的
-- 一致性要求,又不伪造任何推理内容(用户看到的 reasoning 仍然是
-- 上游本轮真实返回的)。
if msg.role == "assistant"
and type(msg.tool_calls) == "table" and #msg.tool_calls > 0
and msg.reasoning_content == nil then
msg.reasoning_content = ""
end
local drop = false
if type(msg.content) == "table" then
local parts = {}
for _, part in ipairs(msg.content) do
if type(part) == "table" and part.type ~= nil and part.type ~= "text" then
-- multimodal part not supported by zen
else
table.insert(parts, part)
end
end
if #parts == 0 then
-- Content collapsed to nothing after stripping unsupported
-- parts. A message that still carries a tool call must
-- NEVER be dropped: the very next message is its tool
-- result, and dropping the call orphans that result. The
-- model then sees a result for a call it never made and
-- re-issues the same tool call on every turn (observed as
-- an infinite "repeated tool call" loop).
if (type(msg.tool_calls) == "table" and #msg.tool_calls > 0)
or msg.tool_call_id ~= nil then
msg.content = ""
else
drop = true
end
else
msg.content = parts
end
end
if not drop then
table.insert(kept, msg)
end
end
req.messages = kept
end
return json.encode(req)
end
function adapter.transform_response(raw_body)
local ok, resp = pcall(json.decode, raw_body)
if not ok or resp == nil then return raw_body end
local unified = {
content = "",
finish_reason = "",
token_usage = { prompt = 0, completion = 0, total = 0 }
}
if type(resp.usage) == "table" then
unified.token_usage.prompt = resp.usage.prompt_tokens or 0
unified.token_usage.completion = resp.usage.completion_tokens or 0
unified.token_usage.total = resp.usage.total_tokens or 0
local hit = 0
if type(resp.usage.prompt_tokens_details) == "table" and resp.usage.prompt_tokens_details.cached_tokens ~= nil then
hit = resp.usage.prompt_tokens_details.cached_tokens
unified.token_usage.prompt_tokens_details = { cached_tokens = hit }
end
if (resp.usage.prompt_cache_hit_tokens or 0) > 0 then
unified.token_usage.prompt_cache_hit_tokens = resp.usage.prompt_cache_hit_tokens
unified.token_usage.prompt_cache_miss_tokens = resp.usage.prompt_cache_miss_tokens or 0
if hit == 0 then
unified.token_usage.prompt_tokens_details = { cached_tokens = resp.usage.prompt_cache_hit_tokens }
end
end
-- 上游报告输出里有多少是思考 token。不给客户端的话无法判断
-- completion_tokens 里多少是「可见回答」、多少是「思考」(都按输出计费)。
local ctd = resp.usage.completion_tokens_details
if type(ctd) == "table" and ctd.reasoning_tokens ~= nil then
unified.token_usage.completion_tokens_details = { reasoning_tokens = ctd.reasoning_tokens }
end
end
-- 本次调用的费用,上游放在**顶层**且是**字符串**(如 "0"、"0.0012")。
-- Go 订阅是包月制,恒为 "0";只有 Zen 按量付费模型才有信息量。
-- 原样透传:网关不解析、不换算、不汇总 —— 它只是上游事实的搬运者。
if resp.cost ~= nil then
unified.cost = tostring(resp.cost)
end
if type(resp.choices) == "table" and #resp.choices > 0 then
local ch = resp.choices[1]
if type(ch.message) == "table" then
unified.content = ch.message.content or ""
local reasoning = ch.message.reasoning_content or ch.message.reasoning
if reasoning then
unified.reasoning_content = reasoning
end
if type(ch.message.tool_calls) == "table" then
local tcs = {}
for _, tc in ipairs(ch.message.tool_calls) do
local args_ok, args = pcall(json.decode, tc["function"].arguments)
if not args_ok then args = {} end
table.insert(tcs, {
id = tc.id,
type = tc.type or "function",
name = tc["function"].name,
arguments = args
})
end
unified.tool_calls = tcs
end
end
unified.finish_reason = ch.finish_reason or ""
end
return json.encode(unified)
end
function adapter.transform_stream_chunk(raw_chunk)
local ok, chunk = pcall(json.decode, raw_chunk)
if not ok then return "" end
-- OpenAI-style streams may attach usage to a chunk with empty choices
-- (the final usage chunk). Preserve it; the gateway emits it as the
-- terminal usage chunk. Note: keys must match Go's TokenUsage json tags
-- (prompt/completion/total); the gateway re-emits standard *_tokens.
local uses = nil
if type(chunk.usage) == "table" then
uses = {
prompt = chunk.usage.prompt_tokens or chunk.usage.prompt or 0,
completion = chunk.usage.completion_tokens or chunk.usage.completion or 0,
total = chunk.usage.total_tokens or chunk.usage.total or 0,
}
-- 独立字段与 prompt_tokens_details **并存**透传,不要写成 elseif
-- 上游命中缓存时两者都发,非流式路径也是两个都带。写成 elseif 会让
-- 流式丢掉 prompt_cache_hit_tokens/miss与同一源的**非流式**产出不一致
-- dsh 优先读 details但只认独立字段的老客户端会看不到缓存
if type(chunk.usage.prompt_tokens_details) == "table" and chunk.usage.prompt_tokens_details.cached_tokens ~= nil then
uses.prompt_tokens_details = { cached_tokens = chunk.usage.prompt_tokens_details.cached_tokens }
end
if (chunk.usage.prompt_cache_hit_tokens or 0) > 0 then
uses.prompt_cache_hit_tokens = chunk.usage.prompt_cache_hit_tokens
uses.prompt_cache_miss_tokens = chunk.usage.prompt_cache_miss_tokens or 0
if uses.prompt_tokens_details == nil then
uses.prompt_tokens_details = { cached_tokens = chunk.usage.prompt_cache_hit_tokens }
end
end
local ctd = chunk.usage.completion_tokens_details
if type(ctd) == "table" and ctd.reasoning_tokens ~= nil then
uses.completion_tokens_details = { reasoning_tokens = ctd.reasoning_tokens }
end
end
if not chunk.choices or #chunk.choices == 0 then
-- 空 choices 的帧既不是内容也不是结束信号。上游发两种:
-- {"choices":[],"usage":{...}} —— 终帧用量
-- {"choices":[],"cost":"0"} —— 单独的费用帧
-- 两者可能落在同一帧上,必须都转出去;早退只带 usage 会把费用丢干净。
local out = { done = false }
if uses ~= nil then out.usage = uses end
if chunk.cost ~= nil then out.cost = tostring(chunk.cost) end
if uses == nil and chunk.cost == nil then return "" end
return json.encode(out)
end
local delta = chunk.choices[1].delta or {}
local fr = chunk.choices[1].finish_reason
local finish = (type(fr) == "string" and fr ~= "") and fr or nil
local unified = {
content = delta.content or "",
done = (finish ~= nil)
}
if finish then
unified.finish_reason = finish
end
if uses ~= nil then
unified.usage = uses
end
-- zen 用 reasoning 字段承载推理文本OpenAI 惯例是 reasoning_content
local reasoning = delta.reasoning_content or delta.reasoning
if reasoning then
unified.reasoning_content = reasoning
end
if delta.tool_calls then
-- pass raw streaming fragments through; OpenAI clients accumulate index+id+name+arguments
unified.tool_calls = delta.tool_calls
end
return json.encode(unified)
end
-- 错误收敛可选钩子zen 错误信封固定为 {error={type,message}}
-- 免费池限流 FreeUsageLimitError 单独标注。返回 nil 走通用兜底。
function adapter.transform_error(status, body)
local ok, resp = pcall(json.decode, body)
if not ok or type(resp) ~= "table" then return nil end
local e = resp.error
if type(e) ~= "table" then return nil end
if e.type == "FreeUsageLimitError" then
return "zen free pool quota exhausted"
end
return e.message
end
return adapter