From 71e9040a453460daf247ca3c8cc97835e33a2983 Mon Sep 17 00:00:00 2001 From: JianFeeeee Date: Fri, 11 Sep 2026 15:00:45 +0800 Subject: [PATCH] feat(adapters): split opencode into opencodezen and opencodego MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Zen (https://opencode.ai/zen/v1) and Go (https://opencode.ai/zen/go/v1) are different services with different requirements, and one shared adapter could not satisfy both. The decisive difference is reasoning_content: * OpenCode Go runs thinking models and REQUIRES the assistant turn's reasoning_content to be echoed back. The shared adapter stripped it (msg.reasoning_content = nil), so every replay of a thinking turn failed with: 400 invalid_request_error: The `reasoning_content` in the thinking mode must be passed back to the API. Reproduced directly: the same request with reasoning_content -> 200, without -> 400. That is why the Go tier never worked in an agent loop. * The Zen free pool must not receive it, so it keeps stripping. Both adapters keep the earlier fixes they share (never drop an assistant turn carrying tool_calls; send stream_options only when streaming; role whitelist; multimodal strip) and the opencode client fingerprint headers — the Go endpoint additionally REQUIRES x-opencode-session, which the adapter already sends. config: localzen -> opencodezen, gozen -> opencodego. Verified: all 25 gozen models answer correctly through the gateway with a thinking + tool_call + tool_result history (was 0/25 before), streaming included; the Zen free models still pass. Test: TestOpenCodeGoVsZenReasoning pins the Go-keeps / Zen-strips split. --- internal/lua/adapters/opencodego.lua | 248 +++++++++++++++++++++ internal/lua/adapters/opencodezen.lua | 247 ++++++++++++++++++++ internal/lua/toolcall_preservation_test.go | 54 +++++ 3 files changed, 549 insertions(+) create mode 100644 internal/lua/adapters/opencodego.lua create mode 100644 internal/lua/adapters/opencodezen.lua diff --git a/internal/lua/adapters/opencodego.lua b/internal/lua/adapters/opencodego.lua new file mode 100644 index 0000000..64675fa --- /dev/null +++ b/internal/lua/adapters/opencodego.lua @@ -0,0 +1,248 @@ +local adapter = {} + +adapter.name = "opencodego" +adapter.version = "1.0.0" +adapter.endpoint = "/chat/completions" + +-- OpenCode Go(包月订阅端点,https://opencode.ai/zen/go/v1) +-- +-- 与 OpenCode Zen(opencodezen.lua,https://opencode.ai/zen/v1)是两个不同的 +-- 服务,行为要求并不相同,因此各有专用适配器: +-- +-- * 本适配器(Go)服务包月订阅池。上游**强制要求** +-- `x-opencode-session` 头,缺失直接 400 "MissingSessionID"。 +-- * Go 的 thinking 模型(deepseek-v4.1-flash 等)**必须回传** assistant 轮的 +-- `reasoning_content`,否则 400: +-- The `reasoning_content` in the thinking mode must be passed back to the API. +-- 这是本适配器与 Zen 适配器最关键的差异 —— Zen 会剥掉它,Go 必须原样保留。 +-- * Go 上游对协议更严格:非流式请求带 `stream_options` 会被拒 +-- ("stream_options should be set along with stream")。 +-- +-- 同样需要 opencode 客户端指纹(UA + x-opencode-*)才能被正确识别与路由。 +adapter.headers = { + ["User-Agent"] = "opencode/1.18.21 ai-sdk/provider-utils/4.0.23 runtime/bun/1.3.14", +} + +-- 每请求生成身份头。沙箱无 os/math,用 meta.timestamp + 请求体哈希派生: +-- 同秒内重复请求 id 相同可接受(zen 只校验存在性,不校验格式)。 +local function rand_id(prefix, seed) + return prefix .. string.sub(sha256_hex(seed), 1, 24) +end + +function adapter.build_headers(meta) + local ts = tostring(meta.timestamp or "") + return { + ["User-Agent"] = adapter.headers["User-Agent"], + ["x-opencode-client"] = "cli", + -- project 固定:同网关实例共享一个工作区身份 + ["x-opencode-project"] = string.sub(sha256_hex("llmsproxy|" .. (meta.source and meta.source.name or "")), 1, 32), + ["x-opencode-session"] = rand_id("ses_", "session|" .. ts), + ["x-opencode-request"] = rand_id("msg_", "request|" .. ts .. "|" .. tostring(meta.body or "")), + } +end + +-- OpenAI /chat/completions format (pass-through, strip provider-specific fields) +-- zen 上游 schema 只接受 text content part(无视觉/音频能力):多模态 part +-- (image_url / input_audio / file 等)一律剥离。剥离后 content 变空的消息 +-- 若不再携带 tool_calls / tool_call_id 才整条丢弃(避免上游 +-- "unknown variant `image_url`, expected `text`");带工具调用的必须保留, +-- 否则会把紧随其后的 tool 结果变成孤儿,模型会反复重发同一个调用。 +-- zen 上游角色白名单只有 system / user / assistant / tool / latest_reminder: +-- OpenAI 的 developer(及 function 等)不在其中,直接透传会触发上游 +-- "unknown variant `developer`, expected one of ..." 错误;统一归一化为 system。 +local ROLE_WHITELIST = { + system = true, + user = true, + assistant = true, + tool = true, + latest_reminder = true, +} + +function adapter.transform_request(raw_body) + local ok, req = pcall(json.decode, raw_body) + if not ok then return raw_body end + req.disable_thinking = nil + req.extra_body = nil + -- stream_options is only valid alongside stream:true; sending it on a + -- non-streaming request is rejected by strict upstreams. + if req.stream then + if type(req.stream_options) ~= "table" then req.stream_options = {} end + req.stream_options.include_usage = true + end + if req.messages then + local kept = {} + for _, msg in ipairs(req.messages) do + if type(msg.role) == "string" and not ROLE_WHITELIST[msg.role] then + msg.role = "system" + end + -- Go 的 thinking 模式**必须**回传 reasoning_content:剥掉它会让 + -- 上游直接 400("The `reasoning_content` in the thinking mode must + -- be passed back to the API"),agent 的每一轮都会失败。 + -- 注意:与 opencodezen.lua 的行为**相反**,不要在这里剥。 + local drop = false + if type(msg.content) == "table" then + local parts = {} + for _, part in ipairs(msg.content) do + if type(part) == "table" and part.type ~= nil and part.type ~= "text" then + -- multimodal part not supported by zen + else + table.insert(parts, part) + end + end + if #parts == 0 then + -- Content collapsed to nothing after stripping unsupported + -- parts. A message that still carries a tool call must + -- NEVER be dropped: the very next message is its tool + -- result, and dropping the call orphans that result. The + -- model then sees a result for a call it never made and + -- re-issues the same tool call on every turn (observed as + -- an infinite "repeated tool call" loop). + if (type(msg.tool_calls) == "table" and #msg.tool_calls > 0) + or msg.tool_call_id ~= nil then + msg.content = "" + else + drop = true + end + else + msg.content = parts + end + end + if not drop then + table.insert(kept, msg) + end + end + req.messages = kept + end + return json.encode(req) +end + +function adapter.transform_response(raw_body) + local ok, resp = pcall(json.decode, raw_body) + if not ok or resp == nil then return raw_body end + + local unified = { + content = "", + finish_reason = "", + token_usage = { prompt = 0, completion = 0, total = 0 } + } + + if type(resp.usage) == "table" then + unified.token_usage.prompt = resp.usage.prompt_tokens or 0 + unified.token_usage.completion = resp.usage.completion_tokens or 0 + unified.token_usage.total = resp.usage.total_tokens or 0 + local hit = 0 + if type(resp.usage.prompt_tokens_details) == "table" and resp.usage.prompt_tokens_details.cached_tokens ~= nil then + hit = resp.usage.prompt_tokens_details.cached_tokens + unified.token_usage.prompt_tokens_details = { cached_tokens = hit } + end + if (resp.usage.prompt_cache_hit_tokens or 0) > 0 then + unified.token_usage.prompt_cache_hit_tokens = resp.usage.prompt_cache_hit_tokens + unified.token_usage.prompt_cache_miss_tokens = resp.usage.prompt_cache_miss_tokens or 0 + if hit == 0 then + unified.token_usage.prompt_tokens_details = { cached_tokens = resp.usage.prompt_cache_hit_tokens } + end + end + end + + if type(resp.choices) == "table" and #resp.choices > 0 then + local ch = resp.choices[1] + if type(ch.message) == "table" then + unified.content = ch.message.content or "" + local reasoning = ch.message.reasoning_content or ch.message.reasoning + if reasoning then + unified.reasoning_content = reasoning + end + if type(ch.message.tool_calls) == "table" then + local tcs = {} + for _, tc in ipairs(ch.message.tool_calls) do + local args_ok, args = pcall(json.decode, tc["function"].arguments) + if not args_ok then args = {} end + table.insert(tcs, { + id = tc.id, + type = tc.type or "function", + name = tc["function"].name, + arguments = args + }) + end + unified.tool_calls = tcs + end + end + unified.finish_reason = ch.finish_reason or "" + end + + return json.encode(unified) +end + +function adapter.transform_stream_chunk(raw_chunk) + local ok, chunk = pcall(json.decode, raw_chunk) + if not ok then return "" end + + -- OpenAI-style streams may attach usage to a chunk with empty choices + -- (the final usage chunk). Preserve it; the gateway emits it as the + -- terminal usage chunk. Note: keys must match Go's TokenUsage json tags + -- (prompt/completion/total); the gateway re-emits standard *_tokens. + local uses = nil + if type(chunk.usage) == "table" then + uses = { + prompt = chunk.usage.prompt_tokens or chunk.usage.prompt or 0, + completion = chunk.usage.completion_tokens or chunk.usage.completion or 0, + total = chunk.usage.total_tokens or chunk.usage.total or 0, + } + if type(chunk.usage.prompt_tokens_details) == "table" and chunk.usage.prompt_tokens_details.cached_tokens ~= nil then + uses.prompt_tokens_details = { cached_tokens = chunk.usage.prompt_tokens_details.cached_tokens } + elseif (chunk.usage.prompt_cache_hit_tokens or 0) > 0 then + uses.prompt_cache_hit_tokens = chunk.usage.prompt_cache_hit_tokens + uses.prompt_cache_miss_tokens = chunk.usage.prompt_cache_miss_tokens or 0 + uses.prompt_tokens_details = { cached_tokens = chunk.usage.prompt_cache_hit_tokens } + end + end + + if not chunk.choices or #chunk.choices == 0 then + if uses ~= nil then + -- usage-only chunk is not a content/finish signal; the gateway + -- emits its own terminal stop chunk and merges this usage. + return json.encode({ usage = uses, done = false }) + end + return "" + end + local delta = chunk.choices[1].delta or {} + local fr = chunk.choices[1].finish_reason + + local finish = (type(fr) == "string" and fr ~= "") and fr or nil + + local unified = { + content = delta.content or "", + done = (finish ~= nil) + } + if finish then + unified.finish_reason = finish + end + if uses ~= nil then + unified.usage = uses + end + -- zen 用 reasoning 字段承载推理文本(OpenAI 惯例是 reasoning_content) + local reasoning = delta.reasoning_content or delta.reasoning + if reasoning then + unified.reasoning_content = reasoning + end + if delta.tool_calls then + -- pass raw streaming fragments through; OpenAI clients accumulate index+id+name+arguments + unified.tool_calls = delta.tool_calls + end + return json.encode(unified) +end + +-- 错误收敛(可选钩子):zen 错误信封固定为 {error={type,message}}; +-- 免费池限流 FreeUsageLimitError 单独标注。返回 nil 走通用兜底。 +function adapter.transform_error(status, body) + local ok, resp = pcall(json.decode, body) + if not ok or type(resp) ~= "table" then return nil end + local e = resp.error + if type(e) ~= "table" then return nil end + if e.type == "FreeUsageLimitError" then + return "zen free pool quota exhausted" + end + return e.message +end + +return adapter diff --git a/internal/lua/adapters/opencodezen.lua b/internal/lua/adapters/opencodezen.lua new file mode 100644 index 0000000..7940a84 --- /dev/null +++ b/internal/lua/adapters/opencodezen.lua @@ -0,0 +1,247 @@ +local adapter = {} + +adapter.name = "opencodezen" +adapter.version = "1.0.0" +adapter.endpoint = "/chat/completions" + +-- OpenCode Zen(按量付费端点,https://opencode.ai/zen/v1) +-- +-- 与 OpenCode Go(opencodego.lua,https://opencode.ai/zen/go/v1)是两个不同的 +-- 服务,行为要求并不相同,因此各有专用适配器: +-- +-- * 本适配器(Zen)服务免费池 / 按量付费池。免费模型必须带 opencode 客户端 +-- 指纹(UA + x-opencode-*),否则上游拒绝:"free tier can only be used in +-- OpenCode"。Zen 的免费池以非 thinking 模型为主,历史上回传 +-- reasoning_content 会引出上游报错,故这里剥掉(Go 侧相反,见 opencodego.lua)。 +-- * Go 订阅读者请看 opencodego.lua:那边**必须**回传 reasoning_content, +-- 否则 thinking 模式直接 400。 +-- +-- 缺 x-opencode-client/session/request/project 身份头会被判为匿名客户端, +-- 部分模型在带 tools 时上游直接失败(流式返回单 chunk +-- "finish_reason":"network_error" 空 content,非流式 503 +-- "Endpoint is unavailable",表现为"空回复")。 +adapter.headers = { + ["User-Agent"] = "opencode/1.18.21 ai-sdk/provider-utils/4.0.23 runtime/bun/1.3.14", +} + +-- 每请求生成身份头。沙箱无 os/math,用 meta.timestamp + 请求体哈希派生: +-- 同秒内重复请求 id 相同可接受(zen 只校验存在性,不校验格式)。 +local function rand_id(prefix, seed) + return prefix .. string.sub(sha256_hex(seed), 1, 24) +end + +function adapter.build_headers(meta) + local ts = tostring(meta.timestamp or "") + return { + ["User-Agent"] = adapter.headers["User-Agent"], + ["x-opencode-client"] = "cli", + -- project 固定:同网关实例共享一个工作区身份 + ["x-opencode-project"] = string.sub(sha256_hex("llmsproxy|" .. (meta.source and meta.source.name or "")), 1, 32), + ["x-opencode-session"] = rand_id("ses_", "session|" .. ts), + ["x-opencode-request"] = rand_id("msg_", "request|" .. ts .. "|" .. tostring(meta.body or "")), + } +end + +-- OpenAI /chat/completions format (pass-through, strip provider-specific fields) +-- zen 上游 schema 只接受 text content part(无视觉/音频能力):多模态 part +-- (image_url / input_audio / file 等)一律剥离。剥离后 content 变空的消息 +-- 若不再携带 tool_calls / tool_call_id 才整条丢弃(避免上游 +-- "unknown variant `image_url`, expected `text`");带工具调用的必须保留, +-- 否则会把紧随其后的 tool 结果变成孤儿,模型会反复重发同一个调用。 +-- zen 上游角色白名单只有 system / user / assistant / tool / latest_reminder: +-- OpenAI 的 developer(及 function 等)不在其中,直接透传会触发上游 +-- "unknown variant `developer`, expected one of ..." 错误;统一归一化为 system。 +local ROLE_WHITELIST = { + system = true, + user = true, + assistant = true, + tool = true, + latest_reminder = true, +} + +function adapter.transform_request(raw_body) + local ok, req = pcall(json.decode, raw_body) + if not ok then return raw_body end + req.disable_thinking = nil + req.extra_body = nil + -- stream_options is only valid alongside stream:true; sending it on a + -- non-streaming request is rejected by strict upstreams. + if req.stream then + if type(req.stream_options) ~= "table" then req.stream_options = {} end + req.stream_options.include_usage = true + end + if req.messages then + local kept = {} + for _, msg in ipairs(req.messages) do + if type(msg.role) == "string" and not ROLE_WHITELIST[msg.role] then + msg.role = "system" + end + -- Zen 免费池不接受回传 reasoning_content(Go 侧相反) + msg.reasoning_content = nil + local drop = false + if type(msg.content) == "table" then + local parts = {} + for _, part in ipairs(msg.content) do + if type(part) == "table" and part.type ~= nil and part.type ~= "text" then + -- multimodal part not supported by zen + else + table.insert(parts, part) + end + end + if #parts == 0 then + -- Content collapsed to nothing after stripping unsupported + -- parts. A message that still carries a tool call must + -- NEVER be dropped: the very next message is its tool + -- result, and dropping the call orphans that result. The + -- model then sees a result for a call it never made and + -- re-issues the same tool call on every turn (observed as + -- an infinite "repeated tool call" loop). + if (type(msg.tool_calls) == "table" and #msg.tool_calls > 0) + or msg.tool_call_id ~= nil then + msg.content = "" + else + drop = true + end + else + msg.content = parts + end + end + if not drop then + table.insert(kept, msg) + end + end + req.messages = kept + end + return json.encode(req) +end + +function adapter.transform_response(raw_body) + local ok, resp = pcall(json.decode, raw_body) + if not ok or resp == nil then return raw_body end + + local unified = { + content = "", + finish_reason = "", + token_usage = { prompt = 0, completion = 0, total = 0 } + } + + if type(resp.usage) == "table" then + unified.token_usage.prompt = resp.usage.prompt_tokens or 0 + unified.token_usage.completion = resp.usage.completion_tokens or 0 + unified.token_usage.total = resp.usage.total_tokens or 0 + local hit = 0 + if type(resp.usage.prompt_tokens_details) == "table" and resp.usage.prompt_tokens_details.cached_tokens ~= nil then + hit = resp.usage.prompt_tokens_details.cached_tokens + unified.token_usage.prompt_tokens_details = { cached_tokens = hit } + end + if (resp.usage.prompt_cache_hit_tokens or 0) > 0 then + unified.token_usage.prompt_cache_hit_tokens = resp.usage.prompt_cache_hit_tokens + unified.token_usage.prompt_cache_miss_tokens = resp.usage.prompt_cache_miss_tokens or 0 + if hit == 0 then + unified.token_usage.prompt_tokens_details = { cached_tokens = resp.usage.prompt_cache_hit_tokens } + end + end + end + + if type(resp.choices) == "table" and #resp.choices > 0 then + local ch = resp.choices[1] + if type(ch.message) == "table" then + unified.content = ch.message.content or "" + local reasoning = ch.message.reasoning_content or ch.message.reasoning + if reasoning then + unified.reasoning_content = reasoning + end + if type(ch.message.tool_calls) == "table" then + local tcs = {} + for _, tc in ipairs(ch.message.tool_calls) do + local args_ok, args = pcall(json.decode, tc["function"].arguments) + if not args_ok then args = {} end + table.insert(tcs, { + id = tc.id, + type = tc.type or "function", + name = tc["function"].name, + arguments = args + }) + end + unified.tool_calls = tcs + end + end + unified.finish_reason = ch.finish_reason or "" + end + + return json.encode(unified) +end + +function adapter.transform_stream_chunk(raw_chunk) + local ok, chunk = pcall(json.decode, raw_chunk) + if not ok then return "" end + + -- OpenAI-style streams may attach usage to a chunk with empty choices + -- (the final usage chunk). Preserve it; the gateway emits it as the + -- terminal usage chunk. Note: keys must match Go's TokenUsage json tags + -- (prompt/completion/total); the gateway re-emits standard *_tokens. + local uses = nil + if type(chunk.usage) == "table" then + uses = { + prompt = chunk.usage.prompt_tokens or chunk.usage.prompt or 0, + completion = chunk.usage.completion_tokens or chunk.usage.completion or 0, + total = chunk.usage.total_tokens or chunk.usage.total or 0, + } + if type(chunk.usage.prompt_tokens_details) == "table" and chunk.usage.prompt_tokens_details.cached_tokens ~= nil then + uses.prompt_tokens_details = { cached_tokens = chunk.usage.prompt_tokens_details.cached_tokens } + elseif (chunk.usage.prompt_cache_hit_tokens or 0) > 0 then + uses.prompt_cache_hit_tokens = chunk.usage.prompt_cache_hit_tokens + uses.prompt_cache_miss_tokens = chunk.usage.prompt_cache_miss_tokens or 0 + uses.prompt_tokens_details = { cached_tokens = chunk.usage.prompt_cache_hit_tokens } + end + end + + if not chunk.choices or #chunk.choices == 0 then + if uses ~= nil then + -- usage-only chunk is not a content/finish signal; the gateway + -- emits its own terminal stop chunk and merges this usage. + return json.encode({ usage = uses, done = false }) + end + return "" + end + local delta = chunk.choices[1].delta or {} + local fr = chunk.choices[1].finish_reason + + local finish = (type(fr) == "string" and fr ~= "") and fr or nil + + local unified = { + content = delta.content or "", + done = (finish ~= nil) + } + if finish then + unified.finish_reason = finish + end + if uses ~= nil then + unified.usage = uses + end + -- zen 用 reasoning 字段承载推理文本(OpenAI 惯例是 reasoning_content) + local reasoning = delta.reasoning_content or delta.reasoning + if reasoning then + unified.reasoning_content = reasoning + end + if delta.tool_calls then + -- pass raw streaming fragments through; OpenAI clients accumulate index+id+name+arguments + unified.tool_calls = delta.tool_calls + end + return json.encode(unified) +end + +-- 错误收敛(可选钩子):zen 错误信封固定为 {error={type,message}}; +-- 免费池限流 FreeUsageLimitError 单独标注。返回 nil 走通用兜底。 +function adapter.transform_error(status, body) + local ok, resp = pcall(json.decode, body) + if not ok or type(resp) ~= "table" then return nil end + local e = resp.error + if type(e) ~= "table" then return nil end + if e.type == "FreeUsageLimitError" then + return "zen free pool quota exhausted" + end + return e.message +end + +return adapter diff --git a/internal/lua/toolcall_preservation_test.go b/internal/lua/toolcall_preservation_test.go index 5aed735..2b953d4 100644 --- a/internal/lua/toolcall_preservation_test.go +++ b/internal/lua/toolcall_preservation_test.go @@ -151,3 +151,57 @@ func TestOpenCodeStreamOptionsOnlyWhenStreaming(t *testing.T) { t.Fatalf("streaming request must carry stream_options.include_usage: %s", sout) } } + +// TestOpenCodeGoVsZenReasoning pins the one behavioural difference that made a +// shared adapter wrong: OpenCode Go's thinking mode REQUIRES the assistant +// turn's reasoning_content to be echoed back ("The `reasoning_content` in the +// thinking mode must be passed back to the API"), while the Zen free pool must +// not receive it. Hence two purpose-built adapters. +func TestOpenCodeGoVsZenReasoning(t *testing.T) { + vm := NewVM(freshAdapterDir(t)) + if err := vm.Start(); err != nil { + t.Fatal(err) + } + defer vm.Stop() + + body := `{"model":"m","messages":[ + {"role":"user","content":"hi"}, + {"role":"assistant","content":"","reasoning_content":"I should read the file.","tool_calls":[{"id":"call_1","type":"function","function":{"name":"read","arguments":"{\"path\":\"/x\"}"}}]}, + {"role":"tool","tool_call_id":"call_1","content":"r"} + ]}` + + // Go: reasoning_content must survive, and the tool call must stay paired. + gout, err := vm.Transform("opencodego", "transform_request", body) + if err != nil { + t.Fatalf("opencodego: %v", err) + } + if !strings.Contains(gout, "I should read the file.") { + t.Errorf("opencodego must pass reasoning_content back (thinking mode requires it): %s", gout) + } + if !strings.Contains(gout, "call_1") || !strings.Contains(gout, "tool_call_id") { + t.Errorf("opencodego must keep the tool call paired with its result: %s", gout) + } + + // Zen: reasoning_content is stripped. + zout, err := vm.Transform("opencodezen", "transform_request", body) + if err != nil { + t.Fatalf("opencodezen: %v", err) + } + if strings.Contains(zout, "I should read the file.") { + t.Errorf("opencodezen must strip reasoning_content: %s", zout) + } + if !strings.Contains(zout, "call_1") { + t.Errorf("opencodezen must still keep the tool call: %s", zout) + } + + // Both must still omit stream_options on non-streaming requests. + for _, a := range []string{"opencodego", "opencodezen"} { + out, err := vm.Transform(a, "transform_request", `{"model":"m","messages":[{"role":"user","content":"hi"}]}`) + if err != nil { + t.Fatalf("%s: %v", a, err) + } + if strings.Contains(out, "stream_options") { + t.Errorf("%s must not send stream_options when not streaming: %s", a, out) + } + } +}