mirror of
https://gitcode.com/JianFeeeee/ModelRouter.git
synced 2026-09-19 16:39:15 +00:00
Four adapters handled tool_calls in transform_stream_chunk but lost them in
transform_response, so any NON-streaming tool-using conversation broke on its
second request: the client received finish_reason:"tool_calls" with no
tool_calls payload, replayed an assistant message whose function
name/arguments were empty, and the upstream rejected the next turn with
400 invalid tool_call function, function/name/arguments cannot be empty
The production audit trail shows 46 such failures on sensenova alone.
- sensenova.lua: forward message.tool_calls, decoding the arguments JSON string
into an object as the unified shape expects.
- gemini.lua: collect functionCall parts from candidates[].content.parts. Also
correct finish_reason, since Gemini reports "STOP" even when it emitted a
function call and clients keyed on it treat that as a finished answer.
- ollama.lua: the field was initialized to an empty table and never filled;
fill it and likewise correct done_reason "stop" -> "tool_calls".
trae is a different failure with the same symptom: trae-local-api's OpenAI
endpoint (/v1/chat/completions, src/server.js:353) never reads the request's
`tools` array — only its Anthropic endpoint does — so the relayed model is never
told the tool schema and instead PRINTS a <tool_call>{...}</tool_call> block into
content, leaving message.tool_calls null and finish_reason "stop". An OpenAI
client sees an ordinary completion and its agent loop ends mid-conversation.
trae.lua now recovers the structured call from that text, strips the block from
user-visible content, and corrects finish_reason. Both tag spellings
(<tool_call>/<toolcall>, the latter is what the same codebase's Anthropic prompt
asks for) and all three argument key names (arguments/params/input) are accepted.
This is a defensive fallback: fixing the upstream shim to honour `tools` remains
the real fix, since the model still guesses parameter names.
Tests: TestNonStreamToolCallsPreserved covers all ten OpenAI-shaped adapters,
TestGeminiNonStreamToolCalls and TestOllamaNonStreamToolCalls cover their native
shapes, TestTraeTextToolCallRecovery covers both tag spellings, prose around the
block, and asserts a plain text answer never gains tool_calls.
Verified end-to-end against mock upstreams reproducing each shape: a full
two-round agent loop (tool call -> tool result -> final answer) now completes for
both the structured and the text-emitted variants.
270 lines
10 KiB
Lua
270 lines
10 KiB
Lua
-- trae 源适配器(trae-local-api 的 OpenAI 兼容代理)
|
||
-- trae-local-api 运行在 http://127.0.0.1:19900,接 Trae CN 账号的云资源。
|
||
-- 协议:OpenAI /v1/chat/completions
|
||
-- 特性:
|
||
-- - 排队:上游 busy 时排队,可能长时间无响应
|
||
-- - 非流式请求偶发返回 SSE 数据(upstream bug)
|
||
-- - 支持 thinking,由上游模型自动决定是否开启
|
||
|
||
local adapter = {}
|
||
adapter.name = "trae"
|
||
adapter.version = "1.0.0"
|
||
adapter.endpoint = "/v1/chat/completions"
|
||
adapter.headers = {}
|
||
|
||
function adapter.transform_request(raw_body)
|
||
local ok, req = pcall(json.decode, raw_body)
|
||
if not ok then return raw_body end
|
||
req.disable_thinking = nil
|
||
req.extra_body = nil
|
||
if req.messages then
|
||
for _, msg in ipairs(req.messages) do
|
||
msg.reasoning_content = nil
|
||
end
|
||
end
|
||
return json.encode(req)
|
||
end
|
||
|
||
-- parse_text_tool_calls extracts tool calls that an upstream emitted as PLAIN
|
||
-- TEXT instead of using the OpenAI tool_calls field.
|
||
--
|
||
-- trae-local-api's OpenAI endpoint (/v1/chat/completions) does not read the
|
||
-- request's `tools` array at all, so the relayed model is never told the tool
|
||
-- schema; it falls back to printing
|
||
-- <tool_call>
|
||
-- {"name": "get_weather", "arguments": {"location": "北京"}}
|
||
-- </tool_call>
|
||
-- into message.content, leaves message.tool_calls null, and reports
|
||
-- finish_reason="stop". A client following the OpenAI contract therefore never
|
||
-- sees a tool call: the agent loop terminates unexpectedly mid-conversation
|
||
-- (and a hand-written replay produces an empty function name next turn).
|
||
--
|
||
-- Both tag spellings are accepted: the same codebase's Anthropic endpoint
|
||
-- instructs models to emit <toolcall>, and models mix the two. Key names vary
|
||
-- too (arguments / params / input), so all are tried.
|
||
--
|
||
-- Returns (tool_calls_array_or_nil, content_with_blocks_removed).
|
||
local function parse_text_tool_calls(content)
|
||
if type(content) ~= "string" or content == "" then return nil, content end
|
||
if not (content:find("<tool_call", 1, true) or content:find("<toolcall", 1, true)) then
|
||
return nil, content
|
||
end
|
||
|
||
local tcs = {}
|
||
local idx = 0
|
||
|
||
local function collect(pattern)
|
||
for payload in content:gmatch(pattern) do
|
||
local ok, obj = pcall(json.decode, payload)
|
||
if ok and type(obj) == "table" then
|
||
-- some models wrap it as {"function":{"name":..,"arguments":..}}
|
||
local fn = obj["function"]
|
||
local name = obj.name or (type(fn) == "table" and fn.name) or nil
|
||
if name then
|
||
local args = obj.arguments or obj.params or obj.input
|
||
if args == nil and type(fn) == "table" then
|
||
args = fn.arguments or fn.params
|
||
end
|
||
if type(args) == "string" then
|
||
local aok, decoded = pcall(json.decode, args)
|
||
args = aok and decoded or {}
|
||
elseif type(args) ~= "table" then
|
||
args = {}
|
||
end
|
||
idx = idx + 1
|
||
table.insert(tcs, {
|
||
id = obj.id or ("call_text_" .. idx),
|
||
type = "function",
|
||
name = name,
|
||
arguments = args
|
||
})
|
||
end
|
||
end
|
||
end
|
||
end
|
||
|
||
-- `<tag ...>` allows attributes; %s* handles the "</tool_call >" spacing
|
||
-- that trae-local-api's own prompt example uses.
|
||
collect("<tool_call[^>]*>%s*(.-)%s*</tool_call%s*>")
|
||
collect("<toolcall[^>]*>%s*(.-)%s*</toolcall%s*>")
|
||
|
||
if #tcs == 0 then return nil, content end
|
||
|
||
-- drop the blocks from user-visible content; keep any surrounding prose
|
||
local stripped = content:gsub("<tool_call[^>]*>%s*.-%s*</tool_call%s*>", "")
|
||
stripped = stripped:gsub("<toolcall[^>]*>%s*.-%s*</toolcall%s*>", "")
|
||
stripped = stripped:gsub("^%s+", ""):gsub("%s+$", "")
|
||
return tcs, stripped
|
||
end
|
||
|
||
function adapter.transform_response(raw_body)
|
||
-- trae-local-api 偶发在非流式请求中返回 SSE 格式数据,
|
||
-- 表现为多个 data: {...} 行或混合了 reasoning_chunk 等。
|
||
-- 剥掉 data: 前缀,取最后一个完整的 JSON 块(通常是
|
||
-- 最终的 stop chunk 或 usage chunk)。
|
||
local use_json = raw_body
|
||
if raw_body:match("^data:") or raw_body:match("\ndata:") then
|
||
-- 取最后一个 data: 行的 JSON
|
||
local last = nil
|
||
for line in raw_body:gmatch("data: ([^\n\r]+)") do
|
||
local trimmed = line:match("^%s*(.-)%s*$")
|
||
if trimmed and trimmed ~= "[DONE]" then
|
||
local ok2, obj = pcall(json.decode, trimmed)
|
||
if ok2 and type(obj) == "table" then
|
||
last = obj
|
||
end
|
||
end
|
||
end
|
||
if last then
|
||
use_json = json.encode(last)
|
||
end
|
||
-- 如果解析失败,回退到原始 body
|
||
end
|
||
|
||
local ok, resp = pcall(json.decode, use_json)
|
||
if not ok or resp == nil then return raw_body end
|
||
|
||
local unified = {
|
||
content = "",
|
||
finish_reason = "",
|
||
token_usage = { prompt = 0, completion = 0, total = 0 }
|
||
}
|
||
|
||
if type(resp.usage) == "table" then
|
||
unified.token_usage.prompt = resp.usage.prompt_tokens or 0
|
||
unified.token_usage.completion = resp.usage.completion_tokens or 0
|
||
unified.token_usage.total = resp.usage.total_tokens or 0
|
||
end
|
||
|
||
if type(resp.choices) == "table" and #resp.choices > 0 then
|
||
local ch = resp.choices[1]
|
||
if type(ch.message) == "table" then
|
||
unified.content = ch.message.content or ""
|
||
if ch.message.reasoning_content then
|
||
unified.reasoning_content = ch.message.reasoning_content
|
||
end
|
||
if type(ch.message.tool_calls) == "table" and #ch.message.tool_calls > 0 then
|
||
local tcs = {}
|
||
for _, tc in ipairs(ch.message.tool_calls) do
|
||
local fn = tc["function"] or {}
|
||
local args = fn.arguments
|
||
if type(args) == "string" then
|
||
local args_ok, decoded = pcall(json.decode, args)
|
||
args = args_ok and decoded or {}
|
||
elseif type(args) ~= "table" then
|
||
args = {}
|
||
end
|
||
table.insert(tcs, {
|
||
id = tc.id,
|
||
type = tc.type or "function",
|
||
name = fn.name,
|
||
arguments = args
|
||
})
|
||
end
|
||
unified.tool_calls = tcs
|
||
end
|
||
end
|
||
unified.finish_reason = ch.finish_reason or ""
|
||
|
||
-- Fallback: trae-local-api's OpenAI endpoint drops the request's
|
||
-- `tools` array entirely, so the relayed model is never told the tool
|
||
-- schema and instead PRINTS a <tool_call>{...}</tool_call> block into
|
||
-- content, leaving message.tool_calls null and finish_reason="stop".
|
||
-- A client following the OpenAI contract then sees a normal completion
|
||
-- and its agent loop terminates mid-conversation. Recover the
|
||
-- structured call so the loop can continue.
|
||
if unified.tool_calls == nil then
|
||
local recovered, cleaned = parse_text_tool_calls(unified.content)
|
||
if recovered then
|
||
unified.tool_calls = recovered
|
||
unified.content = cleaned
|
||
unified.finish_reason = "tool_calls"
|
||
end
|
||
end
|
||
end
|
||
|
||
return json.encode(unified)
|
||
end
|
||
|
||
function adapter.transform_stream_chunk(raw_chunk)
|
||
local ok, chunk = pcall(json.decode, raw_chunk)
|
||
if not ok then return "" end
|
||
|
||
local uses = nil
|
||
if type(chunk.usage) == "table" then
|
||
uses = {
|
||
prompt = chunk.usage.prompt_tokens or chunk.usage.prompt or 0,
|
||
completion = chunk.usage.completion_tokens or chunk.usage.completion or 0,
|
||
total = chunk.usage.total_tokens or chunk.usage.total or 0,
|
||
}
|
||
if type(chunk.usage.prompt_tokens_details) == "table" and chunk.usage.prompt_tokens_details.cached_tokens ~= nil then
|
||
uses.prompt_tokens_details = { cached_tokens = chunk.usage.prompt_tokens_details.cached_tokens }
|
||
elseif (chunk.usage.prompt_cache_hit_tokens or 0) > 0 then
|
||
uses.prompt_cache_hit_tokens = chunk.usage.prompt_cache_hit_tokens
|
||
uses.prompt_cache_miss_tokens = chunk.usage.prompt_cache_miss_tokens or 0
|
||
uses.prompt_tokens_details = { cached_tokens = chunk.usage.prompt_cache_hit_tokens }
|
||
end
|
||
end
|
||
|
||
if not chunk.choices or #chunk.choices == 0 then
|
||
if uses ~= nil then
|
||
return json.encode({ usage = uses, done = false })
|
||
end
|
||
return ""
|
||
end
|
||
local delta = chunk.choices[1].delta or {}
|
||
local fr = chunk.choices[1].finish_reason
|
||
|
||
local finish = (type(fr) == "string" and fr ~= "") and fr or nil
|
||
|
||
local unified = {
|
||
content = delta.content or "",
|
||
done = (finish ~= nil)
|
||
}
|
||
if finish then
|
||
unified.finish_reason = finish
|
||
end
|
||
if uses ~= nil then
|
||
unified.usage = uses
|
||
end
|
||
if delta.reasoning_content then
|
||
unified.reasoning_content = delta.reasoning_content
|
||
end
|
||
if delta.tool_calls then
|
||
unified.tool_calls = delta.tool_calls
|
||
end
|
||
return json.encode(unified)
|
||
end
|
||
|
||
function adapter.transform_error(status, body)
|
||
local ok, resp = pcall(json.decode, body)
|
||
if not ok or type(resp) ~= "table" then
|
||
-- Non-JSON body: extract title from HTML or first line
|
||
local title = string.match(body or "", "<title>(.-)</title>")
|
||
if title and title ~= "" then return title end
|
||
local line = string.match(body or "", "^%s*([^\r\n]+)")
|
||
if line and line ~= "" and not string.match(line, "^<") then
|
||
return string.sub(line, 1, 200)
|
||
end
|
||
return nil
|
||
end
|
||
local e = resp.error
|
||
if type(e) == "table" and type(e.message) == "string" then
|
||
-- trae quota / rate limit annotations
|
||
if e.code then
|
||
return tostring(e.code) .. ": " .. e.message
|
||
end
|
||
return e.message
|
||
end
|
||
if type(e) == "string" then return e end
|
||
if type(resp.message) == "string" and resp.message ~= "" then
|
||
if resp.code ~= nil then
|
||
return tostring(resp.code) .. ": " .. resp.message
|
||
end
|
||
return resp.message
|
||
end
|
||
return nil
|
||
end
|
||
|
||
return adapter
|