Files
ModelRouter/internal/lua/adapters/sensenova.lua
JianFeeeee 2e3d5b79ad fix(adapters): stop dropping non-streaming tool calls (agent loops died on turn 2)
Four adapters handled tool_calls in transform_stream_chunk but lost them in
transform_response, so any NON-streaming tool-using conversation broke on its
second request: the client received finish_reason:"tool_calls" with no
tool_calls payload, replayed an assistant message whose function
name/arguments were empty, and the upstream rejected the next turn with

    400 invalid tool_call function, function/name/arguments cannot be empty

The production audit trail shows 46 such failures on sensenova alone.

- sensenova.lua: forward message.tool_calls, decoding the arguments JSON string
  into an object as the unified shape expects.
- gemini.lua: collect functionCall parts from candidates[].content.parts. Also
  correct finish_reason, since Gemini reports "STOP" even when it emitted a
  function call and clients keyed on it treat that as a finished answer.
- ollama.lua: the field was initialized to an empty table and never filled;
  fill it and likewise correct done_reason "stop" -> "tool_calls".

trae is a different failure with the same symptom: trae-local-api's OpenAI
endpoint (/v1/chat/completions, src/server.js:353) never reads the request's
`tools` array — only its Anthropic endpoint does — so the relayed model is never
told the tool schema and instead PRINTS a <tool_call>{...}</tool_call> block into
content, leaving message.tool_calls null and finish_reason "stop". An OpenAI
client sees an ordinary completion and its agent loop ends mid-conversation.
trae.lua now recovers the structured call from that text, strips the block from
user-visible content, and corrects finish_reason. Both tag spellings
(<tool_call>/<toolcall>, the latter is what the same codebase's Anthropic prompt
asks for) and all three argument key names (arguments/params/input) are accepted.
This is a defensive fallback: fixing the upstream shim to honour `tools` remains
the real fix, since the model still guesses parameter names.

Tests: TestNonStreamToolCallsPreserved covers all ten OpenAI-shaped adapters,
TestGeminiNonStreamToolCalls and TestOllamaNonStreamToolCalls cover their native
shapes, TestTraeTextToolCallRecovery covers both tag spellings, prose around the
block, and asserts a plain text answer never gains tool_calls.

Verified end-to-end against mock upstreams reproducing each shape: a full
two-round agent loop (tool call -> tool result -> final answer) now completes for
both the structured and the text-emitted variants.
2026-08-31 10:23:35 +08:00

175 lines
6.9 KiB
Lua
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

-- sensenova API format adapter
-- streaming: delta only has reasoning_content, no content field
-- non-streaming: has both content and reasoning_content
local adapter = {}
adapter.name = "sensenova"
adapter.version = "1.0.0"
adapter.endpoint = "/chat/completions"
adapter.headers = {}
-- Same as openai - strip provider-specific fields
function adapter.transform_request(raw_body)
local ok, req = pcall(json.decode, raw_body)
if not ok then return raw_body end
req.disable_thinking = nil
req.extra_body = nil
if req.messages then
for _, msg in ipairs(req.messages) do
msg.reasoning_content = nil
end
end
return json.encode(req)
end
-- Same as openai - extract content from response
function adapter.transform_response(raw_body)
local ok, resp = pcall(json.decode, raw_body)
if not ok or resp == nil then return raw_body end
local unified = {
content = "",
finish_reason = "",
token_usage = { prompt = 0, completion = 0, total = 0 }
}
if type(resp.usage) == "table" then
unified.token_usage.prompt = resp.usage.prompt_tokens or 0
unified.token_usage.completion = resp.usage.completion_tokens or 0
unified.token_usage.total = resp.usage.total_tokens or 0
local hit = 0
if type(resp.usage.prompt_tokens_details) == "table" and resp.usage.prompt_tokens_details.cached_tokens ~= nil then
hit = resp.usage.prompt_tokens_details.cached_tokens
unified.token_usage.prompt_tokens_details = { cached_tokens = hit }
end
if (resp.usage.prompt_cache_hit_tokens or 0) > 0 then
unified.token_usage.prompt_cache_hit_tokens = resp.usage.prompt_cache_hit_tokens
unified.token_usage.prompt_cache_miss_tokens = resp.usage.prompt_cache_miss_tokens or 0
if hit == 0 then
unified.token_usage.prompt_tokens_details = { cached_tokens = resp.usage.prompt_cache_hit_tokens }
end
end
end
if type(resp.choices) == "table" and #resp.choices > 0 then
local ch = resp.choices[1]
if type(ch.message) == "table" then
unified.content = ch.message.content or ""
if ch.message.reasoning_content then
unified.reasoning_content = ch.message.reasoning_content
elseif ch.message.reasoning then
-- sensenova-6.8-flash-lite 用 reasoning 字段而不是 reasoning_content
unified.reasoning_content = ch.message.reasoning
end
-- Tool calls MUST be forwarded. Dropping them while keeping
-- finish_reason="tool_calls" makes the client replay an assistant
-- message whose function name/arguments are empty, and sensenova
-- then rejects the next turn with
-- 400 invalid tool_call function, function/name/arguments cannot be empty
-- i.e. a tool-using conversation dies on its second request.
if type(ch.message.tool_calls) == "table" and #ch.message.tool_calls > 0 then
local tcs = {}
for _, tc in ipairs(ch.message.tool_calls) do
local fn = tc["function"] or {}
-- arguments arrives as a JSON *string* on the wire; the
-- unified shape expects a decoded object.
local args = fn.arguments
if type(args) == "string" then
local args_ok, decoded = pcall(json.decode, args)
args = args_ok and decoded or {}
elseif type(args) ~= "table" then
args = {}
end
table.insert(tcs, {
id = tc.id,
type = tc.type or "function",
name = fn.name,
arguments = args
})
end
unified.tool_calls = tcs
end
end
unified.finish_reason = ch.finish_reason or ""
end
return json.encode(unified)
end
-- Sensenova-specific stream handling
-- Upstream puts content in reasoning_content only (no content field)
-- Also sends finish_reason="" (empty string) on every chunk
function adapter.transform_stream_chunk(raw_chunk)
local ok, chunk = pcall(json.decode, raw_chunk)
if not ok then return "" end
local uses = nil
if type(chunk.usage) == "table" then
uses = {
prompt = chunk.usage.prompt_tokens or chunk.usage.prompt or 0,
completion = chunk.usage.completion_tokens or chunk.usage.completion or 0,
total = chunk.usage.total_tokens or chunk.usage.total or 0,
}
if type(chunk.usage.prompt_tokens_details) == "table" and chunk.usage.prompt_tokens_details.cached_tokens ~= nil then
uses.prompt_tokens_details = { cached_tokens = chunk.usage.prompt_tokens_details.cached_tokens }
elseif (chunk.usage.prompt_cache_hit_tokens or 0) > 0 then
uses.prompt_cache_hit_tokens = chunk.usage.prompt_cache_hit_tokens
uses.prompt_cache_miss_tokens = chunk.usage.prompt_cache_miss_tokens or 0
uses.prompt_tokens_details = { cached_tokens = chunk.usage.prompt_cache_hit_tokens }
end
end
if not chunk.choices or #chunk.choices == 0 then
if uses ~= nil then
return json.encode({ usage = uses, done = false })
end
return ""
end
local delta = chunk.choices[1].delta or {}
-- Sensenova: delta has reasoning_content (deepseek-v4-flash) or reasoning
-- (sensenova-6.8-flash-lite) but no content field
local content = delta.content or ""
if content == "" and (delta.reasoning_content or delta.reasoning) then
content = delta.reasoning_content or delta.reasoning
end
-- Sensenova: finish_reason is "" on every chunk, "stop" on last
local fr = chunk.choices[1].finish_reason
local done = (fr == "stop" or fr == "length")
local unified = {
content = content,
done = done,
}
if chunk.choices[1].finish_reason and chunk.choices[1].finish_reason ~= "" then
unified.finish_reason = chunk.choices[1].finish_reason
end
if delta.tool_calls then
unified.tool_calls = delta.tool_calls
end
if uses ~= nil then
unified.usage = uses
end
return json.encode(unified)
end
-- 错误收敛sensenova 为 OpenAI 风格 {error:{message,...}}
-- 配额类错误单独点出便于客户端识别重置周期。
function adapter.transform_error(status, body)
local ok, resp = pcall(json.decode, body)
if not ok or type(resp) ~= "table" then return nil end
local e = resp.error
if type(e) ~= "table" then return nil end
if status == 429 and type(e.code) == "string"
and e.code == "insufficient_quota" then
return "workspace quota exhausted (resets periodically)"
end
if type(e.message) == "string" then return e.message end
return nil
end
return adapter