Files
ModelRouter/internal/lua/adapters/openai.lua
JianFeeeee 131a42a169 fix(adapters): sanitize tool-call ids so one bad upstream can't kill every Claude slot
Anthropic requires tool_use.id / tool_result.tool_use_id to match
^[a-zA-Z0-9_-]{1,64}$ and rejects the WHOLE request otherwise with
REQUEST_BODY_INVALID / "Invalid tool use format". OpenAI has no such rule, so
an OpenAI-compatible model can mint an id like "bash:0"
(xinjianya/moonshotai/kimi-k3 does exactly that).

In a fan-out router that id does not stay local: the client stores it in its
history and replays it to every other source. One such id therefore kills
every Claude slot at once — justwoker, tabitoken and 扇贝 are all
Claude-behind-{OpenAI,Anthropic} — and an AUTO request falls through all four
tiers to whatever tolerant model is left. Observed live: 4 consecutive 503
"all N auto providers failed" with tier 1/2/3 each reporting the same 400.

Both directions are sanitized, in both adapters:
  - request:  tool_calls[].id and tool_call_id, so poisoned history recovers
  - response: non-streaming tool_calls[].id and the first streamed fragment,
              so a bad id never enters a client session in the first place

safe_tool_id is pure and deterministic, so a call and its result are rewritten
identically within one request. A rewritten id keeps an 8-hex digest of the
original, without which distinct ids could collapse ("a:b" and "a_b") into a
duplicate/unpaired tool_use. Already-legal ids pass through byte-identical, so
well-behaved traffic is unaffected. openai.lua carries its own copy because
Lua adapters have no shared prelude.

Streamed argument fragments carry no id and must stay id-less, otherwise
index-based accumulation on the client breaks; a test pins that.
2026-09-05 22:18:31 +08:00

229 lines
9.3 KiB
Lua

local adapter = {}
adapter.name = "openai"
adapter.version = "2.0.0"
adapter.endpoint = "/chat/completions"
adapter.headers = {}
-- Claude-behind-OpenAI upstreams (tabitoken, 扇贝, …) convert /chat/completions
-- to the Anthropic Messages API internally, so they inherit Anthropic's
-- tool id rule: ^[a-zA-Z0-9_-]{1,64}$, enforced by rejecting the WHOLE request
-- with "Invalid tool use format" / "tool_use.id: String should match pattern".
-- Plain OpenAI has no such rule, so an OpenAI-compatible model can mint an id
-- like "bash:0" (observed from moonshotai/kimi-k3). In a fan-out router that id
-- is replayed to every other source, so one such id kills every Claude slot at
-- once and an AUTO request falls through all tiers.
--
-- safe_tool_id is pure and deterministic, so a tool_calls entry and its
-- matching tool_call_id are rewritten identically within one request. A
-- rewritten id keeps an 8-hex digest of the ORIGINAL id, without which two
-- distinct ids could collapse into one ("a:b" and "a_b") and become an
-- unpaired/duplicate tool call. Already-legal ids are returned untouched, so
-- well-behaved traffic is byte-identical to before.
-- (anthropic.lua carries the same helper; Lua adapters have no shared prelude.)
local TOOL_ID_MAX = 64
local function safe_tool_id(id)
if type(id) ~= "string" or id == "" then return id end
local clean = string.gsub(id, "[^A-Za-z0-9_-]", "_")
if clean == id and #clean <= TOOL_ID_MAX then
return clean
end
local digest = string.sub(sha256_hex(id), 1, 8)
local keep = TOOL_ID_MAX - #digest - 1
if #clean > keep then clean = string.sub(clean, 1, keep) end
return clean .. "_" .. digest
end
-- OpenAI /chat/completions format (pass-through, strip provider-specific fields)
function adapter.transform_request(raw_body)
local ok, req = pcall(json.decode, raw_body)
if not ok then return raw_body end
req.disable_thinking = nil
req.extra_body = nil
if req.messages then
for _, msg in ipairs(req.messages) do
msg.reasoning_content = nil
if msg.tool_call_id ~= nil then
msg.tool_call_id = safe_tool_id(msg.tool_call_id)
end
if type(msg.tool_calls) == "table" then
for _, tc in ipairs(msg.tool_calls) do
if type(tc) == "table" and tc.id ~= nil then
tc.id = safe_tool_id(tc.id)
end
end
end
end
end
return json.encode(req)
end
function adapter.transform_response(raw_body)
local ok, resp = pcall(json.decode, raw_body)
if not ok or resp == nil then return raw_body end
local unified = {
content = "",
finish_reason = "",
token_usage = { prompt = 0, completion = 0, total = 0 }
}
if type(resp.usage) == "table" then
unified.token_usage.prompt = resp.usage.prompt_tokens or 0
unified.token_usage.completion = resp.usage.completion_tokens or 0
unified.token_usage.total = resp.usage.total_tokens or 0
-- Cache passthrough: OpenAI v2 prompt_tokens_details.cached_tokens
-- and DeepSeek-legacy prompt_cache_hit/miss_tokens. dsh reads
-- prompt_tokens_details.cached_tokens (falls back to the legacy
-- standalone field), so both shapes reach clients.
local hit = 0
if type(resp.usage.prompt_tokens_details) == "table" and resp.usage.prompt_tokens_details.cached_tokens ~= nil then
hit = resp.usage.prompt_tokens_details.cached_tokens
unified.token_usage.prompt_tokens_details = { cached_tokens = hit }
end
if (resp.usage.prompt_cache_hit_tokens or 0) > 0 then
unified.token_usage.prompt_cache_hit_tokens = resp.usage.prompt_cache_hit_tokens
unified.token_usage.prompt_cache_miss_tokens = resp.usage.prompt_cache_miss_tokens or 0
if hit == 0 then
unified.token_usage.prompt_tokens_details = { cached_tokens = resp.usage.prompt_cache_hit_tokens }
end
end
end
if type(resp.choices) == "table" and #resp.choices > 0 then
local ch = resp.choices[1]
if type(ch.message) == "table" then
unified.content = ch.message.content or ""
if ch.message.reasoning_content then
unified.reasoning_content = ch.message.reasoning_content
end
if type(ch.message.tool_calls) == "table" then
local tcs = {}
for _, tc in ipairs(ch.message.tool_calls) do
local args_ok, args = pcall(json.decode, tc["function"].arguments)
if not args_ok then args = {} end
table.insert(tcs, {
id = safe_tool_id(tc.id),
type = tc.type or "function",
name = tc["function"].name,
arguments = args
})
end
unified.tool_calls = tcs
end
end
unified.finish_reason = ch.finish_reason or ""
end
return json.encode(unified)
end
function adapter.transform_stream_chunk(raw_chunk)
local ok, chunk = pcall(json.decode, raw_chunk)
if not ok then return "" end
-- OpenAI-style streams may attach usage to a chunk with empty choices
-- (the final usage chunk). Keys must match Go's TokenUsage json tags
-- (prompt/completion/total); the gateway re-emits standard *_tokens.
local uses = nil
if type(chunk.usage) == "table" then
uses = {
prompt = chunk.usage.prompt_tokens or chunk.usage.prompt or 0,
completion = chunk.usage.completion_tokens or chunk.usage.completion or 0,
total = chunk.usage.total_tokens or chunk.usage.total or 0,
}
if type(chunk.usage.prompt_tokens_details) == "table" and chunk.usage.prompt_tokens_details.cached_tokens ~= nil then
uses.prompt_tokens_details = { cached_tokens = chunk.usage.prompt_tokens_details.cached_tokens }
elseif (chunk.usage.prompt_cache_hit_tokens or 0) > 0 then
uses.prompt_cache_hit_tokens = chunk.usage.prompt_cache_hit_tokens
uses.prompt_cache_miss_tokens = chunk.usage.prompt_cache_miss_tokens or 0
uses.prompt_tokens_details = { cached_tokens = chunk.usage.prompt_cache_hit_tokens }
end
end
if not chunk.choices or #chunk.choices == 0 then
if uses ~= nil then
-- usage-only chunk is not a content/finish signal; the gateway
-- emits its own terminal stop chunk and merges this usage.
return json.encode({ usage = uses, done = false })
end
return ""
end
local delta = chunk.choices[1].delta or {}
local fr = chunk.choices[1].finish_reason
local finish = (type(fr) == "string" and fr ~= "") and fr or nil
local unified = {
content = delta.content or "",
done = (finish ~= nil)
}
if finish then
unified.finish_reason = finish
end
if uses ~= nil then
unified.usage = uses
end
if delta.reasoning_content then
unified.reasoning_content = delta.reasoning_content
end
if delta.tool_calls then
-- Sanitize on the way OUT too: an id this upstream happily minted (it
-- does not validate them) becomes a landmine once the client replays
-- it to a Claude upstream. Only the first fragment of a streamed call
-- carries an id; later argument fragments have none and are untouched.
for _, tc in ipairs(delta.tool_calls) do
if type(tc) == "table" and tc.id ~= nil then
tc.id = safe_tool_id(tc.id)
end
end
unified.tool_calls = delta.tool_calls
end
return json.encode(unified)
end
-- 错误收敛:标准 OpenAI 信封 {error:{message,...}}
function adapter.transform_error(status, body)
-- Non-JSON body (nginx HTML error pages, plain text): extract a short
-- human-readable reason instead of letting the raw body reach the log.
local ok, resp = pcall(json.decode, body)
if not ok or type(resp) ~= "table" then
-- HTML error page: pull the <title> text (e.g. "413 Request Entity Too Large")
local title = string.match(body or "", "<title>(.-)</title>")
if title and title ~= "" then return title end
-- Bare text: first non-empty line, capped
local line = string.match(body or "", "^%s*([^\r\n]+)")
if line and line ~= "" and not string.match(line, "^<") then
return string.sub(line, 1, 200)
end
return nil
end
-- Standard OpenAI envelope: {error:{message,...}} or {error:"..."}
local e = resp.error
if type(e) == "table" and type(e.message) == "string" then
return e.message
end
if type(e) == "string" then return e end
-- Flat envelope used by many OpenAI-compatible gateways:
-- {"code":20012,"message":"Model does not exist..."}
-- {"code":"INVALID_API_KEY","message":"Invalid API key"}
if type(resp.message) == "string" and resp.message ~= "" then
if resp.code ~= nil then
return tostring(resp.code) .. ": " .. resp.message
end
return resp.message
end
-- Some gateways use {detail:"..."} (FastAPI style)
if type(resp.detail) == "string" and resp.detail ~= "" then
return resp.detail
end
return nil
end
return adapter