fix: anthropic tool-call round-trip, cache zero-hit parity, round-robin load balancing

anthropic.lua v3.0.0:
- Issue 1: tool_result/tool_use round-trip
- Issue 3: thinking default OFF (opt-in via extra_body.thinking)
- Issue 4: tool_choice mapping
- Issue 5: collect_blocks preserves unknown part types
- message_stop no longer emits done=true (was overwriting tool_calls finish_reason)
- cache_read_input_tokens normalized even at 0

gemini.lua:
- transform_response was missing cachedContentTokenCount

openai.lua (Issue 6):
- transform_error handles flat envelopes, nginx HTML, bare text

chat.go mergeUsage:
- Keep PromptTokensDetails even when CachedTokens=0

scheduler.go:
- Remove sort.SliceStable by Pref; round-robin cursor is the only LB mechanism

provider.go ModelAvailable:
- Also check Pref() > prefMin, persistently failing slots exit cands

presets.go:
- 17 built-in source templates

Tests: 6 new test functions, 2 updated for new semantics
This commit is contained in:
JianFeeeee
2026-08-28 12:02:46 +08:00
parent 94cbcb6771
commit 624fd74b45
15 changed files with 1170 additions and 114 deletions

View File

@ -1,143 +1,301 @@
local adapter = {}
adapter.name = "anthropic"
adapter.version = "2.0.0"
adapter.version = "3.0.0"
adapter.endpoint = "/v1/messages"
adapter.headers = {
["anthropic-version"] = "2023-06-01"
["anthropic-version"] = "2023-06-01",
}
function adapter.transform_request(raw_body)
local ok, req = pcall(json.decode, raw_body)
if not ok then return raw_body end
-- ============================================================
-- helpers
-- ============================================================
-- 将 OpenAI 风格 content字符串或 [{type:*}] 数组)拆成文本/图片块
local function collect_blocks(content)
if type(content) == "string" then
return { { type = "text", text = content } }
end
local blocks = {}
for _, p in ipairs(content or {}) do
local ROLE_WHITELIST = {
system = true, user = true, assistant = true, tool = true,
}
local function collect_blocks(content)
if type(content) == "string" then
if content == "" then return {} end
return { { type = "text", text = content } }
end
if type(content) ~= "table" then return {} end
local blocks = {}
for _, p in ipairs(content) do
if type(p) == "string" then
table.insert(blocks, { type = "text", text = p })
elseif type(p) == "table" then
if p.type == "text" then
table.insert(blocks, { type = "text", text = p.text })
table.insert(blocks, { type = "text", text = p.text or "" })
elseif p.type == "image_url" and type(p.image_url) == "table" and p.image_url.url then
local mt, b64 = string.match(p.image_url.url, "^data:([^,]+);base64,(.+)$")
local url = p.image_url.url
local mt, b64 = string.match(url, "^data:([^,]+);base64,(.+)$")
if b64 then
table.insert(blocks, { type = "image", source = { type = "base64", media_type = mt or "image/png", data = b64 } })
else
table.insert(blocks, { type = "image", source = { type = "url", url = p.image_url.url } })
table.insert(blocks, { type = "image", source = { type = "url", url = url } })
end
else
-- Unknown content type: preserve for forward compatibility
table.insert(blocks, p)
end
end
return blocks
end
local function text_of(content)
if type(content) == "string" then return content end
local t = ""
for _, p in ipairs(content or {}) do
if p.type == "text" and p.text then t = t .. p.text end
return blocks
end
local function text_of(content)
if type(content) == "string" then return content end
local t = ""
for _, p in ipairs(content or {}) do
if type(p) == "string" then
t = t .. p
elseif type(p) == "table" and p.type == "text" and p.text then
t = t .. p.text
end
return t
end
return t
end
--- Append blocks to the last user message (merge) or create a new one.
local function append_user(msgs, blocks)
if #msgs > 0 and msgs[#msgs].role == "user" then
for _, b in ipairs(blocks) do
table.insert(msgs[#msgs].content, b)
end
else
table.insert(msgs, { role = "user", content = blocks })
end
end
-- ============================================================
-- transform_request (OpenAI → Anthropic)
-- ============================================================
function adapter.transform_request(raw_body)
local ok, req = pcall(json.decode, raw_body)
if not ok or type(req) ~= "table" then return raw_body end
local msgs = {}
local system = ""
local pending_tool = {} -- accumulated tool_result blocks
for _, m in ipairs(req.messages or {}) do
if m.role == "system" then
local role = m.role or "user"
if role == "system" then
system = system .. text_of(m.content) .. "\n"
elseif role == "assistant" then
-- Build content array: text/image blocks from content + tool_use from tool_calls
local blocks = collect_blocks(m.content)
if type(m.tool_calls) == "table" then
for _, tc in ipairs(m.tool_calls) do
if type(tc) == "table" then
local fn = tc["function"] or {}
local args = fn.arguments
local input = {}
if type(args) == "string" and args ~= "" then
local ok2, parsed = pcall(json.decode, args)
if ok2 and type(parsed) == "table" then input = parsed end
elseif type(args) == "table" then
input = args
end
table.insert(blocks, {
type = "tool_use",
id = tc.id or "",
name = fn.name or "",
input = input,
})
end
end
end
table.insert(msgs, { role = "assistant", content = blocks })
elseif role == "tool" then
-- Accumulate consecutive tool results; will be flushed as one
-- user message with tool_result content blocks.
table.insert(pending_tool, {
type = "tool_result",
tool_use_id = m.tool_call_id or "",
content = text_of(m.content),
})
else
table.insert(msgs, { role = m.role, content = collect_blocks(m.content) })
-- user / any other role: flush pending tool results first
if #pending_tool > 0 then
append_user(msgs, pending_tool)
pending_tool = {}
end
append_user(msgs, collect_blocks(m.content))
end
end
local anthropic_req = {
model = req.model or "claude-sonnet-4-20250514",
max_tokens = req.max_tokens or 4096,
messages = msgs,
stream = req.stream or false,
}
if not req.disable_thinking then
anthropic_req.thinking = { type = "enabled", budget_tokens = 4096 }
-- Flush any trailing tool results
if #pending_tool > 0 then
append_user(msgs, pending_tool)
end
-- ── Build Anthropic request ──────────────────────────────
local anthropic_req = {
model = req.model or "claude-sonnet-4-20250514",
max_tokens = req.max_tokens or 4096,
messages = msgs,
stream = req.stream or false,
}
-- ── tools ────────────────────────────────────────────────
local has_tools = false
if req.tools and type(req.tools) == "table" then
local tools = {}
for _, t in ipairs(req.tools) do
if type(t) == "table" and t.type == "function"
and type(t["function"]) == "table" then
local fn = t["function"]
table.insert(tools, {
name = fn.name or "",
description = fn.description or "",
input_schema = fn.parameters or { type = "object", properties = {} },
})
end
end
if #tools > 0 then
anthropic_req.tools = tools
has_tools = true
end
end
-- ── tool_choice ──────────────────────────────────────────
-- OpenAI → Anthropic mapping:
-- "auto" → {type:"auto"}
-- "none" → remove tools entirely (Anthropic has no "none")
-- "required" → {type:"any"}
-- {type:"function", function:{name:"X"}} → {type:"tool", name:"X"}
if has_tools and req.tool_choice then
local tc = req.tool_choice
if type(tc) == "string" then
if tc == "auto" then
anthropic_req.tool_choice = { type = "auto" }
elseif tc == "none" then
anthropic_req.tools = nil
anthropic_req.tool_choice = nil
elseif tc == "required" then
anthropic_req.tool_choice = { type = "any" }
end
elseif type(tc) == "table" and tc.type == "function" then
local fn = tc["function"] or {}
anthropic_req.tool_choice = { type = "tool", name = fn.name or "" }
end
end
-- ── thinking (opt-in via extra_body) ──────────────────────
-- Default: OFF. Client sends extra_body.thinking to enable.
-- Example: {"extra_body": {"thinking": {"type": "enabled", "budget_tokens": 4096}}}
if type(req.extra_body) == "table" and type(req.extra_body.thinking) == "table" then
anthropic_req.thinking = req.extra_body.thinking
end
-- ── system ───────────────────────────────────────────────
if system ~= "" then
anthropic_req.system = system
-- Strip trailing newline from accumulation
anthropic_req.system = string.match(system, "^(.-)\n*$")
end
return json.encode(anthropic_req)
end
-- ============================================================
-- transform_response (Anthropic → OpenAI, non-streaming)
-- ============================================================
function adapter.transform_response(raw_body)
local ok, resp = pcall(json.decode, raw_body)
if not ok then return raw_body end
if not ok or resp == nil then return raw_body end
local unified = {
content = "",
finish_reason = "",
token_usage = { prompt = 0, completion = 0, total = 0 }
token_usage = { prompt = 0, completion = 0, total = 0 },
}
if resp.usage then
unified.token_usage.prompt = resp.usage.input_tokens or 0
unified.token_usage.prompt = resp.usage.input_tokens or 0
unified.token_usage.completion = resp.usage.output_tokens or 0
unified.token_usage.total = (resp.usage.input_tokens or 0) + (resp.usage.output_tokens or 0)
-- Anthropic reports cache_read_input_tokens; normalize into
-- OpenAI-standard prompt_tokens_details.cached_tokens so clients
-- (dsh) see the cache hit count.
local cacheRead = resp.usage.cache_read_input_tokens or 0
if cacheRead > 0 then
unified.token_usage.prompt_tokens_details = { cached_tokens = cacheRead }
unified.token_usage.total = (resp.usage.input_tokens or 0) + (resp.usage.output_tokens or 0)
-- Emit details whenever Anthropic reports the field, even at 0, so a
-- reported cache miss stays distinguishable from "not reported".
if resp.usage.cache_read_input_tokens ~= nil then
unified.token_usage.prompt_tokens_details = {
cached_tokens = resp.usage.cache_read_input_tokens
}
end
end
if resp.content and #resp.content > 0 then
local tcs = {}
for _, block in ipairs(resp.content) do
if block.type == "text" then
unified.content = unified.content .. (block.text or "")
elseif block.type == "thinking" and block.thinking then
unified.reasoning_content = (unified.reasoning_content or "") .. block.thinking
elseif block.type == "tool_use" then
table.insert(tcs, {
id = block.id or "",
type = "function",
name = block.name or "",
arguments = block.input or {},
})
end
end
if #tcs > 0 then unified.tool_calls = tcs end
end
unified.finish_reason = resp.stop_reason or ""
if unified.finish_reason == "tool_use" then
unified.finish_reason = "tool_calls"
elseif unified.finish_reason == "max_tokens" then
unified.finish_reason = "length"
elseif unified.finish_reason == "end_turn"
or unified.finish_reason == "stop_sequence" then
unified.finish_reason = "stop"
end
return json.encode(unified)
end
-- ============================================================
-- transform_stream_chunk (Anthropic SSE → OpenAI SSE delta)
-- ============================================================
function adapter.transform_stream_chunk(raw_chunk)
local ok, chunk = pcall(json.decode, raw_chunk)
if not ok then return "" end
-- ── message_start: initial usage ─────────────────────────
if chunk.type == "message_start" then
local uses = nil
if chunk.message and type(chunk.message.usage) == "table" then
local u = chunk.message.usage
local p = u.input_tokens or 0
local c = u.output_tokens or 0
if p > 0 or c > 0 then
uses = { prompt = p, completion = c, total = p + c }
local cacheRead = u.cache_read_input_tokens or 0
if cacheRead > 0 then
uses.prompt_tokens_details = { cached_tokens = cacheRead }
local uses = { prompt = p, completion = c, total = p + c }
if u.cache_read_input_tokens ~= nil then
uses.prompt_tokens_details = {
cached_tokens = u.cache_read_input_tokens
}
end
return json.encode({ usage = uses, done = false })
end
end
if uses ~= nil then
return json.encode({ usage = uses, done = false })
end
return ""
end
-- ── message_delta: stop_reason + final usage ─────────────
if chunk.type == "message_delta" then
local uses = nil
if type(chunk.usage) == "table" then
local u = chunk.usage
local p = u.input_tokens or 0
local c = u.output_tokens or 0
if p > 0 or c > 0 then
uses = { prompt = p, completion = c, total = p + c }
end
end
local finish = nil
if chunk.delta and chunk.delta.stop_reason ~= nil then
-- Anthropic stop_reason -> OpenAI finish_reason
local sr = chunk.delta.stop_reason
if sr == "max_tokens" then
finish = "length"
@ -147,58 +305,95 @@ function adapter.transform_stream_chunk(raw_chunk)
finish = "stop"
end
end
local uses = nil
if type(chunk.usage) == "table" then
local u = chunk.usage
local p = u.input_tokens or 0
local c = u.output_tokens or 0
if p > 0 or c > 0 then
uses = { prompt = p, completion = c, total = p + c }
end
end
if uses ~= nil then
-- completion is final here; prompt is merged from message_start
return json.encode({ content = "", done = (finish ~= nil), finish_reason = finish, usage = uses })
end
return json.encode({ content = "", done = (finish ~= nil), finish_reason = finish })
end
if chunk.type == "content_block_start" and chunk.content_block
and chunk.content_block.type == "tool_use" then
-- first fragment of a tool call: emit index + id + name, empty args
return json.encode({
content = "", done = false,
tool_calls = { {
index = chunk.index or 0,
id = chunk.content_block.id or "",
type = "function",
["function"] = { name = chunk.content_block.name or "", arguments = "" }
} }
})
-- ── content_block_start: begin text / thinking / tool_use ─
if chunk.type == "content_block_start" and chunk.content_block then
local cb = chunk.content_block
if cb.type == "tool_use" then
-- Pass Anthropic content_block index through; Go-side rewrites
-- to sequential OpenAI tool_call ordinal for multi-tool streams.
return json.encode({
content = "", done = false,
tool_calls = { {
index = chunk.index or 0,
id = cb.id or "",
type = "function",
["function"] = { name = cb.name or "", arguments = "" },
} },
})
end
return "" -- text / thinking block start: no OpenAI equivalent
end
-- ── content_block_delta: incremental content ──────────────
if chunk.type == "content_block_delta" and chunk.delta then
if chunk.delta.type == "input_json_delta" then
-- incremental JSON fragment; clients accumulate across chunks
local unified = { content = "", done = false, tool_calls = { {
index = chunk.index or 0,
id = "",
type = "function",
["function"] = { name = "", arguments = chunk.delta.partial_json or "" }
} } }
return json.encode(unified)
local d = chunk.delta
if d.type == "input_json_delta" then
-- Tool call argument fragment; Go-side rewrites index.
return json.encode({
content = "", done = false,
tool_calls = { {
index = chunk.index or 0,
id = "",
type = "function",
["function"] = { name = "", arguments = d.partial_json or "" },
} },
})
end
if chunk.delta.type == "thinking_delta" and chunk.delta.thinking then
return json.encode({ content = "", done = false, reasoning_content = chunk.delta.thinking })
if d.type == "thinking_delta" and d.thinking then
return json.encode({ content = "", done = false, reasoning_content = d.thinking })
end
if d.type == "text_delta" and d.text then
return json.encode({ content = d.text, done = false })
end
return json.encode({ content = chunk.delta.text or "", done = false })
end
-- ── content_block_stop / message_stop ─────────────────────
if chunk.type == "message_stop" then
return json.encode({ content = "", done = true })
end
if chunk.type == "content_block_stop" then
return json.encode({ content = "", done = false })
-- The message_delta event already emitted the true finish_reason.
-- Do NOT emit done=true here: an empty finish_reason would
-- overwrite the real one (tool_calls) in the Go gateway's
-- lastFinish tracker, causing the final SSE chunk to say
-- finish_reason=stop instead of tool_calls.
return ""
end
return ""
end
-- 错误收敛Anthropic 信封 {type:"error", error:{type, message}}
-- ============================================================
-- transform_error (Anthropic error → human-readable string)
-- ============================================================
function adapter.transform_error(status, body)
local ok, resp = pcall(json.decode, body)
if not ok or type(resp) ~= "table" then return nil end
-- Anthropic envelope: {type:"error", error:{type, message}}
if resp.type == "error" and type(resp.error) == "table"
and type(resp.error.message) == "string" then
return resp.error.message
end
-- Flat envelope: {error: {message: "..."}}
if type(resp.error) == "table" and type(resp.error.message) == "string" then
return resp.error.message
end
return nil
end

View File

@ -68,6 +68,15 @@ function adapter.transform_response(raw_body)
unified.token_usage.prompt = resp.usageMetadata.promptTokenCount or 0
unified.token_usage.completion = resp.usageMetadata.candidatesTokenCount or 0
unified.token_usage.total = resp.usageMetadata.totalTokenCount or 0
-- Gemini reports context-cache reads as cachedContentTokenCount;
-- normalize into OpenAI-standard prompt_tokens_details.cached_tokens
-- so clients and the audit trail see the hit count. Emitted even when
-- 0 so a reported miss stays distinguishable from "not reported".
if resp.usageMetadata.cachedContentTokenCount ~= nil then
unified.token_usage.prompt_tokens_details = {
cached_tokens = resp.usageMetadata.cachedContentTokenCount
}
end
end
if resp.candidates and #resp.candidates > 0 then
@ -100,9 +109,12 @@ function adapter.transform_stream_chunk(raw_chunk)
local t = chunk.usageMetadata.totalTokenCount or 0
if p > 0 or c > 0 or t > 0 then
uses = { prompt = p, completion = c, total = t }
local cacheRead = chunk.usageMetadata.cachedContentTokenCount or 0
if cacheRead > 0 then
uses.prompt_tokens_details = { cached_tokens = cacheRead }
-- Emit details whenever the field is present, even at 0, so a
-- reported cache miss stays distinguishable from "not reported".
if chunk.usageMetadata.cachedContentTokenCount ~= nil then
uses.prompt_tokens_details = {
cached_tokens = chunk.usageMetadata.cachedContentTokenCount
}
end
end
end

View File

@ -136,13 +136,43 @@ end
-- 错误收敛:标准 OpenAI 信封 {error:{message,...}}
function adapter.transform_error(status, body)
-- Non-JSON body (nginx HTML error pages, plain text): extract a short
-- human-readable reason instead of letting the raw body reach the log.
local ok, resp = pcall(json.decode, body)
if not ok or type(resp) ~= "table" then return nil end
if not ok or type(resp) ~= "table" then
-- HTML error page: pull the <title> text (e.g. "413 Request Entity Too Large")
local title = string.match(body or "", "<title>(.-)</title>")
if title and title ~= "" then return title end
-- Bare text: first non-empty line, capped
local line = string.match(body or "", "^%s*([^\r\n]+)")
if line and line ~= "" and not string.match(line, "^<") then
return string.sub(line, 1, 200)
end
return nil
end
-- Standard OpenAI envelope: {error:{message,...}} or {error:"..."}
local e = resp.error
if type(e) == "table" and type(e.message) == "string" then
return e.message
end
if type(e) == "string" then return e end
-- Flat envelope used by many OpenAI-compatible gateways:
-- {"code":20012,"message":"Model does not exist..."}
-- {"code":"INVALID_API_KEY","message":"Invalid API key"}
if type(resp.message) == "string" and resp.message ~= "" then
if resp.code ~= nil then
return tostring(resp.code) .. ": " .. resp.message
end
return resp.message
end
-- Some gateways use {detail:"..."} (FastAPI style)
if type(resp.detail) == "string" and resp.detail ~= "" then
return resp.detail
end
return nil
end