mirror of
https://gitcode.com/JianFeeeee/ModelRouter.git
synced 2026-09-20 00:48:00 +00:00
anthropic.lua v3.0.0: - Issue 1: tool_result/tool_use round-trip - Issue 3: thinking default OFF (opt-in via extra_body.thinking) - Issue 4: tool_choice mapping - Issue 5: collect_blocks preserves unknown part types - message_stop no longer emits done=true (was overwriting tool_calls finish_reason) - cache_read_input_tokens normalized even at 0 gemini.lua: - transform_response was missing cachedContentTokenCount openai.lua (Issue 6): - transform_error handles flat envelopes, nginx HTML, bare text chat.go mergeUsage: - Keep PromptTokensDetails even when CachedTokens=0 scheduler.go: - Remove sort.SliceStable by Pref; round-robin cursor is the only LB mechanism provider.go ModelAvailable: - Also check Pref() > prefMin, persistently failing slots exit cands presets.go: - 17 built-in source templates Tests: 6 new test functions, 2 updated for new semantics
180 lines
6.9 KiB
Lua
180 lines
6.9 KiB
Lua
local adapter = {}
|
|
|
|
adapter.name = "openai"
|
|
adapter.version = "2.0.0"
|
|
adapter.endpoint = "/chat/completions"
|
|
adapter.headers = {}
|
|
|
|
-- OpenAI /chat/completions format (pass-through, strip provider-specific fields)
|
|
function adapter.transform_request(raw_body)
|
|
local ok, req = pcall(json.decode, raw_body)
|
|
if not ok then return raw_body end
|
|
req.disable_thinking = nil
|
|
req.extra_body = nil
|
|
if req.messages then
|
|
for _, msg in ipairs(req.messages) do
|
|
msg.reasoning_content = nil
|
|
end
|
|
end
|
|
return json.encode(req)
|
|
end
|
|
|
|
function adapter.transform_response(raw_body)
|
|
local ok, resp = pcall(json.decode, raw_body)
|
|
if not ok or resp == nil then return raw_body end
|
|
|
|
local unified = {
|
|
content = "",
|
|
finish_reason = "",
|
|
token_usage = { prompt = 0, completion = 0, total = 0 }
|
|
}
|
|
|
|
if type(resp.usage) == "table" then
|
|
unified.token_usage.prompt = resp.usage.prompt_tokens or 0
|
|
unified.token_usage.completion = resp.usage.completion_tokens or 0
|
|
unified.token_usage.total = resp.usage.total_tokens or 0
|
|
-- Cache passthrough: OpenAI v2 prompt_tokens_details.cached_tokens
|
|
-- and DeepSeek-legacy prompt_cache_hit/miss_tokens. dsh reads
|
|
-- prompt_tokens_details.cached_tokens (falls back to the legacy
|
|
-- standalone field), so both shapes reach clients.
|
|
local hit = 0
|
|
if type(resp.usage.prompt_tokens_details) == "table" and resp.usage.prompt_tokens_details.cached_tokens ~= nil then
|
|
hit = resp.usage.prompt_tokens_details.cached_tokens
|
|
unified.token_usage.prompt_tokens_details = { cached_tokens = hit }
|
|
end
|
|
if (resp.usage.prompt_cache_hit_tokens or 0) > 0 then
|
|
unified.token_usage.prompt_cache_hit_tokens = resp.usage.prompt_cache_hit_tokens
|
|
unified.token_usage.prompt_cache_miss_tokens = resp.usage.prompt_cache_miss_tokens or 0
|
|
if hit == 0 then
|
|
unified.token_usage.prompt_tokens_details = { cached_tokens = resp.usage.prompt_cache_hit_tokens }
|
|
end
|
|
end
|
|
end
|
|
|
|
if type(resp.choices) == "table" and #resp.choices > 0 then
|
|
local ch = resp.choices[1]
|
|
if type(ch.message) == "table" then
|
|
unified.content = ch.message.content or ""
|
|
if ch.message.reasoning_content then
|
|
unified.reasoning_content = ch.message.reasoning_content
|
|
end
|
|
if type(ch.message.tool_calls) == "table" then
|
|
local tcs = {}
|
|
for _, tc in ipairs(ch.message.tool_calls) do
|
|
local args_ok, args = pcall(json.decode, tc["function"].arguments)
|
|
if not args_ok then args = {} end
|
|
table.insert(tcs, {
|
|
id = tc.id,
|
|
type = tc.type or "function",
|
|
name = tc["function"].name,
|
|
arguments = args
|
|
})
|
|
end
|
|
unified.tool_calls = tcs
|
|
end
|
|
end
|
|
unified.finish_reason = ch.finish_reason or ""
|
|
end
|
|
|
|
return json.encode(unified)
|
|
end
|
|
|
|
function adapter.transform_stream_chunk(raw_chunk)
|
|
local ok, chunk = pcall(json.decode, raw_chunk)
|
|
if not ok then return "" end
|
|
|
|
-- OpenAI-style streams may attach usage to a chunk with empty choices
|
|
-- (the final usage chunk). Keys must match Go's TokenUsage json tags
|
|
-- (prompt/completion/total); the gateway re-emits standard *_tokens.
|
|
local uses = nil
|
|
if type(chunk.usage) == "table" then
|
|
uses = {
|
|
prompt = chunk.usage.prompt_tokens or chunk.usage.prompt or 0,
|
|
completion = chunk.usage.completion_tokens or chunk.usage.completion or 0,
|
|
total = chunk.usage.total_tokens or chunk.usage.total or 0,
|
|
}
|
|
if type(chunk.usage.prompt_tokens_details) == "table" and chunk.usage.prompt_tokens_details.cached_tokens ~= nil then
|
|
uses.prompt_tokens_details = { cached_tokens = chunk.usage.prompt_tokens_details.cached_tokens }
|
|
elseif (chunk.usage.prompt_cache_hit_tokens or 0) > 0 then
|
|
uses.prompt_cache_hit_tokens = chunk.usage.prompt_cache_hit_tokens
|
|
uses.prompt_cache_miss_tokens = chunk.usage.prompt_cache_miss_tokens or 0
|
|
uses.prompt_tokens_details = { cached_tokens = chunk.usage.prompt_cache_hit_tokens }
|
|
end
|
|
end
|
|
|
|
if not chunk.choices or #chunk.choices == 0 then
|
|
if uses ~= nil then
|
|
-- usage-only chunk is not a content/finish signal; the gateway
|
|
-- emits its own terminal stop chunk and merges this usage.
|
|
return json.encode({ usage = uses, done = false })
|
|
end
|
|
return ""
|
|
end
|
|
local delta = chunk.choices[1].delta or {}
|
|
local fr = chunk.choices[1].finish_reason
|
|
|
|
local finish = (type(fr) == "string" and fr ~= "") and fr or nil
|
|
|
|
local unified = {
|
|
content = delta.content or "",
|
|
done = (finish ~= nil)
|
|
}
|
|
if finish then
|
|
unified.finish_reason = finish
|
|
end
|
|
if uses ~= nil then
|
|
unified.usage = uses
|
|
end
|
|
if delta.reasoning_content then
|
|
unified.reasoning_content = delta.reasoning_content
|
|
end
|
|
if delta.tool_calls then
|
|
unified.tool_calls = delta.tool_calls
|
|
end
|
|
return json.encode(unified)
|
|
end
|
|
|
|
-- 错误收敛:标准 OpenAI 信封 {error:{message,...}}
|
|
function adapter.transform_error(status, body)
|
|
-- Non-JSON body (nginx HTML error pages, plain text): extract a short
|
|
-- human-readable reason instead of letting the raw body reach the log.
|
|
local ok, resp = pcall(json.decode, body)
|
|
if not ok or type(resp) ~= "table" then
|
|
-- HTML error page: pull the <title> text (e.g. "413 Request Entity Too Large")
|
|
local title = string.match(body or "", "<title>(.-)</title>")
|
|
if title and title ~= "" then return title end
|
|
-- Bare text: first non-empty line, capped
|
|
local line = string.match(body or "", "^%s*([^\r\n]+)")
|
|
if line and line ~= "" and not string.match(line, "^<") then
|
|
return string.sub(line, 1, 200)
|
|
end
|
|
return nil
|
|
end
|
|
|
|
-- Standard OpenAI envelope: {error:{message,...}} or {error:"..."}
|
|
local e = resp.error
|
|
if type(e) == "table" and type(e.message) == "string" then
|
|
return e.message
|
|
end
|
|
if type(e) == "string" then return e end
|
|
|
|
-- Flat envelope used by many OpenAI-compatible gateways:
|
|
-- {"code":20012,"message":"Model does not exist..."}
|
|
-- {"code":"INVALID_API_KEY","message":"Invalid API key"}
|
|
if type(resp.message) == "string" and resp.message ~= "" then
|
|
if resp.code ~= nil then
|
|
return tostring(resp.code) .. ": " .. resp.message
|
|
end
|
|
return resp.message
|
|
end
|
|
|
|
-- Some gateways use {detail:"..."} (FastAPI style)
|
|
if type(resp.detail) == "string" and resp.detail ~= "" then
|
|
return resp.detail
|
|
end
|
|
|
|
return nil
|
|
end
|
|
|
|
return adapter
|