feat: ModelRouter — unified OpenAI-compatible multi-source LLM gateway

- Lua adapters per upstream (transform_request/response/stream_chunk, build_headers signing hooks)
- AUTO priority routing with per-model kind (chat/image), explicit source/model routing
- Per-source concurrency caps with queueing, exponential backoff, AUTO failover
- OpenAI-compatible API: chat completions, SSE streaming, image generations, models
- Gateway key auth, web UI for adapter/source management, runtime persistence
- e2e test running the real binary against mocked upstreams
This commit is contained in:
root
2026-08-05 15:24:51 +08:00
parent 8631b08253
commit f7f76e097d
32 changed files with 4382 additions and 2 deletions

View File

@ -0,0 +1,86 @@
local adapter = {}
adapter.name = "anthropic"
adapter.version = "2.0.0"
adapter.endpoint = "/v1/messages"
adapter.headers = {
["anthropic-version"] = "2023-06-01"
}
function adapter.transform_request(raw_body)
local ok, req = pcall(json.decode, raw_body)
if not ok then return raw_body end
local msgs = {}
local system = ""
for _, m in ipairs(req.messages or {}) do
if m.role == "system" then
system = system .. m.content .. "\n"
else
table.insert(msgs, { role = m.role, content = m.content })
end
end
local anthropic_req = {
model = req.model or "claude-sonnet-4-20250514",
max_tokens = req.max_tokens or 4096,
messages = msgs,
stream = req.stream or false,
}
if not req.disable_thinking then
anthropic_req.thinking = { type = "enabled", budget_tokens = 4096 }
end
if system ~= "" then
anthropic_req.system = system
end
return json.encode(anthropic_req)
end
function adapter.transform_response(raw_body)
local ok, resp = pcall(json.decode, raw_body)
if not ok then return raw_body end
local unified = {
content = "",
finish_reason = "",
token_usage = { prompt = 0, completion = 0, total = 0 }
}
if resp.usage then
unified.token_usage.prompt = resp.usage.input_tokens or 0
unified.token_usage.completion = resp.usage.output_tokens or 0
unified.token_usage.total = (resp.usage.input_tokens or 0) + (resp.usage.output_tokens or 0)
end
if resp.content and #resp.content > 0 then
for _, block in ipairs(resp.content) do
if block.type == "text" then
unified.content = unified.content .. (block.text or "")
end
end
end
unified.finish_reason = resp.stop_reason or ""
return json.encode(unified)
end
function adapter.transform_stream_chunk(raw_chunk)
local ok, chunk = pcall(json.decode, raw_chunk)
if not ok then return "" end
if chunk.type == "message_start" then return "" end
if chunk.type == "message_delta" then
return json.encode({ content = "", done = (chunk.delta and chunk.delta.stop_reason ~= nil) })
end
if chunk.type == "content_block_delta" and chunk.delta then
return json.encode({ content = chunk.delta.text or "", done = false })
end
if chunk.type == "message_stop" then
return json.encode({ content = "", done = true })
end
return ""
end
return adapter

View File

@ -0,0 +1,78 @@
local adapter = {}
adapter.name = "deepseek"
adapter.version = "2.1.0"
adapter.endpoint = "/chat/completions"
adapter.headers = {}
function adapter.transform_request(raw_body)
local ok, req = pcall(json.decode, raw_body)
if not ok then return raw_body end
req.model = req.model or "deepseek-chat"
req.stream = req.stream or false
if req.disable_thinking then
req.extra_body = req.extra_body or {}
req.extra_body.thinking = { type = "disabled" }
end
req.disable_thinking = nil
return json.encode(req)
end
function adapter.transform_response(raw_body)
local ok, resp = pcall(json.decode, raw_body)
if not ok or resp == nil then return raw_body end
local unified = {
content = "",
finish_reason = "",
token_usage = { prompt = 0, completion = 0, total = 0 }
}
if type(resp.usage) == "table" then
unified.token_usage.prompt = resp.usage.prompt_tokens or 0
unified.token_usage.completion = resp.usage.completion_tokens or 0
unified.token_usage.total = resp.usage.total_tokens or 0
end
if type(resp.choices) == "table" and #resp.choices > 0 then
local ch = resp.choices[1]
if type(ch.message) == "table" then
unified.content = ch.message.content or ""
if ch.message.reasoning_content then
unified.reasoning_content = ch.message.reasoning_content
end
if type(ch.message.tool_calls) == "table" then
local tcs = {}
for _, tc in ipairs(ch.message.tool_calls) do
local args_ok, args = pcall(json.decode, tc["function"].arguments)
if not args_ok then args = {} end
table.insert(tcs, {
id = tc.id,
type = tc.type or "function",
name = tc["function"].name,
arguments = args
})
end
unified.tool_calls = tcs
end
end
unified.finish_reason = ch.finish_reason or ""
end
return json.encode(unified)
end
function adapter.transform_stream_chunk(raw_chunk)
local ok, chunk = pcall(json.decode, raw_chunk)
if not ok then return "" end
if not chunk.choices or #chunk.choices == 0 then return "" end
local delta = chunk.choices[1].delta or {}
local fr = chunk.choices[1].finish_reason
return json.encode({
content = delta.content or "",
done = (fr ~= nil)
})
end
return adapter

View File

@ -0,0 +1,89 @@
local adapter = {}
adapter.name = "gemini"
adapter.version = "2.0.0"
adapter.endpoint = "/v1/models"
adapter.headers = {}
-- Gemini API: POST /v1/models/{model}:generateContent
-- Auth: API key in query param ?key=XXX or Authorization: Bearer XXX
function adapter.transform_request(raw_body)
local ok, req = pcall(json.decode, raw_body)
if not ok then return raw_body end
local contents = {}
for _, m in ipairs(req.messages or {}) do
table.insert(contents, {
role = (m.role == "assistant") and "model" or m.role,
parts = { { text = m.content } }
})
end
local gemini_req = {
contents = contents,
generationConfig = {
temperature = req.temperature or 0.7,
maxOutputTokens = req.max_tokens or 4096,
}
}
if req.stream then
gemini_req.stream = true
end
return json.encode(gemini_req)
end
-- Gemini 的 endpoint 动态拼接:/v1/models/{model}:generateContent
function adapter.transform_response(raw_body)
local ok, resp = pcall(json.decode, raw_body)
if not ok then return raw_body end
local unified = {
content = "",
finish_reason = "",
token_usage = { prompt = 0, completion = 0, total = 0 }
}
if resp.usageMetadata then
unified.token_usage.prompt = resp.usageMetadata.promptTokenCount or 0
unified.token_usage.completion = resp.usageMetadata.candidatesTokenCount or 0
unified.token_usage.total = resp.usageMetadata.totalTokenCount or 0
end
if resp.candidates and #resp.candidates > 0 then
local cand = resp.candidates[1]
if cand.content and cand.content.parts then
for _, part in ipairs(cand.content.parts) do
if part.text then
unified.content = unified.content .. part.text
end
end
end
if cand.finishReason then
unified.finish_reason = cand.finishReason
end
end
return json.encode(unified)
end
function adapter.transform_stream_chunk(raw_chunk)
local ok, chunk = pcall(json.decode, raw_chunk)
if not ok then return "" end
if not chunk.candidates or #chunk.candidates == 0 then return "" end
local cand = chunk.candidates[1]
local content = ""
if cand.content and cand.content.parts then
for _, part in ipairs(cand.content.parts) do
content = content .. (part.text or "")
end
end
return json.encode({
content = content,
done = (cand.finishReason ~= nil)
})
end
return adapter

View File

@ -0,0 +1,75 @@
local adapter = {}
adapter.name = "github"
adapter.version = "2.0.0"
adapter.endpoint = "/chat/completions"
adapter.headers = {}
-- GitHub Models: Azure-like endpoint, auth via Bearer token (PAT)
-- BaseURL example: https://models.inference.ai.azure.com
function adapter.transform_request(raw_body)
local ok, req = pcall(json.decode, raw_body)
if not ok then return raw_body end
req.model = req.model or "gpt-4o"
req.temperature = req.temperature or 0.7
req.max_tokens = req.max_tokens or 4096
req.stream = req.stream or false
req.disable_thinking = nil
req.extra_body = nil
return json.encode(req)
end
function adapter.transform_response(raw_body)
local ok, resp = pcall(json.decode, raw_body)
if not ok or resp == nil then return raw_body end
local unified = {
content = "",
finish_reason = "",
token_usage = { prompt = 0, completion = 0, total = 0 }
}
if type(resp.usage) == "table" then
unified.token_usage.prompt = resp.usage.prompt_tokens or 0
unified.token_usage.completion = resp.usage.completion_tokens or 0
unified.token_usage.total = resp.usage.total_tokens or 0
end
if type(resp.choices) == "table" and #resp.choices > 0 then
local ch = resp.choices[1]
if type(ch.message) == "table" then
unified.content = ch.message.content or ""
if type(ch.message.tool_calls) == "table" then
local tcs = {}
for _, tc in ipairs(ch.message.tool_calls) do
local args_ok, args = pcall(json.decode, tc["function"].arguments)
if not args_ok then args = {} end
table.insert(tcs, {
id = tc.id,
type = tc.type or "function",
name = tc["function"].name,
arguments = args
})
end
unified.tool_calls = tcs
end
end
unified.finish_reason = ch.finish_reason or ""
end
return json.encode(unified)
end
function adapter.transform_stream_chunk(raw_chunk)
local ok, chunk = pcall(json.decode, raw_chunk)
if not ok then return "" end
if not chunk.choices or #chunk.choices == 0 then return "" end
local delta = chunk.choices[1].delta or {}
local fr = chunk.choices[1].finish_reason
return json.encode({
content = delta.content or "",
done = (fr ~= nil)
})
end
return adapter

View File

@ -0,0 +1,74 @@
local adapter = {}
adapter.name = "groq"
adapter.version = "2.0.0"
adapter.endpoint = "/openai/v1/chat/completions"
adapter.headers = {}
-- Groq API is OpenAI-compatible
function adapter.transform_request(raw_body)
local ok, req = pcall(json.decode, raw_body)
if not ok then return raw_body end
req.model = req.model or "llama3-70b-8192"
req.temperature = req.temperature or 0.7
req.max_tokens = req.max_tokens or 4096
req.stream = req.stream or false
req.disable_thinking = nil
req.extra_body = nil
return json.encode(req)
end
function adapter.transform_response(raw_body)
local ok, resp = pcall(json.decode, raw_body)
if not ok or resp == nil then return raw_body end
local unified = {
content = "",
finish_reason = "",
token_usage = { prompt = 0, completion = 0, total = 0 }
}
if type(resp.usage) == "table" then
unified.token_usage.prompt = resp.usage.prompt_tokens or 0
unified.token_usage.completion = resp.usage.completion_tokens or 0
unified.token_usage.total = resp.usage.total_tokens or 0
end
if type(resp.choices) == "table" and #resp.choices > 0 then
local ch = resp.choices[1]
if type(ch.message) == "table" then
unified.content = ch.message.content or ""
if type(ch.message.tool_calls) == "table" then
local tcs = {}
for _, tc in ipairs(ch.message.tool_calls) do
local args_ok, args = pcall(json.decode, tc["function"].arguments)
if not args_ok then args = {} end
table.insert(tcs, {
id = tc.id,
type = tc.type or "function",
name = tc["function"].name,
arguments = args
})
end
unified.tool_calls = tcs
end
end
unified.finish_reason = ch.finish_reason or ""
end
return json.encode(unified)
end
function adapter.transform_stream_chunk(raw_chunk)
local ok, chunk = pcall(json.decode, raw_chunk)
if not ok then return "" end
if not chunk.choices or #chunk.choices == 0 then return "" end
local delta = chunk.choices[1].delta or {}
local fr = chunk.choices[1].finish_reason
return json.encode({
content = delta.content or "",
done = (fr ~= nil)
})
end
return adapter

View File

@ -0,0 +1,107 @@
local adapter = {}
adapter.name = "kimicode"
adapter.version = "1.0.0"
adapter.endpoint = "/v1/chat/completions"
adapter.headers = {}
-- KimiCode / Kimi K2 属于 OpenAI 兼容协议;但部分云端 API 会校验调用方
-- "app"(只放行特定 agent要求每次请求带上按 secret 计算的应用签名。
-- 这里演示 build_headers 钩子:基于 timestamp + 请求体哈希生成签名头。
--
-- 配置要求source.meta:
-- meta:
-- app_id: <申请到的 app id>
-- app_key: <你的 key由网关的 base_url 复用 api_key 亦可)>
-- app_secret: <签名密钥>
-- app_agent: code-agent # 若云端要求声明 agent 身份
function adapter.transform_request(raw_body)
local ok, req = pcall(json.decode, raw_body)
if not ok then return raw_body end
req.model = req.model or "kimi-k2"
req.disable_thinking = nil
req.extra_body = nil
return json.encode(req)
end
-- 可选的动态签名钩子。meta 由 Go 注入:
-- meta.url / meta.method / meta.body / meta.api_key / meta.timestamp / meta.source.meta
function adapter.build_headers(meta)
local h = {
["Content-Type"] = "application/json",
["X-App-Id"] = tostring((meta.source.meta or {}).app_id or ""),
["X-Timestamp"] = tostring(meta.timestamp),
}
local agent = (meta.source.meta or {}).app_agent
if agent and agent ~= "" then
h["X-Agent"] = agent
end
-- 校验 app通常要求 Authorization 用 app secret 派生签名
local secret = (meta.source.meta or {}).app_secret
local api_key = meta.source.meta and meta.source.meta.api_key or meta.api_key
if secret and secret ~= "" then
local body_hash = sha256_hex(meta.body)
local sign_string = tostring(meta.timestamp) .. meta.method .. meta.url .. body_hash
local sign = hmac_sha256_hex(secret, sign_string)
h["Authorization"] = "Bearer " .. api_key
h["X-App-Sign"] = sign
else
h["Authorization"] = "Bearer " .. api_key
end
return h
end
function adapter.transform_response(raw_body)
local ok, resp = pcall(json.decode, raw_body)
if not ok or resp == nil then return raw_body end
local unified = {
content = "",
finish_reason = "",
token_usage = { prompt = 0, completion = 0, total = 0 }
}
if type(resp.usage) == "table" then
unified.token_usage.prompt = resp.usage.prompt_tokens or 0
unified.token_usage.completion = resp.usage.completion_tokens or 0
unified.token_usage.total = resp.usage.total_tokens or 0
end
if type(resp.choices) == "table" and #resp.choices > 0 then
local ch = resp.choices[1]
if type(ch.message) == "table" then
unified.content = ch.message.content or ""
if ch.message.reasoning_content then
unified.reasoning_content = ch.message.reasoning_content
end
if type(ch.message.tool_calls) == "table" then
local tcs = {}
for _, tc in ipairs(ch.message.tool_calls) do
local args_ok, args = pcall(json.decode, tc["function"].arguments)
if not args_ok then args = {} end
table.insert(tcs, {
id = tc.id,
type = tc.type or "function",
name = tc["function"].name,
arguments = args
})
end
unified.tool_calls = tcs
end
end
unified.finish_reason = ch.finish_reason or ""
end
return json.encode(unified)
end
function adapter.transform_stream_chunk(raw_chunk)
local ok, chunk = pcall(json.decode, raw_chunk)
if not ok then return "" end
if not chunk.choices or #chunk.choices == 0 then return "" end
local delta = chunk.choices[1].delta or {}
local fr = chunk.choices[1].finish_reason
return json.encode({
content = delta.content or "",
done = (fr ~= nil)
})
end
return adapter

View File

@ -0,0 +1,74 @@
local adapter = {}
adapter.name = "mistral"
adapter.version = "2.0.0"
adapter.endpoint = "/v1/chat/completions"
adapter.headers = {}
-- Mistral API is OpenAI-compatible, just passes through
function adapter.transform_request(raw_body)
local ok, req = pcall(json.decode, raw_body)
if not ok then return raw_body end
req.model = req.model or "mistral-large-latest"
req.temperature = req.temperature or 0.7
req.max_tokens = req.max_tokens or 4096
req.stream = req.stream or false
req.disable_thinking = nil
req.extra_body = nil
return json.encode(req)
end
function adapter.transform_response(raw_body)
local ok, resp = pcall(json.decode, raw_body)
if not ok or resp == nil then return raw_body end
local unified = {
content = "",
finish_reason = "",
token_usage = { prompt = 0, completion = 0, total = 0 }
}
if type(resp.usage) == "table" then
unified.token_usage.prompt = resp.usage.prompt_tokens or 0
unified.token_usage.completion = resp.usage.completion_tokens or 0
unified.token_usage.total = resp.usage.total_tokens or 0
end
if type(resp.choices) == "table" and #resp.choices > 0 then
local ch = resp.choices[1]
if type(ch.message) == "table" then
unified.content = ch.message.content or ""
if type(ch.message.tool_calls) == "table" then
local tcs = {}
for _, tc in ipairs(ch.message.tool_calls) do
local args_ok, args = pcall(json.decode, tc["function"].arguments)
if not args_ok then args = {} end
table.insert(tcs, {
id = tc.id,
type = tc.type or "function",
name = tc["function"].name,
arguments = args
})
end
unified.tool_calls = tcs
end
end
unified.finish_reason = ch.finish_reason or ""
end
return json.encode(unified)
end
function adapter.transform_stream_chunk(raw_chunk)
local ok, chunk = pcall(json.decode, raw_chunk)
if not ok then return "" end
if not chunk.choices or #chunk.choices == 0 then return "" end
local delta = chunk.choices[1].delta or {}
local fr = chunk.choices[1].finish_reason
return json.encode({
content = delta.content or "",
done = (fr ~= nil)
})
end
return adapter

View File

@ -0,0 +1,63 @@
local adapter = {}
adapter.name = "ollama"
adapter.version = "2.0.0"
adapter.endpoint = "/api/chat"
adapter.headers = {}
-- Ollama API 格式:{ model, messages, stream, options:{temperature,num_predict} }
function adapter.transform_request(raw_body)
local ok, req = pcall(json.decode, raw_body)
if not ok then return raw_body end
local ollama_req = {
model = req.model or "llama3",
stream = req.stream or false,
options = {
temperature = req.temperature or 0.7,
num_predict = req.max_tokens or 2048
}
}
-- 转换 messages 格式Ollama 兼容 OpenAI 的 messages 格式)
if req.messages then
local msgs = {}
for _, m in ipairs(req.messages) do
table.insert(msgs, { role = m.role, content = m.content })
end
ollama_req.messages = msgs
end
return json.encode(ollama_req)
end
function adapter.transform_response(raw_body)
local ok, resp = pcall(json.decode, raw_body)
if not ok then return raw_body end
local unified = {
content = "",
finish_reason = resp.done_reason or "",
tool_calls = {},
usage = { prompt = 0, completion = 0, total = 0 }
}
if resp.message then
unified.content = resp.message.content or ""
end
return json.encode(unified)
end
function adapter.transform_stream_chunk(raw_chunk)
local ok, chunk = pcall(json.decode, raw_chunk)
if not ok then return "" end
if not chunk.message then return "" end
return json.encode({
content = chunk.message.content or "",
done = chunk.done or false
})
end
return adapter

View File

@ -0,0 +1,80 @@
local adapter = {}
adapter.name = "openai"
adapter.version = "2.0.0"
adapter.endpoint = "/chat/completions"
adapter.headers = {}
-- OpenAI /chat/completions format (pass-through, strip provider-specific fields)
function adapter.transform_request(raw_body)
local ok, req = pcall(json.decode, raw_body)
if not ok then return raw_body end
req.disable_thinking = nil
req.extra_body = nil
if req.messages then
for _, msg in ipairs(req.messages) do
msg.reasoning_content = nil
end
end
return json.encode(req)
end
function adapter.transform_response(raw_body)
local ok, resp = pcall(json.decode, raw_body)
if not ok or resp == nil then return raw_body end
local unified = {
content = "",
finish_reason = "",
token_usage = { prompt = 0, completion = 0, total = 0 }
}
if type(resp.usage) == "table" then
unified.token_usage.prompt = resp.usage.prompt_tokens or 0
unified.token_usage.completion = resp.usage.completion_tokens or 0
unified.token_usage.total = resp.usage.total_tokens or 0
end
if type(resp.choices) == "table" and #resp.choices > 0 then
local ch = resp.choices[1]
if type(ch.message) == "table" then
unified.content = ch.message.content or ""
if ch.message.reasoning_content then
unified.reasoning_content = ch.message.reasoning_content
end
if type(ch.message.tool_calls) == "table" then
local tcs = {}
for _, tc in ipairs(ch.message.tool_calls) do
local args_ok, args = pcall(json.decode, tc["function"].arguments)
if not args_ok then args = {} end
table.insert(tcs, {
id = tc.id,
type = tc.type or "function",
name = tc["function"].name,
arguments = args
})
end
unified.tool_calls = tcs
end
end
unified.finish_reason = ch.finish_reason or ""
end
return json.encode(unified)
end
function adapter.transform_stream_chunk(raw_chunk)
local ok, chunk = pcall(json.decode, raw_chunk)
if not ok then return "" end
if not chunk.choices or #chunk.choices == 0 then return "" end
local delta = chunk.choices[1].delta or {}
local fr = chunk.choices[1].finish_reason
return json.encode({
content = delta.content or "",
done = (fr ~= nil)
})
end
return adapter