mirror of
https://gitcode.com/JianFeeeee/ModelRouter.git
synced 2026-09-20 08:57:57 +00:00
Three related forwarding defects found by auditing every adapter with a
tool-calling replay (assistant turn with content:[] + tool_calls).
1) content:[] -> content:{} (all 12 openai-adapter sources, plus
deepseek/trae/sensenova/agentrouter/github/groq/kimicode/mistral)
Lua adapters json.decode the request and re-encode it, and an empty Lua
table is indistinguishable from an empty JSON array — the encoder emits
{} for both. Agent clients serialise a tool-calling assistant turn with
no text as content:[], so every pass-through adapter rewrote it to
content:{} — not valid OpenAI (content is string|array|null). Verified
against a live upstream: content:[] produced "400 invalid arguments"
while content:"" was accepted.
Fixed once at the decode boundary (types.ChatMessage.UnmarshalJSON):
empty-array content normalises to "" and an empty tool_calls array is
dropped, so every adapter — including future ones — sees a valid shape.
2) gemini dropped tool_calls and never emitted functionCall /
functionResponse; the tool role also stayed as an invalid role inside
contents and system was not moved to systemInstruction.
3) ollama copied only role/content, dropping tool_calls and the call
attribution entirely (it needs tool_name, not tool_call_id).
Test: TestAdaptersPreserveToolCalls asserts, for every adapter, that the
call id (or function name where the wire format has no id), the function
name, the tool result and the trailing user turn all survive, plus a
negative control for plain text.
242 lines
9.5 KiB
Go
242 lines
9.5 KiB
Go
// Package types defines the unified (OpenAI-compatible) wire format that the
|
|
// gateway exposes to its clients, plus the unified internal representation.
|
|
package types
|
|
|
|
import (
|
|
"bytes"
|
|
"encoding/json"
|
|
"errors"
|
|
"strings"
|
|
"time"
|
|
)
|
|
|
|
// ErrBusy is the soft "source at capacity" sentinel shared by the provider
|
|
// layer (returns it) and the scheduler layer (reacts to it): busy is not a
|
|
// failure, so no cooldown/preference penalty is recorded, and gateways map
|
|
// it to HTTP 429. Defined here so the scheduler does not depend on the
|
|
// provider package (which pulls in the Lua runtime).
|
|
var ErrBusy = errors.New("provider busy")
|
|
|
|
// ---- OpenAI wire request (gateway input) ----
|
|
|
|
type ChatRequest struct {
|
|
Model string `json:"model"`
|
|
Messages []ChatMessage `json:"messages"`
|
|
Temperature *float64 `json:"temperature,omitempty"`
|
|
MaxTokens int `json:"max_tokens,omitempty"`
|
|
Stream bool `json:"stream,omitempty"`
|
|
Tools []interface{} `json:"tools,omitempty"`
|
|
ToolChoice interface{} `json:"tool_choice,omitempty"`
|
|
DisableThinking bool `json:"disable_thinking"`
|
|
ExtraBody map[string]interface{} `json:"-"`
|
|
}
|
|
|
|
func (r *ChatRequest) MarshalJSON() ([]byte, error) {
|
|
type Alias ChatRequest
|
|
data, err := json.Marshal((*Alias)(r))
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
if len(r.ExtraBody) == 0 {
|
|
return data, nil
|
|
}
|
|
var raw map[string]interface{}
|
|
if err := json.Unmarshal(data, &raw); err != nil {
|
|
return nil, err
|
|
}
|
|
for k, v := range r.ExtraBody {
|
|
raw[k] = v
|
|
}
|
|
return json.Marshal(raw)
|
|
}
|
|
|
|
// ChatMessage supports both plain string content and multimodal arrays
|
|
// (RawMessage preserves whatever the client sent for the adapter to process).
|
|
type ChatMessage struct {
|
|
Role string `json:"role"`
|
|
Content json.RawMessage `json:"content,omitempty"`
|
|
ReasoningContent string `json:"reasoning_content,omitempty"`
|
|
ToolCallID string `json:"tool_call_id,omitempty"`
|
|
ToolCalls json.RawMessage `json:"tool_calls,omitempty"`
|
|
}
|
|
|
|
func StringContent(s string) json.RawMessage { b, _ := json.Marshal(s); return b }
|
|
|
|
// isEmptyJSONArray reports whether raw is the literal empty JSON array [].
|
|
func isEmptyJSONArray(raw json.RawMessage) bool {
|
|
t := bytes.TrimSpace(raw)
|
|
return len(t) == 2 && t[0] == '[' && t[1] == ']'
|
|
}
|
|
|
|
// UnmarshalJSON decodes a chat message, normalising an empty-array content
|
|
// ("content":[]) to an empty string and dropping an empty tool_calls array.
|
|
//
|
|
// Why this exists: Lua adapters json.decode the request and re-encode it, and
|
|
// an empty Lua table is indistinguishable from an empty JSON array — the
|
|
// encoder emits {} for both. Agent clients serialise an assistant turn that
|
|
// carries tool_calls and no text as content:[], so every pass-through adapter
|
|
// turned it into content:{} — a shape that is not valid OpenAI (content is
|
|
// string | array of parts | null) and that real upstreams reject with
|
|
// "400 invalid arguments". Normalising at the decode boundary fixes every
|
|
// adapter at once, including ones added later.
|
|
func (m *ChatMessage) UnmarshalJSON(b []byte) error {
|
|
type alias ChatMessage
|
|
var a alias
|
|
if err := json.Unmarshal(b, &a); err != nil {
|
|
return err
|
|
}
|
|
*m = ChatMessage(a)
|
|
if isEmptyJSONArray(m.Content) {
|
|
m.Content = StringContent("")
|
|
}
|
|
if isEmptyJSONArray(m.ToolCalls) {
|
|
m.ToolCalls = nil
|
|
}
|
|
return nil
|
|
}
|
|
|
|
type ToolCall struct {
|
|
ID string `json:"id"`
|
|
Type string `json:"type"`
|
|
Name string `json:"name"`
|
|
Arguments map[string]interface{} `json:"arguments"`
|
|
}
|
|
|
|
// ---- Unified internal representation (what adapters produce) ----
|
|
|
|
type UnifiedResponse struct {
|
|
Model string `json:"model,omitempty"`
|
|
Content string `json:"content"`
|
|
ReasoningContent string `json:"reasoning_content,omitempty"`
|
|
FinishReason string `json:"finish_reason,omitempty"`
|
|
TokenUsage TokenUsage `json:"token_usage"`
|
|
ToolCalls []ToolCall `json:"tool_calls,omitempty"`
|
|
// ImageData used by image-generation adapters.
|
|
ImageData []ImageData `json:"image_data,omitempty"`
|
|
}
|
|
|
|
type TokenUsage struct {
|
|
Prompt int `json:"prompt"`
|
|
Completion int `json:"completion"`
|
|
Total int `json:"total"`
|
|
// PromptTokensDetails mirrors the OpenAI v2 usage.prompt_tokens_details
|
|
// object so cache-hit counts reported by OpenAI-compatible upstreams
|
|
// (and by adapters that normalize their own cache fields into it) pass
|
|
// through to clients that read it — dsh reads cached_tokens from here.
|
|
PromptTokensDetails *PromptTokensDetails `json:"prompt_tokens_details,omitempty"`
|
|
// PromptCacheHit / PromptCacheMiss carry the DeepSeek-legacy standalone
|
|
// fields; dsh falls back to prompt_cache_hit_tokens when
|
|
// prompt_tokens_details.cached_tokens is absent.
|
|
PromptCacheHit int `json:"prompt_cache_hit_tokens,omitempty"`
|
|
PromptCacheMiss int `json:"prompt_cache_miss_tokens,omitempty"`
|
|
}
|
|
|
|
// PromptTokensDetails is the OpenAI v2 prompt_tokens_details object. Only
|
|
// CachedTokens is emitted (omitempty drops the whole object when zero).
|
|
type PromptTokensDetails struct {
|
|
// CachedTokens is always emitted (even 0) so clients can distinguish
|
|
// "upstream reports cache, this request missed" from "no cache data".
|
|
CachedTokens int `json:"cached_tokens"`
|
|
}
|
|
|
|
// MarshalJSON emits both the legacy short keys (prompt/completion/total, used
|
|
// by the internal unified representation and older clients) and the OpenAI
|
|
// standard keys (prompt_tokens/completion_tokens/total_tokens). Standard
|
|
// clients such as DSH and DevEco Code read the *_tokens fields.
|
|
func (t TokenUsage) MarshalJSON() ([]byte, error) {
|
|
// Auto-generate prompt_tokens_details from legacy DeepSeek fields when
|
|
// the upstream adapter only set the standalone hit count (openai.lua
|
|
// does this in Lua, but other adapters or the standardSSEChunk fallback
|
|
// may not). dsh reads prompt_tokens_details.cached_tokens first.
|
|
pdetails := t.PromptTokensDetails
|
|
if pdetails == nil && t.PromptCacheHit > 0 {
|
|
pdetails = &PromptTokensDetails{CachedTokens: t.PromptCacheHit}
|
|
}
|
|
return json.Marshal(struct {
|
|
PromptTokens int `json:"prompt_tokens"`
|
|
CompletionTokens int `json:"completion_tokens"`
|
|
TotalTokens int `json:"total_tokens"`
|
|
Prompt int `json:"prompt"`
|
|
Completion int `json:"completion"`
|
|
Total int `json:"total"`
|
|
PromptTokensDetails *PromptTokensDetails `json:"prompt_tokens_details,omitempty"`
|
|
PromptCacheHitTokens int `json:"prompt_cache_hit_tokens,omitempty"`
|
|
PromptCacheMissTokens int `json:"prompt_cache_miss_tokens,omitempty"`
|
|
}{
|
|
PromptTokens: t.Prompt,
|
|
CompletionTokens: t.Completion,
|
|
TotalTokens: t.Total,
|
|
Prompt: t.Prompt,
|
|
Completion: t.Completion,
|
|
Total: t.Total,
|
|
PromptTokensDetails: pdetails,
|
|
PromptCacheHitTokens: t.PromptCacheHit,
|
|
PromptCacheMissTokens: t.PromptCacheMiss,
|
|
})
|
|
}
|
|
|
|
type ImageData struct {
|
|
B64JSON string `json:"b64_json,omitempty"`
|
|
URL string `json:"url,omitempty"`
|
|
Revised string `json:"revised_prompt,omitempty"`
|
|
}
|
|
|
|
// ---- Image generation (OpenAI /v1/images/generations wire) ----
|
|
|
|
type ImageGenRequest struct {
|
|
Model string `json:"model"`
|
|
Prompt string `json:"prompt"`
|
|
N int `json:"n,omitempty"`
|
|
Size string `json:"size,omitempty"`
|
|
ResponseFormat string `json:"response_format,omitempty"`
|
|
}
|
|
|
|
type ImageGenResponse struct {
|
|
Created int64 `json:"created"`
|
|
Data []ImageData `json:"data"`
|
|
}
|
|
|
|
// ---- Unified streaming chunk produced by adapters ----
|
|
|
|
// UnifiedChunk is one streamed delta. ToolCalls carries the raw upstream
|
|
// streaming tool_calls array (incremental fragments with an index field), which
|
|
// OpenAI-compatible clients accumulate themselves.
|
|
type UnifiedChunk struct {
|
|
Content string `json:"content"`
|
|
Done bool `json:"done"`
|
|
// FinishReason carries the upstream finish/stop reason ("tool_calls",
|
|
// "length", ...) when the adapter provides it; the gateway emits it on
|
|
// the terminating chunk instead of the default "stop".
|
|
FinishReason string `json:"finish_reason,omitempty"`
|
|
ToolCalls json.RawMessage `json:"tool_calls,omitempty"`
|
|
ReasoningContent string `json:"reasoning_content,omitempty"`
|
|
// Usage carries the upstream token usage when the stream chunk provides
|
|
// it (OpenAI-style streams attach usage to some chunks; the final one
|
|
// often has empty choices). Gateway uses it to emit exact usage in the
|
|
// final stream chunk instead of estimates.
|
|
Usage *TokenUsage `json:"usage,omitempty"`
|
|
}
|
|
|
|
// Meta passed to Lua build_headers hook
|
|
type BuildMeta struct {
|
|
URL string `json:"url"`
|
|
Method string `json:"method"`
|
|
Body string `json:"body"`
|
|
APIKey string `json:"api_key"`
|
|
Timestamp int64 `json:"timestamp"`
|
|
Source map[string]interface{} `json:"source"`
|
|
}
|
|
|
|
func Now() int64 { return time.Now().Unix() }
|
|
|
|
// OneLine flattens an error string to a single line capped at n chars.
|
|
// Shared by the provider layer (upstream error reasons) and the gateway
|
|
// layer (client-facing AUTO-chain summaries).
|
|
func OneLine(s string, n int) string {
|
|
s = strings.Join(strings.Fields(s), " ")
|
|
if len(s) > n {
|
|
s = s[:n] + "..."
|
|
}
|
|
return s
|
|
}
|