mirror of
https://gitcode.com/JianFeeeee/ModelRouter.git
synced 2026-10-03 15:44:05 +00:00
feat: AUTO chain rewrite — silent failover+busy skip+pref round-robin+503 tier summary; chain edits reset slot cooldowns (P0/P1); stats by_status + audit jsonl rotation; UI priority-page health badges & status-code card; ctx-menu capture-phase close (outside-press guard); main.go ops warnings; local bundled-Lua verified tests (3 latent bugs fixed); plan.md
This commit is contained in:
@ -3,6 +3,7 @@ package gateway
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"fmt"
|
||||
"net/http"
|
||||
"strings"
|
||||
@ -17,15 +18,15 @@ import (
|
||||
|
||||
// chatRequest mirrors the OpenAI chat completions request the gateway accepts.
|
||||
type chatRequest struct {
|
||||
Model string `json:"model"`
|
||||
Messages []types.ChatMessage `json:"messages"`
|
||||
Temperature *float64 `json:"temperature,omitempty"`
|
||||
MaxTokens int `json:"max_tokens,omitempty"`
|
||||
Stream bool `json:"stream,omitempty"`
|
||||
Tools []interface{} `json:"tools,omitempty"`
|
||||
ToolChoice interface{} `json:"tool_choice,omitempty"`
|
||||
DisableThinking bool `json:"disable_thinking"`
|
||||
ExtraBody map[string]interface{} `json:"extra_body,omitempty"`
|
||||
Model string `json:"model"`
|
||||
Messages []types.ChatMessage `json:"messages"`
|
||||
Temperature *float64 `json:"temperature,omitempty"`
|
||||
MaxTokens int `json:"max_tokens,omitempty"`
|
||||
Stream bool `json:"stream,omitempty"`
|
||||
Tools []interface{} `json:"tools,omitempty"`
|
||||
ToolChoice interface{} `json:"tool_choice,omitempty"`
|
||||
DisableThinking bool `json:"disable_thinking"`
|
||||
ExtraBody map[string]interface{} `json:"extra_body,omitempty"`
|
||||
}
|
||||
|
||||
// ChatCompletion is the non-streaming OpenAI response object.
|
||||
@ -60,9 +61,9 @@ type ChatChunk struct {
|
||||
}
|
||||
|
||||
type ChunkChoice struct {
|
||||
Index int `json:"index"`
|
||||
Delta RespMessage `json:"delta"`
|
||||
FinishReason *string `json:"finish_reason"`
|
||||
Index int `json:"index"`
|
||||
Delta RespMessage `json:"delta"`
|
||||
FinishReason *string `json:"finish_reason"`
|
||||
}
|
||||
|
||||
var seq int64
|
||||
@ -279,9 +280,9 @@ func (g *Gateway) handleChat(w http.ResponseWriter, r *http.Request) {
|
||||
model = g.core.DefaultModel()
|
||||
}
|
||||
if isAuto(model) {
|
||||
plans := g.autoPlans()
|
||||
if len(plans) == 0 {
|
||||
writeError(w, http.StatusServiceUnavailable, "no_provider", "no auto slot available (quota exhausted or none configured)")
|
||||
chain := g.core.AutoChain()
|
||||
if chain == nil || len(chain.Tiers) == 0 {
|
||||
writeError(w, http.StatusServiceUnavailable, "no_provider", "no auto slot configured")
|
||||
return
|
||||
}
|
||||
if msg := g.checkModelScope(r.Context(), "AUTO"); msg != "" {
|
||||
@ -306,12 +307,21 @@ func (g *Gateway) handleChat(w http.ResponseWriter, r *http.Request) {
|
||||
Type: "chat",
|
||||
OK: false,
|
||||
}
|
||||
// quotaExhausted reports a slot whose token window has been used up;
|
||||
// exhausted slots are dropped from scheduling without penalty.
|
||||
quotaExhausted := func(sl *scheduler.Slot) bool {
|
||||
if sl.Quota <= 0 {
|
||||
return false
|
||||
}
|
||||
win := AutoPeriodSeconds(sl.Period, sl.Hours)
|
||||
return g.stats.WindowTokens(sl.Model, sl.Source, win) >= sl.Quota
|
||||
}
|
||||
if req.Stream {
|
||||
rec.Type = "stream"
|
||||
g.streamChatAuto(w, ctx, plans, inner, rec)
|
||||
g.streamChatAuto(w, ctx, chain, inner, rec, quotaExhausted)
|
||||
return
|
||||
}
|
||||
g.singleChatAuto(w, ctx, plans, inner, rec)
|
||||
g.singleChatAuto(w, ctx, chain, inner, rec, quotaExhausted)
|
||||
return
|
||||
}
|
||||
if !isAuto(model) {
|
||||
@ -322,7 +332,7 @@ func (g *Gateway) handleChat(w http.ResponseWriter, r *http.Request) {
|
||||
}
|
||||
cands, effective := g.resolveCands(r.Context(), &req)
|
||||
if len(cands) == 0 {
|
||||
writeError(w, http.StatusServiceUnavailable, "no_provider", "no LLM source configured")
|
||||
writeError(w, http.StatusNotFound, "model_not_found", fmt.Sprintf("model %q is not configured", model))
|
||||
return
|
||||
}
|
||||
if effective == "" {
|
||||
@ -472,17 +482,43 @@ func estimateTextTokens(parts ...interface{}) int64 {
|
||||
return int64(n/3 + 1)
|
||||
}
|
||||
|
||||
// toScheduler adapts concrete providers to the scheduler.Provider interface.
|
||||
// It lives here (not in the scheduler package) so scheduler tests do not pull
|
||||
// in the provider package and with it the Lua runtime's link requirements.
|
||||
func toScheduler(cands []*provider.Provider) []scheduler.Provider {
|
||||
out := make([]scheduler.Provider, len(cands))
|
||||
for i, p := range cands {
|
||||
out[i] = p
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// upstreamErrStatus maps a scheduling error to its HTTP status: a failed
|
||||
// AUTO chain answers 503 with its per-tier summary, a busy source (every
|
||||
// concurrency slot in use) is a transient capacity condition answered with
|
||||
// 429 so clients fail fast, while other upstream failures stay 502.
|
||||
func upstreamErrStatus(err error) int {
|
||||
var ce *scheduler.ChainErr
|
||||
if errors.As(err, &ce) {
|
||||
return http.StatusServiceUnavailable
|
||||
}
|
||||
if errors.Is(err, provider.ErrBusy) {
|
||||
return http.StatusTooManyRequests
|
||||
}
|
||||
return http.StatusBadGateway
|
||||
}
|
||||
|
||||
func (g *Gateway) singleChat(w http.ResponseWriter, ctx context.Context, cands []*provider.Provider, req *types.ChatRequest, effective string, rec *Req) {
|
||||
rec.LatMs = 0
|
||||
t0 := time.Now()
|
||||
resp, usedSrc, usedModel, err := g.core.Scheduler().Chat(ctx, scheduler.FromRegistry(cands), req)
|
||||
resp, usedSrc, usedModel, err := g.core.Scheduler().Chat(ctx, toScheduler(cands), req)
|
||||
rec.LatMs = time.Since(t0).Milliseconds()
|
||||
if err != nil {
|
||||
rec.OK = false
|
||||
rec.Status = http.StatusBadGateway
|
||||
rec.Status = upstreamErrStatus(err)
|
||||
rec.Err = err.Error()
|
||||
g.writeRec(rec)
|
||||
writeError(w, http.StatusBadGateway, "upstream_error", err.Error())
|
||||
writeError(w, rec.Status, "upstream_error", err.Error())
|
||||
return
|
||||
}
|
||||
rec.OK = true
|
||||
@ -538,12 +574,12 @@ func (g *Gateway) streamChat(w http.ResponseWriter, ctx context.Context, cands [
|
||||
rec.LatMs = time.Since(t0).Milliseconds()
|
||||
g.writeRec(rec)
|
||||
}()
|
||||
chunks, _, usedModel, err := g.core.Scheduler().ChatStream(ctx, scheduler.FromRegistry(cands), req)
|
||||
chunks, _, usedModel, err := g.core.Scheduler().ChatStream(ctx, toScheduler(cands), req)
|
||||
if err != nil {
|
||||
rec.OK = false
|
||||
rec.Status = http.StatusBadGateway
|
||||
rec.Status = upstreamErrStatus(err)
|
||||
rec.Err = err.Error()
|
||||
writeError(w, http.StatusBadGateway, "upstream_error", err.Error())
|
||||
writeError(w, rec.Status, "upstream_error", err.Error())
|
||||
return
|
||||
}
|
||||
if usedModel != "" {
|
||||
@ -611,146 +647,66 @@ func (g *Gateway) streamChat(w http.ResponseWriter, ctx context.Context, cands [
|
||||
}
|
||||
}
|
||||
|
||||
// autoPlan is one schedulable AUTO slot: a model id pinned to its provider
|
||||
// with an optional token quota window. Quota-exhausted slots are skipped.
|
||||
type autoPlan struct {
|
||||
p *provider.Provider
|
||||
model string
|
||||
tier int
|
||||
quota int64
|
||||
win int64
|
||||
}
|
||||
|
||||
// autoPlans builds the schedulable AUTO slots from the persisted rules. A
|
||||
// slot is schedulable while its model is available and (when quota > 0) the
|
||||
// tokens used within its reset window are below the quota. Slots are returned
|
||||
// tiered high→low, and within the same tier the order is rotated round-robin
|
||||
// so concurrent requests spread evenly across equal-priority sources (still
|
||||
// with failover to the next slot if one errors).
|
||||
func (g *Gateway) autoPlans() []autoPlan {
|
||||
rules := g.core.AutoRules()
|
||||
if len(rules) == 0 {
|
||||
return nil
|
||||
}
|
||||
plans := make([]autoPlan, 0, len(rules))
|
||||
for _, e := range rules {
|
||||
p := g.core.ProviderForSlot(e.Model, e.Source)
|
||||
if p == nil {
|
||||
continue
|
||||
}
|
||||
if m := p.ModelByID(e.Model); m != nil && m.Kind == "image" {
|
||||
continue
|
||||
}
|
||||
win := AutoPeriodSeconds(e.Period, e.Hours)
|
||||
if e.TokenQuota > 0 && g.stats.WindowTokens(e.Model, e.Source, win) >= e.TokenQuota {
|
||||
continue
|
||||
}
|
||||
plans = append(plans, autoPlan{p: p, model: e.Model, tier: e.Tier, quota: e.TokenQuota, win: win})
|
||||
}
|
||||
return g.rotateSameTier(plans)
|
||||
}
|
||||
|
||||
// rotateSameTier reorders the leading plan of each consecutive same-tier run
|
||||
// using a global round-robin counter, so requests distribute across
|
||||
// equal-priority sources while preserving tier ordering and in-tier failover.
|
||||
func (g *Gateway) rotateSameTier(plans []autoPlan) []autoPlan {
|
||||
if len(plans) < 2 {
|
||||
// allow single slot without varying
|
||||
return plans
|
||||
}
|
||||
rot := int(g.autoRR.Add(1))
|
||||
out := make([]autoPlan, 0, len(plans))
|
||||
for i := 0; i < len(plans); {
|
||||
j := i
|
||||
for j < len(plans) && plans[j].tier == plans[i].tier {
|
||||
j++
|
||||
}
|
||||
run := plans[i:j]
|
||||
if len(run) > 1 {
|
||||
off := rot % len(run)
|
||||
run = append(run[off:], run[:off]...)
|
||||
}
|
||||
out = append(out, run...)
|
||||
i = j
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// singleChatAuto runs a non-streaming AUTO request slot by slot: each slot
|
||||
// pins its own model; a slot whose provider errors out is skipped. The first
|
||||
// slot to answer wins; when every slot fails, the recorded error is from the
|
||||
// last one.
|
||||
func (g *Gateway) singleChatAuto(w http.ResponseWriter, ctx context.Context, plans []autoPlan, req *types.ChatRequest, rec *Req) {
|
||||
// singleChatAuto runs a non-streaming AUTO request down the chain (see
|
||||
// scheduler.ChainChat): tiers descending, per-tier round-robin ordered by
|
||||
// preference, cooldown as the only hard skip, busy slots skipped without
|
||||
// penalty and a bounded busy wait. When every tier fails, the response is a
|
||||
// 503 carrying the per-tier error summary (which source/model failed why).
|
||||
func (g *Gateway) singleChatAuto(w http.ResponseWriter, ctx context.Context, chain *scheduler.Chain, req *types.ChatRequest, rec *Req, quotaExhausted func(*scheduler.Slot) bool) {
|
||||
rec.LatMs = 0
|
||||
t0 := time.Now()
|
||||
var lastErr error
|
||||
var lastSrc, lastModel string
|
||||
for _, pl := range plans {
|
||||
if !pl.p.Available() {
|
||||
continue
|
||||
resp, usedSrc, usedModel, err := g.core.Scheduler().ChainChat(ctx, chain, req, quotaExhausted)
|
||||
rec.LatMs = time.Since(t0).Milliseconds()
|
||||
if err != nil {
|
||||
rec.OK = false
|
||||
rec.Status = upstreamErrStatus(err)
|
||||
rec.Err = err.Error()
|
||||
if ce, ok := err.(*scheduler.ChainErr); ok && len(ce.Tiers) > 0 {
|
||||
rec.Source = ce.Tiers[0].Source
|
||||
rec.Model = ce.Tiers[0].Model
|
||||
}
|
||||
r := *req
|
||||
r.Model = pl.model
|
||||
lastSrc, lastModel = pl.p.Name(), pl.model
|
||||
resp, usedSrc, usedModel, err := g.core.Scheduler().Chat(ctx, scheduler.FromRegistry([]*provider.Provider{pl.p}), &r)
|
||||
if err != nil {
|
||||
lastErr = err
|
||||
continue
|
||||
}
|
||||
rec.LatMs = time.Since(t0).Milliseconds()
|
||||
rec.OK = true
|
||||
rec.Status = http.StatusOK
|
||||
rec.Prompt = int64(resp.TokenUsage.Prompt)
|
||||
if rec.Prompt == 0 {
|
||||
rec.Prompt = estimatePromptTokens(&r)
|
||||
}
|
||||
rec.Compl = int64(resp.TokenUsage.Completion)
|
||||
if rec.Compl == 0 {
|
||||
rec.Compl = estimateTextTokens(resp.Content, resp.ReasoningContent, resp.ToolCalls)
|
||||
}
|
||||
rec.Source = usedSrc
|
||||
rec.Model = usedModel
|
||||
g.writeRec(rec)
|
||||
msg := RespMessage{Role: "assistant", Content: resp.Content}
|
||||
if resp.ReasoningContent != "" {
|
||||
msg.ReasoningContent = resp.ReasoningContent
|
||||
}
|
||||
if len(resp.ToolCalls) > 0 {
|
||||
msg.ToolCalls = toolCallsWire(resp.ToolCalls)
|
||||
}
|
||||
out := ChatCompletion{
|
||||
ID: newID(),
|
||||
Object: "chat.completion",
|
||||
Created: time.Now().Unix(),
|
||||
Model: usedModel,
|
||||
Choices: []ChatChoice{{Index: 0, Message: msg, FinishReason: resp.FinishReason}},
|
||||
}
|
||||
if resp.TokenUsage.Total > 0 || resp.TokenUsage.Prompt > 0 || resp.TokenUsage.Completion > 0 {
|
||||
out.Usage = &resp.TokenUsage
|
||||
}
|
||||
writeJSON(w, http.StatusOK, out)
|
||||
writeError(w, rec.Status, "upstream_error", err.Error())
|
||||
return
|
||||
}
|
||||
rec.LatMs = time.Since(t0).Milliseconds()
|
||||
if lastErr == nil {
|
||||
lastErr = fmt.Errorf("no provider available")
|
||||
rec.OK = true
|
||||
rec.Status = http.StatusOK
|
||||
rec.Prompt = int64(resp.TokenUsage.Prompt)
|
||||
if rec.Prompt == 0 {
|
||||
rec.Prompt = estimatePromptTokens(req)
|
||||
}
|
||||
rec.OK = false
|
||||
rec.Status = http.StatusBadGateway
|
||||
rec.Err = lastErr.Error()
|
||||
if rec.Model == "" {
|
||||
rec.Model = lastModel
|
||||
}
|
||||
if rec.Source == "" {
|
||||
rec.Source = lastSrc
|
||||
rec.Compl = int64(resp.TokenUsage.Completion)
|
||||
if rec.Compl == 0 {
|
||||
rec.Compl = estimateTextTokens(resp.Content, resp.ReasoningContent, resp.ToolCalls)
|
||||
}
|
||||
rec.Source = usedSrc
|
||||
rec.Model = usedModel
|
||||
g.writeRec(rec)
|
||||
writeError(w, http.StatusBadGateway, "upstream_error", lastErr.Error())
|
||||
msg := RespMessage{Role: "assistant", Content: resp.Content}
|
||||
if resp.ReasoningContent != "" {
|
||||
msg.ReasoningContent = resp.ReasoningContent
|
||||
}
|
||||
if len(resp.ToolCalls) > 0 {
|
||||
msg.ToolCalls = toolCallsWire(resp.ToolCalls)
|
||||
}
|
||||
out := ChatCompletion{
|
||||
ID: newID(),
|
||||
Object: "chat.completion",
|
||||
Created: time.Now().Unix(),
|
||||
Model: usedModel,
|
||||
Choices: []ChatChoice{{Index: 0, Message: msg, FinishReason: resp.FinishReason}},
|
||||
}
|
||||
if resp.TokenUsage.Total > 0 || resp.TokenUsage.Prompt > 0 || resp.TokenUsage.Completion > 0 {
|
||||
out.Usage = &resp.TokenUsage
|
||||
}
|
||||
writeJSON(w, http.StatusOK, out)
|
||||
}
|
||||
|
||||
// streamChatAuto streams an AUTO request. It stays pinned to the first slot
|
||||
// whose stream begins; a slot that fails to connect is skipped.
|
||||
func (g *Gateway) streamChatAuto(w http.ResponseWriter, ctx context.Context, plans []autoPlan, req *types.ChatRequest, rec *Req) {
|
||||
// streamChatAuto streams an AUTO request down the chain. A slot is abandoned
|
||||
// only before its first chunk (connect error / non-200 / busy); once a stream
|
||||
// starts it stays pinned. Total failure writes a JSON 503 (with the per-tier
|
||||
// summary) before any SSE byte is sent.
|
||||
func (g *Gateway) streamChatAuto(w http.ResponseWriter, ctx context.Context, chain *scheduler.Chain, req *types.ChatRequest, rec *Req, quotaExhausted func(*scheduler.Slot) bool) {
|
||||
rec.LatMs = 0
|
||||
t0 := time.Now()
|
||||
rec.OK = true
|
||||
@ -759,6 +715,23 @@ func (g *Gateway) streamChatAuto(w http.ResponseWriter, ctx context.Context, pla
|
||||
rec.LatMs = time.Since(t0).Milliseconds()
|
||||
g.writeRec(rec)
|
||||
}()
|
||||
chunks, usedSrc, usedModel, err := g.core.Scheduler().ChainChatStream(ctx, chain, req, quotaExhausted)
|
||||
if err != nil {
|
||||
rec.OK = false
|
||||
rec.Status = upstreamErrStatus(err)
|
||||
rec.Err = err.Error()
|
||||
if ce, ok := err.(*scheduler.ChainErr); ok && len(ce.Tiers) > 0 {
|
||||
rec.Source = ce.Tiers[0].Source
|
||||
rec.Model = ce.Tiers[0].Model
|
||||
}
|
||||
writeError(w, rec.Status, "upstream_error", err.Error())
|
||||
return
|
||||
}
|
||||
if usedModel != "" {
|
||||
rec.Model = usedModel
|
||||
}
|
||||
rec.Source = usedSrc
|
||||
rec.Prompt = estimatePromptTokens(req)
|
||||
w.Header().Set("Content-Type", "text/event-stream")
|
||||
w.Header().Set("Cache-Control", "no-cache")
|
||||
w.Header().Set("Connection", "keep-alive")
|
||||
@ -780,68 +753,38 @@ func (g *Gateway) streamChatAuto(w http.ResponseWriter, ctx context.Context, pla
|
||||
return true
|
||||
}
|
||||
if !send(ChatChunk{
|
||||
ID: id, Object: "chat.completion.chunk", Created: created, Model: "auto",
|
||||
ID: id, Object: "chat.completion.chunk", Created: created, Model: rec.Model,
|
||||
Choices: []ChunkChoice{{Index: 0, Delta: RespMessage{Role: "assistant"}}},
|
||||
}) {
|
||||
return
|
||||
}
|
||||
var lastErr error
|
||||
var lastSrc, lastModel string
|
||||
for _, pl := range plans {
|
||||
if !pl.p.Available() {
|
||||
continue
|
||||
for ck := range chunks {
|
||||
chunk := ChatChunk{
|
||||
ID: id, Object: "chat.completion.chunk", Created: created, Model: rec.Model,
|
||||
}
|
||||
r := *req
|
||||
r.Model = pl.model
|
||||
lastSrc, lastModel = pl.p.Name(), pl.model
|
||||
chunks, usedSrc, usedModel, err := g.core.Scheduler().ChatStream(ctx, scheduler.FromRegistry([]*provider.Provider{pl.p}), &r)
|
||||
if err != nil {
|
||||
lastErr = err
|
||||
continue
|
||||
delta := RespMessage{Role: "assistant", Content: ck.Content}
|
||||
if ck.ReasoningContent != "" {
|
||||
delta.ReasoningContent = ck.ReasoningContent
|
||||
}
|
||||
if usedModel != "" {
|
||||
rec.Model = usedModel
|
||||
rec.Source = usedSrc
|
||||
if len(ck.ToolCalls) > 0 {
|
||||
delta.ToolCalls = ck.ToolCalls
|
||||
}
|
||||
rec.Prompt = estimatePromptTokens(&r)
|
||||
for ck := range chunks {
|
||||
chunk := ChatChunk{
|
||||
ID: id, Object: "chat.completion.chunk", Created: created, Model: rec.Model,
|
||||
}
|
||||
delta := RespMessage{Role: "assistant", Content: ck.Content}
|
||||
if ck.ReasoningContent != "" {
|
||||
delta.ReasoningContent = ck.ReasoningContent
|
||||
}
|
||||
if len(ck.ToolCalls) > 0 {
|
||||
delta.ToolCalls = ck.ToolCalls
|
||||
}
|
||||
choice := ChunkChoice{Index: 0, Delta: delta}
|
||||
if ck.Done {
|
||||
stop := "stop"
|
||||
choice.FinishReason = &stop
|
||||
}
|
||||
chunk.Choices = []ChunkChoice{choice}
|
||||
rec.Compl += int64(len(ck.Content)+len(ck.ReasoningContent)+len(ck.ToolCalls)) / 3
|
||||
if !send(chunk) {
|
||||
return
|
||||
}
|
||||
choice := ChunkChoice{Index: 0, Delta: delta}
|
||||
if ck.Done {
|
||||
stop := "stop"
|
||||
choice.FinishReason = &stop
|
||||
}
|
||||
chunk.Choices = []ChunkChoice{choice}
|
||||
rec.Compl += int64(len(ck.Content)+len(ck.ReasoningContent)+len(ck.ToolCalls)) / 3
|
||||
if !send(chunk) {
|
||||
return
|
||||
}
|
||||
return
|
||||
}
|
||||
if lastErr == nil {
|
||||
lastErr = fmt.Errorf("no provider available")
|
||||
}
|
||||
rec.OK = false
|
||||
rec.Status = http.StatusBadGateway
|
||||
rec.Err = lastErr.Error()
|
||||
if rec.Model == "" {
|
||||
rec.Model = lastModel
|
||||
}
|
||||
if rec.Source == "" {
|
||||
rec.Source = lastSrc
|
||||
}
|
||||
errEvent, _ := json.Marshal(map[string]interface{}{"error": map[string]string{"message": lastErr.Error(), "type": "upstream_error"}})
|
||||
fmt.Fprintf(w, "data: %s\n\n", errEvent)
|
||||
stop := "stop"
|
||||
send(ChatChunk{
|
||||
ID: id, Object: "chat.completion.chunk", Created: created, Model: rec.Model,
|
||||
Choices: []ChunkChoice{{Index: 0, Delta: RespMessage{}, FinishReason: &stop}},
|
||||
})
|
||||
fmt.Fprintf(w, "data: [DONE]\n\n")
|
||||
if flusher != nil {
|
||||
flusher.Flush()
|
||||
@ -889,13 +832,13 @@ func (g *Gateway) handleImage(w http.ResponseWriter, r *http.Request) {
|
||||
defer done()
|
||||
rec := &Req{Key: keyID(reqKey(r.Context())), Type: "image", Model: model, Source: firstSource(cands), OK: false}
|
||||
t0 := time.Now()
|
||||
resp, usedSrc, err := g.core.Scheduler().Image(r.Context(), scheduler.FromRegistry(cands), &req)
|
||||
resp, usedSrc, err := g.core.Scheduler().Image(r.Context(), toScheduler(cands), &req)
|
||||
rec.LatMs = time.Since(t0).Milliseconds()
|
||||
if err != nil {
|
||||
rec.Status = http.StatusBadGateway
|
||||
rec.Status = upstreamErrStatus(err)
|
||||
rec.Err = err.Error()
|
||||
g.writeRec(rec)
|
||||
writeError(w, http.StatusBadGateway, "upstream_error", err.Error())
|
||||
writeError(w, rec.Status, "upstream_error", err.Error())
|
||||
return
|
||||
}
|
||||
if usedSrc != "" {
|
||||
@ -909,4 +852,4 @@ func (g *Gateway) handleImage(w http.ResponseWriter, r *http.Request) {
|
||||
Created: time.Now().Unix(),
|
||||
Data: resp.ImageData,
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
@ -40,6 +40,7 @@ func newTestGateway(t *testing.T, srcs ...config.Source) *Gateway {
|
||||
cfg := &config.Config{
|
||||
AdapterDir: filepath.Join(t.TempDir(), "adapters"),
|
||||
RuntimeFile: filepath.Join(t.TempDir(), "runtime.json"),
|
||||
GatewayKeys: []string{"sk-test"},
|
||||
Sources: srcs,
|
||||
}
|
||||
if err := cfg.ApplyDefaults(); err != nil {
|
||||
@ -69,6 +70,192 @@ func doReq(t *testing.T, g *Gateway, method, path, body string) *httptest.Respon
|
||||
return rr
|
||||
}
|
||||
|
||||
// upstreamCtrl toggles a mocked upstream's behavior between requests.
|
||||
type upstreamCtrl struct {
|
||||
status int // 0 = healthy; else every request fails with that status
|
||||
hits int // chat call count
|
||||
}
|
||||
|
||||
// upstream returns a mocked OpenAI upstream driven by ctrl.status.
|
||||
func upstream(t *testing.T, ctrl *upstreamCtrl) *httptest.Server {
|
||||
t.Helper()
|
||||
return httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
ctrl.hits++
|
||||
if ctrl.status != 0 {
|
||||
w.WriteHeader(ctrl.status)
|
||||
fmt.Fprint(w, `{"error":"boom"}`)
|
||||
return
|
||||
}
|
||||
w.Header().Set("Content-Type", "application/json")
|
||||
fmt.Fprintf(w, `{"choices":[{"message":{"content":"pong"},"finish_reason":"stop"}],"usage":{"prompt_tokens":3,"completion_tokens":1,"total_tokens":4}}`)
|
||||
}))
|
||||
}
|
||||
|
||||
// TestChatAutoChainTierFailover: AUTO chain, first slot hard-fails, the pass
|
||||
// moves on within the same tier and the request is served by the next slot.
|
||||
func TestChatAutoChainTierFailover(t *testing.T) {
|
||||
a, b := &upstreamCtrl{status: 500}, &upstreamCtrl{}
|
||||
aUp := upstream(t, a)
|
||||
bUp := upstream(t, b)
|
||||
defer aUp.Close()
|
||||
defer bUp.Close()
|
||||
g := newTestGateway(t,
|
||||
config.Source{Name: "a", BaseURL: aUp.URL, Adapter: "openai", Models: []config.Model{{ID: "a-m", Priority: 100}}},
|
||||
config.Source{Name: "b", BaseURL: bUp.URL, Adapter: "openai", Models: []config.Model{{ID: "b-m", Priority: 10}}},
|
||||
)
|
||||
rr := doReq(t, g, "POST", "/v1/chat/completions",
|
||||
`{"model":"AUTO","messages":[{"role":"user","content":"hi"}]}`)
|
||||
if rr.Code != 200 {
|
||||
t.Fatalf("status=%d body=%s", rr.Code, rr.Body.String())
|
||||
}
|
||||
var cc ChatCompletion
|
||||
_ = json.Unmarshal(rr.Body.Bytes(), &cc)
|
||||
if cc.Model != "b-m" {
|
||||
t.Fatalf("AUTO served %q, want b-m", cc.Model)
|
||||
}
|
||||
if a.hits == 0 || b.hits == 0 {
|
||||
t.Fatalf("hit counts a=%d b=%d, want both > 0", a.hits, b.hits)
|
||||
}
|
||||
}
|
||||
|
||||
// TestChatAutoChain503Summary: every AUTO slot fails -> 503 whose message
|
||||
// names each failed tier/source/model.
|
||||
func TestChatAutoChain503Summary(t *testing.T) {
|
||||
a, b := &upstreamCtrl{status: 500}, &upstreamCtrl{status: 500}
|
||||
aUp := upstream(t, a)
|
||||
bUp := upstream(t, b)
|
||||
defer aUp.Close()
|
||||
defer bUp.Close()
|
||||
g := newTestGateway(t,
|
||||
config.Source{Name: "a", BaseURL: aUp.URL, Adapter: "openai", Models: []config.Model{{ID: "a-m", Priority: 100}}},
|
||||
config.Source{Name: "b", BaseURL: bUp.URL, Adapter: "openai", Models: []config.Model{{ID: "b-m", Priority: 10}}},
|
||||
)
|
||||
rr := doReq(t, g, "POST", "/v1/chat/completions",
|
||||
`{"model":"AUTO","messages":[{"role":"user","content":"hi"}]}`)
|
||||
if rr.Code != http.StatusServiceUnavailable {
|
||||
t.Fatalf("status=%d body=%s", rr.Code, rr.Body.String())
|
||||
}
|
||||
if !strings.Contains(rr.Body.String(), "all auto tiers failed") ||
|
||||
!strings.Contains(rr.Body.String(), "a/a-m") ||
|
||||
!strings.Contains(rr.Body.String(), "b/b-m") {
|
||||
t.Fatalf("503 must summarize every tier, body=%s", rr.Body.String())
|
||||
}
|
||||
}
|
||||
|
||||
// TestChatAutoQuotaSkip: a slot whose token quota is exhausted is dropped
|
||||
// from scheduling; with no other slot the chain answers 503 naming the quota.
|
||||
func TestChatAutoQuotaSkip(t *testing.T) {
|
||||
ctrl := &upstreamCtrl{}
|
||||
up := upstream(t, ctrl)
|
||||
defer up.Close()
|
||||
g := newTestGateway(t,
|
||||
config.Source{Name: "a", BaseURL: up.URL, Adapter: "openai", Models: []config.Model{{ID: "a-m", Priority: 100}}},
|
||||
)
|
||||
// one slot with an hourly quota of 1 token
|
||||
rr := doReq(t, g, "PUT", "/api/auto",
|
||||
`{"rules":[{"model":"a-m","tier":0,"token_quota":1,"period":"hour"}]}`)
|
||||
if rr.Code != 200 {
|
||||
t.Fatalf("put auto status=%d body=%s", rr.Code, rr.Body.String())
|
||||
}
|
||||
// first request consumes 4 tokens -> quota exhausted
|
||||
rr = doReq(t, g, "POST", "/v1/chat/completions",
|
||||
`{"model":"AUTO","messages":[{"role":"user","content":"hi"}]}`)
|
||||
if rr.Code != 200 {
|
||||
t.Fatalf("first status=%d body=%s", rr.Code, rr.Body.String())
|
||||
}
|
||||
// second request must skip the exhausted slot and fail 503
|
||||
rr = doReq(t, g, "POST", "/v1/chat/completions",
|
||||
`{"model":"AUTO","messages":[{"role":"user","content":"hi"}]}`)
|
||||
if rr.Code != http.StatusServiceUnavailable {
|
||||
t.Fatalf("quota status=%d body=%s", rr.Code, rr.Body.String())
|
||||
}
|
||||
if !strings.Contains(rr.Body.String(), "quota exhausted") {
|
||||
t.Fatalf("503 must name the quota reason, body=%s", rr.Body.String())
|
||||
}
|
||||
if ctrl.hits != 1 {
|
||||
t.Fatalf("upstream hits = %d, want 1 (exhausted slot must not be called)", ctrl.hits)
|
||||
}
|
||||
}
|
||||
|
||||
// TestAutoStatesReportChainHealth: GET /api/auto reports per-slot health for
|
||||
// the priority-page UI; a chain edit resets the failure state to zero.
|
||||
func TestAutoStatesReportChainHealth(t *testing.T) {
|
||||
ctrl := &upstreamCtrl{status: 500}
|
||||
up := upstream(t, ctrl)
|
||||
defer up.Close()
|
||||
g := newTestGateway(t,
|
||||
config.Source{Name: "a", BaseURL: up.URL, Adapter: "openai", Models: []config.Model{{ID: "a-m", Priority: 100}}},
|
||||
)
|
||||
doReq(t, g, "POST", "/v1/chat/completions",
|
||||
`{"model":"AUTO","messages":[{"role":"user","content":"hi"}]}`)
|
||||
|
||||
fetch := func() []core.AutoSlotState {
|
||||
rr := doReq(t, g, "GET", "/api/auto", "")
|
||||
if rr.Code != 200 {
|
||||
t.Fatalf("get auto status=%d body=%s", rr.Code, rr.Body.String())
|
||||
}
|
||||
var body struct {
|
||||
Rules []config.ModelScope `json:"rules"`
|
||||
States []core.AutoSlotState `json:"states"`
|
||||
}
|
||||
if err := json.Unmarshal(rr.Body.Bytes(), &body); err != nil {
|
||||
t.Fatalf("unmarshal: %v", err)
|
||||
}
|
||||
return body.States
|
||||
}
|
||||
|
||||
st := fetch()
|
||||
if len(st) != 1 || st[0].Model != "a-m" || st[0].Source != "a" {
|
||||
t.Fatalf("want 1 slot a/a-m, got %#v", st)
|
||||
}
|
||||
if st[0].FailCount == 0 || !st[0].Cooling {
|
||||
t.Fatalf("slot must report the failure (fail=%d cooling=%v)", st[0].FailCount, st[0].Cooling)
|
||||
}
|
||||
|
||||
doReq(t, g, "PUT", "/api/auto",
|
||||
`{"rules":[{"model":"a-m","tier":0}]}`)
|
||||
st = fetch()
|
||||
if st[0].FailCount != 0 || st[0].Cooling {
|
||||
t.Fatalf("edit must reset health, got %#v", st[0])
|
||||
}
|
||||
}
|
||||
|
||||
// TestAutoSaveResetsCooldown: editing the AUTO chain clears the cooldown of
|
||||
// its slots, so a fixed upstream is schedulable again without waiting (P1).
|
||||
func TestAutoSaveResetsCooldown(t *testing.T) {
|
||||
ctrl := &upstreamCtrl{status: 500}
|
||||
up := upstream(t, ctrl)
|
||||
defer up.Close()
|
||||
g := newTestGateway(t,
|
||||
config.Source{Name: "a", BaseURL: up.URL, Adapter: "openai", Models: []config.Model{{ID: "a-m", Priority: 100}}},
|
||||
)
|
||||
rr := doReq(t, g, "POST", "/v1/chat/completions",
|
||||
`{"model":"AUTO","messages":[{"role":"user","content":"hi"}]}`)
|
||||
if rr.Code != http.StatusServiceUnavailable {
|
||||
t.Fatalf("expect 503 while upstream down, got %d", rr.Code)
|
||||
}
|
||||
p := g.core.ProviderForSlot("a-m", "a")
|
||||
if p == nil || p.ModelAvailable("a-m") {
|
||||
t.Fatal("a-m must be cooling after the failure")
|
||||
}
|
||||
// editing the chain (same rules) must clear the cooldown immediately
|
||||
rr = doReq(t, g, "PUT", "/api/auto",
|
||||
`{"rules":[{"model":"a-m","tier":0}]}`)
|
||||
if rr.Code != 200 {
|
||||
t.Fatalf("put auto status=%d body=%s", rr.Code, rr.Body.String())
|
||||
}
|
||||
if !p.ModelAvailable("a-m") {
|
||||
t.Fatal("SaveAutoRules must reset the slot cooldown")
|
||||
}
|
||||
// healed upstream -> AUTO serves again on the next request
|
||||
ctrl.status = 0
|
||||
rr = doReq(t, g, "POST", "/v1/chat/completions",
|
||||
`{"model":"AUTO","messages":[{"role":"user","content":"hi"}]}`)
|
||||
if rr.Code != 200 {
|
||||
t.Fatalf("AUTO after reset status=%d body=%s", rr.Code, rr.Body.String())
|
||||
}
|
||||
}
|
||||
|
||||
func TestChatSingle(t *testing.T) {
|
||||
up := mockUpstream()
|
||||
defer up.Close()
|
||||
@ -446,4 +633,4 @@ func TestAPIChatInternal(t *testing.T) {
|
||||
if !strings.Contains(rr.Body.String(), "pong") {
|
||||
t.Fatalf("api chat body=%s", rr.Body.String())
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@ -142,7 +142,10 @@ func (g *Gateway) allowedModels(ctx context.Context) []config.ModelScope {
|
||||
// current rules; PUT /api/auto replaces them (admin only).
|
||||
func (g *Gateway) handleAutoAPI(w http.ResponseWriter, r *http.Request) {
|
||||
if r.Method == http.MethodGet {
|
||||
writeJSON(w, http.StatusOK, map[string]interface{}{"rules": g.core.AutoRules()})
|
||||
writeJSON(w, http.StatusOK, map[string]interface{}{
|
||||
"rules": g.core.AutoRules(),
|
||||
"states": g.core.AutoSlotStates(),
|
||||
})
|
||||
return
|
||||
}
|
||||
if reqRole(r.Context()) != "admin" {
|
||||
|
||||
@ -15,7 +15,6 @@ import (
|
||||
"net/url"
|
||||
"strings"
|
||||
"sync"
|
||||
"sync/atomic"
|
||||
"time"
|
||||
|
||||
"llmsproxy/internal/config"
|
||||
@ -27,12 +26,11 @@ var uiFS embed.FS
|
||||
|
||||
// Gateway is the HTTP handler for the OpenAI-compatible endpoint + web UI.
|
||||
type Gateway struct {
|
||||
core *core.Core
|
||||
ui http.Handler
|
||||
stats *Stats
|
||||
probeMu sync.Mutex
|
||||
lastProbe time.Time
|
||||
autoRR atomic.Uint64
|
||||
core *core.Core
|
||||
ui http.Handler
|
||||
stats *Stats
|
||||
probeMu sync.Mutex
|
||||
lastProbe time.Time
|
||||
}
|
||||
|
||||
func New(c *core.Core, gatewayKeys []string) (*Gateway, error) {
|
||||
@ -131,6 +129,8 @@ func (g *Gateway) routes(w http.ResponseWriter, r *http.Request) {
|
||||
g.handleChat(w, r)
|
||||
case r.URL.Path == "/api/status":
|
||||
g.handleStatusAPI(w, r)
|
||||
case r.URL.Path == "/api/status/reset":
|
||||
g.handleResetHealth(w, r)
|
||||
case r.URL.Path == "/api/stats" || strings.HasPrefix(r.URL.Path, "/api/stats/"):
|
||||
g.handleStatsAPI(w, r)
|
||||
case r.URL.Path == "/api/keys" || strings.HasPrefix(r.URL.Path, "/api/keys/"):
|
||||
@ -389,6 +389,22 @@ func (g *Gateway) ensureProbe(ctx context.Context) {
|
||||
}
|
||||
}
|
||||
|
||||
// handleResetHealth (admin) clears the per-source backoff state so a fixed
|
||||
// upstream or an edited AUTO priority chain becomes schedulable immediately.
|
||||
func (g *Gateway) handleResetHealth(w http.ResponseWriter, r *http.Request) {
|
||||
if r.Method != http.MethodPost {
|
||||
writeError(w, http.StatusMethodNotAllowed, "method_not_allowed", "use POST")
|
||||
return
|
||||
}
|
||||
if reqRole(r.Context()) != "admin" {
|
||||
writeError(w, http.StatusForbidden, "forbidden", "admin role required")
|
||||
return
|
||||
}
|
||||
g.core.ResetHealth()
|
||||
g.stats.AppendAudit("config", map[string]interface{}{"action": "reset_health", "key": keyID(reqKey(r.Context()))})
|
||||
writeJSON(w, http.StatusOK, map[string]interface{}{"ok": true})
|
||||
}
|
||||
|
||||
func (g *Gateway) handleStatusAPI(w http.ResponseWriter, r *http.Request) {
|
||||
g.ensureProbe(r.Context())
|
||||
host := r.Host
|
||||
|
||||
@ -3,7 +3,11 @@ package gateway
|
||||
import (
|
||||
"bufio"
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"sort"
|
||||
"strconv"
|
||||
"sync"
|
||||
"time"
|
||||
)
|
||||
@ -56,6 +60,7 @@ type Stats struct {
|
||||
bySrc map[string]*Stat
|
||||
byKeyModel map[string]map[string]*Stat
|
||||
byKeySrc map[string]map[string]*Stat
|
||||
byStatus map[int]*Stat // per http status code aggregates (incl. 402/400)
|
||||
recs []Req
|
||||
maxRecs int
|
||||
auditPath string
|
||||
@ -64,6 +69,15 @@ type Stats struct {
|
||||
|
||||
const hourSec = 3600
|
||||
|
||||
// auditRotateBytes rotates the audit file once it grows past this size (the
|
||||
// file is renamed to <path>.<unix>.old and a fresh one is started); pruning
|
||||
// keeps at most auditKeepOld rotated files. Both are vars so tests can shrink
|
||||
// the threshold.
|
||||
var (
|
||||
auditRotateBytes int64 = 64 << 20
|
||||
auditKeepOld = 10
|
||||
)
|
||||
|
||||
func NewStats(maxRecords int) *Stats {
|
||||
if maxRecords <= 0 {
|
||||
maxRecords = 3000
|
||||
@ -74,6 +88,7 @@ func NewStats(maxRecords int) *Stats {
|
||||
bySrc: map[string]*Stat{},
|
||||
byKeyModel: map[string]map[string]*Stat{},
|
||||
byKeySrc: map[string]map[string]*Stat{},
|
||||
byStatus: map[int]*Stat{},
|
||||
modelHour: map[string]map[int64]int64{},
|
||||
maxRecs: maxRecords,
|
||||
}
|
||||
@ -98,6 +113,10 @@ func inc(m map[string]*Stat, name string, r Req) {
|
||||
a = &Stat{}
|
||||
m[name] = a
|
||||
}
|
||||
incStatus(a, name, r)
|
||||
}
|
||||
|
||||
func incStatus(a *Stat, name string, r Req) {
|
||||
a.Reqs++
|
||||
if r.OK {
|
||||
a.OK++
|
||||
@ -160,6 +179,15 @@ func (s *Stats) Record(r Req) {
|
||||
s.byKeySrc[r.Key] = ks
|
||||
}
|
||||
inc(ks, r.Source, r)
|
||||
if r.Status > 0 {
|
||||
name := strconv.Itoa(r.Status)
|
||||
a := s.byStatus[r.Status]
|
||||
if a == nil {
|
||||
a = &Stat{}
|
||||
s.byStatus[r.Status] = a
|
||||
}
|
||||
incStatus(a, name, r)
|
||||
}
|
||||
// window bucket for quota enforcement (per source-model pair, per unix hour)
|
||||
tok := r.Prompt + r.Compl
|
||||
if tok > 0 && r.Model != "" {
|
||||
@ -187,28 +215,32 @@ func (s *Stats) Record(r Req) {
|
||||
s.recs = s.recs[len(s.recs)-s.maxRecs:]
|
||||
}
|
||||
if s.auditPath != "" {
|
||||
if f, err := os.OpenFile(s.auditPath, os.O_CREATE|os.O_APPEND|os.O_WRONLY, 0644); err == nil {
|
||||
if b, err := json.Marshal(r); err == nil {
|
||||
_, _ = f.Write(append(b, '\n'))
|
||||
}
|
||||
_ = f.Close()
|
||||
s.rotateAuditLocked()
|
||||
appendAuditLine(s.auditPath, r)
|
||||
}
|
||||
}
|
||||
|
||||
// rotateAuditLocked renames the audit file to <path>.<unix>.old once it
|
||||
// exceeds auditRotateBytes and prunes old files beyond auditKeepOld, keeping
|
||||
// the newest ones. Caller must hold s.mu.
|
||||
func (s *Stats) rotateAuditLocked() {
|
||||
if s.auditPath == "" || auditRotateBytes <= 0 {
|
||||
return
|
||||
}
|
||||
if fi, err := os.Stat(s.auditPath); err == nil && fi.Size() < auditRotateBytes {
|
||||
return
|
||||
}
|
||||
ts := time.Now().Unix()
|
||||
if os.Rename(s.auditPath, fmt.Sprintf("%s.%d.old", s.auditPath, ts)) == nil {
|
||||
old, _ := filepath.Glob(s.auditPath + ".*.old")
|
||||
sort.Sort(sort.Reverse(sort.StringSlice(old)))
|
||||
for i := auditKeepOld; i < len(old); i++ {
|
||||
_ = os.Remove(old[i])
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// AppendAudit writes a generic event line (access log entry, login event,
|
||||
// config change, …) to the same audit file without touching the aggregates.
|
||||
func (s *Stats) AppendAudit(obj string, data map[string]interface{}) {
|
||||
s.mu.Lock()
|
||||
path := s.auditPath
|
||||
s.mu.Unlock()
|
||||
if path == "" {
|
||||
return
|
||||
}
|
||||
row := map[string]interface{}{"obj": obj, "time": time.Now().UnixMilli()}
|
||||
for k, v := range data {
|
||||
row[k] = v
|
||||
}
|
||||
func appendAuditLine(path string, row interface{}) {
|
||||
f, err := os.OpenFile(path, os.O_CREATE|os.O_APPEND|os.O_WRONLY, 0644)
|
||||
if err != nil {
|
||||
return
|
||||
@ -219,6 +251,22 @@ func (s *Stats) AppendAudit(obj string, data map[string]interface{}) {
|
||||
}
|
||||
}
|
||||
|
||||
// AppendAudit writes a generic event line (access log entry, login event,
|
||||
// config change, …) to the same audit file without touching the aggregates.
|
||||
func (s *Stats) AppendAudit(obj string, data map[string]interface{}) {
|
||||
row := map[string]interface{}{"obj": obj, "time": time.Now().UnixMilli()}
|
||||
for k, v := range data {
|
||||
row[k] = v
|
||||
}
|
||||
s.mu.Lock()
|
||||
defer s.mu.Unlock()
|
||||
if s.auditPath == "" {
|
||||
return
|
||||
}
|
||||
s.rotateAuditLocked()
|
||||
appendAuditLine(s.auditPath, row)
|
||||
}
|
||||
|
||||
// ModelTokens returns the tokens consumed per model for one gateway key id
|
||||
// (used for per-model token quota enforcement).
|
||||
func (s *Stats) ModelTokens(key string) map[string]int64 {
|
||||
@ -376,12 +424,22 @@ func (s *Stats) Snapshot(limit int, key string) map[string]interface{} {
|
||||
total.LatMax = a.LatMax
|
||||
}
|
||||
}
|
||||
bs := make([]agrRow, 0, len(s.byStatus))
|
||||
for code := range s.byStatus {
|
||||
bs = append(bs, agrRow{Name: strconv.Itoa(code), Stat: *s.byStatus[code]})
|
||||
}
|
||||
sort.Slice(bs, func(i, j int) bool {
|
||||
ci, _ := strconv.Atoi(bs[i].Name)
|
||||
cj, _ := strconv.Atoi(bs[j].Name)
|
||||
return ci < cj
|
||||
})
|
||||
return map[string]interface{}{
|
||||
"active": s.active,
|
||||
"total": total,
|
||||
"by_key": rows(byKey),
|
||||
"by_model": rows(byModel),
|
||||
"by_source": rows(bySrc),
|
||||
"by_status": bs,
|
||||
"records": append([]Req(nil), recs...),
|
||||
}
|
||||
}
|
||||
88
internal/gateway/stats_test.go
Normal file
88
internal/gateway/stats_test.go
Normal file
@ -0,0 +1,88 @@
|
||||
package gateway
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"testing"
|
||||
)
|
||||
|
||||
func TestStatsByStatus(t *testing.T) {
|
||||
s := NewStats(100)
|
||||
s.Record(Req{Key: "k", Model: "m", Source: "s", Status: 200, OK: true})
|
||||
s.Record(Req{Key: "k", Model: "m", Source: "s", Status: 402, OK: false})
|
||||
s.Record(Req{Key: "k", Model: "m", Source: "s", Status: 400, OK: false})
|
||||
snap := s.Snapshot(0, "")
|
||||
bs, ok := snap["by_status"].([]agrRow)
|
||||
if !ok {
|
||||
t.Fatalf("by_status missing: %#v", snap["by_status"])
|
||||
}
|
||||
if len(bs) != 3 {
|
||||
t.Fatalf("want 3 status buckets, got %d: %#v", len(bs), bs)
|
||||
}
|
||||
if bs[0].Name != "200" || bs[0].OK != 1 || bs[0].Err != 0 {
|
||||
t.Fatalf("bucket 200 wrong: %#v", bs[0])
|
||||
}
|
||||
if bs[1].Name != "400" || bs[1].Err != 1 {
|
||||
t.Fatalf("bucket 400 wrong: %#v", bs[1])
|
||||
}
|
||||
if bs[2].Name != "402" || bs[2].Err != 1 {
|
||||
t.Fatalf("bucket 402 wrong: %#v", bs[2])
|
||||
}
|
||||
}
|
||||
|
||||
func TestAuditRotation(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
path := filepath.Join(dir, "audit.jsonl")
|
||||
s := NewStats(10)
|
||||
s.LoadAudit(path)
|
||||
|
||||
oldRotate, oldKeep := auditRotateBytes, auditKeepOld
|
||||
auditRotateBytes, auditKeepOld = 64, 10
|
||||
defer func() { auditRotateBytes, auditKeepOld = oldRotate, oldKeep }()
|
||||
|
||||
oldFiles := func() []string {
|
||||
matches, _ := filepath.Glob(path + ".*.old")
|
||||
return matches
|
||||
}
|
||||
|
||||
for i := 0; i < 3; i++ {
|
||||
s.AppendAudit("ev", map[string]interface{}{"i": i})
|
||||
}
|
||||
if got := len(oldFiles()); got != 1 {
|
||||
t.Fatalf("want 1 rotated file after first overflow, got %d", got)
|
||||
}
|
||||
if b, err := os.ReadFile(path); err != nil || len(b) == 0 {
|
||||
t.Fatalf("active audit file must continue appending: %v %d bytes", err, len(b))
|
||||
}
|
||||
|
||||
// seed 12 fake old files; the next rotation must prune back to keep=10
|
||||
for i := 1; i <= 12; i++ {
|
||||
name := fmt.Sprintf("%s.%010d.old", path, i)
|
||||
_ = os.WriteFile(name, []byte("x\n"), 0644)
|
||||
}
|
||||
s.AppendAudit("ev", map[string]interface{}{"i": 98})
|
||||
s.AppendAudit("ev", map[string]interface{}{"i": 99})
|
||||
if got := len(oldFiles()); got != auditKeepOld {
|
||||
t.Fatalf("want keeper %d old files, got %d", auditKeepOld, got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestAuditRotationRecords(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
path := filepath.Join(dir, "audit.jsonl")
|
||||
s := NewStats(10)
|
||||
s.LoadAudit(path)
|
||||
|
||||
oldRotate := auditRotateBytes
|
||||
auditRotateBytes = 64
|
||||
defer func() { auditRotateBytes = oldRotate }()
|
||||
|
||||
for i := 0; i < 5; i++ {
|
||||
s.Record(Req{Key: "k", Model: "m", Source: "s", Status: 200, OK: true})
|
||||
}
|
||||
matches, _ := filepath.Glob(path + ".*.old")
|
||||
if len(matches) != 1 {
|
||||
t.Fatalf("Record must rotate too: got %d old files", len(matches))
|
||||
}
|
||||
}
|
||||
@ -306,6 +306,14 @@ html[data-theme="dark"] .dropzone.dragover, html[data-theme="dark"] .dropzone:ho
|
||||
.scr-block .scr-tag { flex:0 0 auto; font-family:ui-monospace,Menlo,Consolas,monospace; font-size:10.5px;
|
||||
padding:2px 7px; border-radius:9px; background:rgba(0,0,0,.24); color:#ffe9a8;
|
||||
border:1px solid rgba(255,220,130,.35); cursor:pointer; }
|
||||
.scr-block .scr-htag { flex:0 0 auto; display:flex; gap:4px; align-items:center;
|
||||
font-family:ui-monospace,Menlo,Consolas,monospace; font-size:10px; font-weight:700; cursor:help; }
|
||||
.scr-htag .ht-cool { padding:2px 6px; border-radius:9px; background:rgba(255,80,80,.28); color:#ffd9d9;
|
||||
border:1px solid rgba(255,120,120,.5); }
|
||||
.scr-htag .ht-fail { padding:2px 6px; border-radius:9px; background:rgba(255,160,60,.2); color:#ffd9a8;
|
||||
border:1px solid rgba(255,180,90,.42); }
|
||||
.scr-htag .ht-pref { padding:2px 6px; border-radius:9px; background:rgba(120,180,255,.18); color:#cfe3ff;
|
||||
border:1px solid rgba(150,190,255,.38); }
|
||||
.scr-block .scr-grip { flex:0 0 auto; display:flex; flex-direction:column; gap:2px; padding:6px 4px;
|
||||
margin-left:2px; border-radius:6px; cursor:grab; background:rgba(255,255,255,.18);
|
||||
box-shadow:inset 0 1px 2px rgba(0,0,0,.18); transition:background .12s; touch-action:none; }
|
||||
@ -471,9 +479,11 @@ const STR = {
|
||||
seedWarnTitle:'请更换初始管理员密钥', seedWarnText:'当前登录的是配置文件中的初始密钥,明文写入 config.yaml、存在泄露风险。请在下方创建新的管理员密钥,用新密钥登录后删除此初始密钥。', seedWarnGo:'去更换密钥', seedWarnLater:'稍后', seedWarnDismiss:'本次不再提示',
|
||||
sortTitle:'拖拽积木配置模型优先级', sortHint:'每行 = 一个优先级档位,行从上到下优先级递减;同一行的模型并排,视为同优先级。按住积木右侧 ⠿ 把手拖动:拖到行内 = 放入该档位或调整同档顺序,拖到行与行之间的缝隙 = 提升或降低到新档位。生图模型不参与排序。', sortDragGrip:'拖拽前须按住把手',
|
||||
sortSave:'保存排序', sortReset:'重置', sortAdd:'添加档位', sortSaved:'排序已保存并热重载', sortNoChange:'无变更', sortHintSave:'点击保存排序后生效',
|
||||
sortCooling:'冷却', sortFail:'失败', sortHealthTip:'冷却 / 失败次数 / 偏好分 实时状态', sortHealthReset:'链上冷却已复位',
|
||||
sortSource:'源', sortPrio:'优先级 %s', sortEmpty:'该源暂无模型',
|
||||
kpiActive:'活跃请求', kpiReqs:'总请求', kpiOk:'成功率', kpiTokens:'Tokens', kpiLat:'平均延迟', kpiMaxLat:'最大延迟',
|
||||
dashModel:'模型用量', dashSrc:'源用量与延迟', dashKey:'密钥用量', dashRecs:'请求记录', exportCsv:'导出 CSV', expWeek:'近一周', expMonth:'近一月', expYear:'近一年', expRange:'自定义范围', expStart:'开始日期', expEnd:'结束日期', expDownload:'下载', expKeysCsv:'导出密钥用量',
|
||||
dashStatus:'状态码分布', thCode:'状态码', statusTag:'状态码分类统计(含 402 欠费 / 400 schema 错误;两者不计入上游退避但单独计数)',
|
||||
thModel:'模型', thSrc:'源', thKey:'密钥', thReqs:'请求', thOk:'成功', thErr:'失败',
|
||||
thPrompt:'输入 Tokens', thCompl:'输出 Tokens', thAvgLat:'平均延迟', thMaxLat:'最长延迟',
|
||||
thTime:'时间', thType:'类型', thStatus:'状态', thLatMs:'延迟',
|
||||
@ -525,9 +535,11 @@ kMeTitle:'My key', kMeRole:'Role', kMeModels:'Models I can use', kMeHint:'Keys c
|
||||
seedWarnTitle:'Replace the initial admin key', seedWarnText:'You are logged in with the seed key from config.yaml. It is plaintext in the config file and a security risk. Create a new admin key below, log in with it, then delete this seed key.', seedWarnGo:'Change my key', seedWarnLater:'Later', seedWarnDismiss:'Don\'t ask again',
|
||||
sortTitle:'Drag blocks to set model priority', sortHint:'Each row = one priority tier, rows go high→low; models on the same row sit side by side and share that priority. Grab the ⠿ handle on the right of a block to drag: drop into a row = join that tier or reorder within it, drop into the gap between rows = move up/down a tier. Image models stay out.', sortDragGrip:'grab the handle to drag',
|
||||
sortSave:'Save order', sortReset:'Reset', sortAdd:'Add slot', sortSaved:'Order saved & hot-reloaded', sortNoChange:'No changes', sortHintSave:'Click Save for it to take effect',
|
||||
sortCooling:'cooling', sortFail:'fail', sortHealthTip:'live cooldown / failures / preference score', sortHealthReset:'chain cooldowns reset',
|
||||
sortSource:'source', sortPrio:'priority %s', sortEmpty:'no models in this source',
|
||||
kpiActive:'Active requests', kpiReqs:'Requests', kpiOk:'Success rate', kpiTokens:'Tokens', kpiLat:'Avg latency', kpiMaxLat:'Max latency',
|
||||
dashModel:'Model usage', dashSrc:'Source usage & latency', dashKey:'Key usage', dashRecs:'Request records', exportCsv:'Export CSV', expWeek:'Last week', expMonth:'Last month', expYear:'Last year', expRange:'Custom range', expStart:'Start date', expEnd:'End date', expDownload:'Download',
|
||||
dashStatus:'Status codes', thCode:'Code', statusTag:'Per-status aggregates — 402 quota / 400 schema errors are counted here but never back off the provider',
|
||||
thModel:'Model', thSrc:'Source', thKey:'Key', thReqs:'Requests', thOk:'OK', thErr:'Err',
|
||||
thPrompt:'Prompt Tokens', thCompl:'Completion Tokens', thAvgLat:'Avg latency', thMaxLat:'Max latency',
|
||||
thTime:'Time', thType:'Type', thStatus:'Status', thLatMs:'Latency',
|
||||
@ -658,6 +670,7 @@ async function renderStatus() {
|
||||
<div class="card"><h2>${t('dashModel')}</h2><div id="tb-model"></div></div>
|
||||
${s.sources ? `<div class="card"><h2>${t('dashSrc')}</h2><div id="tb-src"></div></div>` : ''}
|
||||
</div>
|
||||
<div class="card"><h2>${t('dashStatus')} <span class="muted" style="font-weight:400;font-size:12px">${t('statusTag')}</span></h2><div id="tb-status"></div></div>
|
||||
<div class="card"><h2>${t('dashKey')}<span class="grow"></span><button class="ghost small" onclick="openExportModal()">${t('exportCsv')}</button></h2><div id="tb-key"></div></div>
|
||||
<div class="card"><h2><span>${t('dashRecs')}</span><span class="grow"></span><button class="ghost small" onclick="openExportModal()">${t('exportCsv')}</button></h2>
|
||||
<div class="filter-line">
|
||||
@ -732,6 +745,7 @@ async function paintStats() {
|
||||
<div class="kpi"><div class="k-lab">${t('kpiLat')}</div><div class="k-val">${fmtMs(avg)}</div><div class="k-sub">${t('kpiMaxLat')} ${fmtMs(tot.latency_max_ms)}</div></div>`;
|
||||
paintModelTable(st.by_model || []);
|
||||
paintSrcTable(st.by_source || []);
|
||||
paintStatusTable(st.by_status || []);
|
||||
paintKeyTable(st.by_key || [], st.key_names || {});
|
||||
paintRecords(st.records || [], st.key_names || {});
|
||||
renderKeySelect((st.by_key || []).map(k => k.name));
|
||||
@ -762,6 +776,13 @@ function paintSrcTable(rows) {
|
||||
<td class="num">${fmtTok(r.tokens)}</td>
|
||||
<td class="num">${fmtMs(fmtLat(r.latency_sum_ms, r.reqs))}</td><td class="num">${fmtMs(r.latency_max_ms)}</td></tr>`).join('') + '</table></div>';
|
||||
}
|
||||
function paintStatusTable(rows) {
|
||||
const el = $('#tb-status'); if (!el) return;
|
||||
if (!rows.length) { el.innerHTML = `<div class="muted">${t('noUsage')}</div>`; return; }
|
||||
el.innerHTML = `<div class="tbl-wrap"><table><tr><th>${t('thCode')}</th><th class="num">${t('thReqs')}</th><th class="num">${t('thOk')}</th><th class="num">${t('thErr')}</th></tr>` +
|
||||
rows.map(r => `<tr><td><b class="${+r.name >= 400 ? 'errc' : 'okc'}">${esc(r.name)}</b></td>
|
||||
<td class="num">${fmtN(r.reqs)}</td><td class="num okc">${fmtN(r.ok)}</td><td class="num errc">${fmtN(r.err)}</td></tr>`).join('') + '</table></div>';
|
||||
}
|
||||
function paintKeyTable(rows, keyNames) {
|
||||
const el = $('#tb-key'); if (!el) return;
|
||||
if (!rows.length) { el.innerHTML = `<div class="muted">${t('noUsage')}</div>`; return; }
|
||||
@ -1139,10 +1160,15 @@ function srcColor(name) {
|
||||
}
|
||||
function srcShort(name) { return (name || '?').slice(0, 2).toUpperCase(); }
|
||||
const sortState = { lanes: [], origin: null, drag: null };
|
||||
let sortStateMap = new Map();
|
||||
async function renderSort() {
|
||||
const j = await api('/api/sources');
|
||||
let autoR = [];
|
||||
try { autoR = (await api('/api/auto')).rules || []; } catch (e) {}
|
||||
try {
|
||||
const a = await api('/api/auto');
|
||||
autoR = a.rules || [];
|
||||
sortStateMap = new Map((a.states || []).map(st => [st.model + '|' + (st.source || '*'), st]));
|
||||
} catch (e) {}
|
||||
const byModel = new Map();
|
||||
const byPair = new Map();
|
||||
const sourceRows = new Map();
|
||||
@ -1194,6 +1220,16 @@ async function renderSort() {
|
||||
</div>`;
|
||||
paintSort();
|
||||
}
|
||||
function healthTag(it) {
|
||||
const st = sortStateMap.get(it.id + '|' + (it.src || '*'));
|
||||
if (!st) return '';
|
||||
const bits = [];
|
||||
if (st.cooling) bits.push(`<span class="ht-cool">${esc(t('sortCooling'))}</span>`);
|
||||
if (st.fail_count > 0) bits.push(`<span class="ht-fail">${esc(t('sortFail') + '×' + st.fail_count)}</span>`);
|
||||
if (st.pref !== 0) bits.push(`<span class="ht-pref">${esc(st.pref > 0 ? '+' + st.pref : '' + st.pref)}</span>`);
|
||||
if (!bits.length) return '';
|
||||
return `<span class="scr-htag" title="${escAttr(t('sortHealthTip'))}">${bits.join('')}</span>`;
|
||||
}
|
||||
function scrBlockHtml(it, isFirst, li, ji, extraClass) {
|
||||
const c = srcColor(it.src);
|
||||
const s = srcShort(it.src);
|
||||
@ -1206,6 +1242,7 @@ function scrBlockHtml(it, isFirst, li, ji, extraClass) {
|
||||
<span class="scr-ico">${esc(it.src === '*' ? '+' : s)}</span>
|
||||
<span class="scr-name">${esc(it.id)}<em class="scr-srcname">${esc(it.src === '*' ? t('kAnySrc') : it.src)}</em></span>
|
||||
${it.meta ? `<span class="scr-tag">${esc(quantBadge(it.meta.quota, it.meta.period, it.meta.hours))}</span>` : ''}
|
||||
${healthTag(it)}
|
||||
<span class="scr-x" title="${escAttr(t('kDelB2'))}" onclick="event.stopPropagation();scrDelSlot('${li}','${ji}')">×</span>
|
||||
<span class="scr-grip"><i></i><i></i><i></i></span>
|
||||
</div>`;
|
||||
@ -1521,7 +1558,7 @@ async function saveSort() {
|
||||
try {
|
||||
await persistAuto();
|
||||
sortState.origin = JSON.stringify(sortState.lanes);
|
||||
toast(t('sortSaved'));
|
||||
toast(t('sortSaved') + ' · ' + t('sortHealthReset'));
|
||||
} catch (e) { toast(e.message); }
|
||||
}
|
||||
|
||||
@ -1986,10 +2023,20 @@ function showCtx(x, y, items) {
|
||||
const r = w.getBoundingClientRect();
|
||||
w.style.left = Math.max(6, Math.min(x, window.innerWidth - r.width - 6)) + 'px';
|
||||
w.style.top = Math.max(6, Math.min(y, window.innerHeight - r.height - 6)) + 'px';
|
||||
setTimeout(() => document.addEventListener('click', hideCtx2, { once: true }), 10);
|
||||
}
|
||||
function hideCtx2() { hideCtx(); }
|
||||
function hideCtx() { if (ctxEl) { ctxEl.remove(); ctxEl = null; } }
|
||||
// Close the ctx menu on any primary click/press OUTSIDE the menu. Both
|
||||
// listeners run in the CAPTURE phase, so they fire even when the clicked
|
||||
// element stops propagation (priority blocks and key bricks call
|
||||
// stopPropagation in their own click handlers, which would otherwise keep the
|
||||
// menu open forever). Presses INSIDE the menu are left alone: the menu's own
|
||||
// click handler closes it after running the item action.
|
||||
document.addEventListener('click', e => {
|
||||
if (e.button === 0 && !(ctxEl && ctxEl.contains(e.target))) hideCtx();
|
||||
}, true);
|
||||
document.addEventListener('mousedown', e => {
|
||||
if (e.button === 0 && !(ctxEl && ctxEl.contains(e.target))) hideCtx();
|
||||
}, true);
|
||||
/* cross-canvas brick dragging */
|
||||
function bindBrickDrag(b) {
|
||||
b.addEventListener('dragstart', e => {
|
||||
|
||||
Reference in New Issue
Block a user