mirror of
https://gitcode.com/JianFeeeee/ModelRouter.git
synced 2026-09-20 17:07:59 +00:00
feat(types): pass through upstream cache tokens in TokenUsage
dsh displays cache-hit %, but llmsproxy dropped every upstream's cache fields — deepseek prompt_cache_hit_tokens, OpenAI prompt_tokens_details. cached_tokens, anthropic cache_read_input_tokens, gemini cachedContentTokenCount. Changes: - TokenUsage: add PromptTokensDetails (with CachedTokens) + PromptCacheHit/Miss - MarshalJSON: emit prompt_tokens_details.cached_tokens (OpenAI v2 standard) and prompt_cache_hit/miss_tokens (DeepSeek legacy) — dsh reads the former first, falls back to the latter - mergeUsage: preserve cache fields across stream chunks - standardSSEChunk: parse the upstream raw prompt_tokens_details too - deepseek.lua: forward prompt_cache_hit/miss_tokens + create prompt_tokens_details from them - openai.lua: forward prompt_tokens_details.cached_tokens and legacy prompt_cache_hit/miss_tokens; normalize legacy hits into the standard object so dsh sees them regardless of upstream format - anthropic.lua: map cache_read_input_tokens → prompt_tokens_details - gemini.lua: map cachedContentTokenCount → prompt_tokens_details
This commit is contained in:
@ -33,6 +33,22 @@ function adapter.transform_response(raw_body)
|
||||
unified.token_usage.prompt = resp.usage.prompt_tokens or 0
|
||||
unified.token_usage.completion = resp.usage.completion_tokens or 0
|
||||
unified.token_usage.total = resp.usage.total_tokens or 0
|
||||
-- Cache passthrough: OpenAI v2 prompt_tokens_details.cached_tokens
|
||||
-- and DeepSeek-legacy prompt_cache_hit/miss_tokens. dsh reads
|
||||
-- prompt_tokens_details.cached_tokens (falls back to the legacy
|
||||
-- standalone field), so both shapes reach clients.
|
||||
local hit = 0
|
||||
if type(resp.usage.prompt_tokens_details) == "table" and (resp.usage.prompt_tokens_details.cached_tokens or 0) > 0 then
|
||||
hit = resp.usage.prompt_tokens_details.cached_tokens
|
||||
unified.token_usage.prompt_tokens_details = { cached_tokens = hit }
|
||||
end
|
||||
if (resp.usage.prompt_cache_hit_tokens or 0) > 0 then
|
||||
unified.token_usage.prompt_cache_hit_tokens = resp.usage.prompt_cache_hit_tokens
|
||||
unified.token_usage.prompt_cache_miss_tokens = resp.usage.prompt_cache_miss_tokens or 0
|
||||
if hit == 0 then
|
||||
unified.token_usage.prompt_tokens_details = { cached_tokens = resp.usage.prompt_cache_hit_tokens }
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
if type(resp.choices) == "table" and #resp.choices > 0 then
|
||||
@ -77,6 +93,13 @@ function adapter.transform_stream_chunk(raw_chunk)
|
||||
completion = chunk.usage.completion_tokens or chunk.usage.completion or 0,
|
||||
total = chunk.usage.total_tokens or chunk.usage.total or 0,
|
||||
}
|
||||
if type(chunk.usage.prompt_tokens_details) == "table" and (chunk.usage.prompt_tokens_details.cached_tokens or 0) > 0 then
|
||||
uses.prompt_tokens_details = { cached_tokens = chunk.usage.prompt_tokens_details.cached_tokens }
|
||||
elseif (chunk.usage.prompt_cache_hit_tokens or 0) > 0 then
|
||||
uses.prompt_cache_hit_tokens = chunk.usage.prompt_cache_hit_tokens
|
||||
uses.prompt_cache_miss_tokens = chunk.usage.prompt_cache_miss_tokens or 0
|
||||
uses.prompt_tokens_details = { cached_tokens = chunk.usage.prompt_cache_hit_tokens }
|
||||
end
|
||||
end
|
||||
|
||||
if not chunk.choices or #chunk.choices == 0 then
|
||||
|
||||
Reference in New Issue
Block a user