feat(types): pass through upstream cache tokens in TokenUsage

dsh displays cache-hit %, but llmsproxy dropped every upstream's cache
fields — deepseek prompt_cache_hit_tokens, OpenAI prompt_tokens_details.
cached_tokens, anthropic cache_read_input_tokens, gemini cachedContentTokenCount.

Changes:
- TokenUsage: add PromptTokensDetails (with CachedTokens) + PromptCacheHit/Miss
- MarshalJSON: emit prompt_tokens_details.cached_tokens (OpenAI v2 standard)
  and prompt_cache_hit/miss_tokens (DeepSeek legacy) — dsh reads the former
  first, falls back to the latter
- mergeUsage: preserve cache fields across stream chunks
- standardSSEChunk: parse the upstream raw prompt_tokens_details too
- deepseek.lua: forward prompt_cache_hit/miss_tokens + create
  prompt_tokens_details from them
- openai.lua: forward prompt_tokens_details.cached_tokens and legacy
  prompt_cache_hit/miss_tokens; normalize legacy hits into the standard
  object so dsh sees them regardless of upstream format
- anthropic.lua: map cache_read_input_tokens → prompt_tokens_details
- gemini.lua: map cachedContentTokenCount → prompt_tokens_details
This commit is contained in:
dev
2026-08-25 07:57:32 +08:00
parent 334b984c25
commit 045ecf47bc
7 changed files with 112 additions and 21 deletions

View File

@ -82,6 +82,13 @@ function adapter.transform_response(raw_body)
unified.token_usage.prompt = resp.usage.input_tokens or 0
unified.token_usage.completion = resp.usage.output_tokens or 0
unified.token_usage.total = (resp.usage.input_tokens or 0) + (resp.usage.output_tokens or 0)
-- Anthropic reports cache_read_input_tokens; normalize into
-- OpenAI-standard prompt_tokens_details.cached_tokens so clients
-- (dsh) see the cache hit count.
local cacheRead = resp.usage.cache_read_input_tokens or 0
if cacheRead > 0 then
unified.token_usage.prompt_tokens_details = { cached_tokens = cacheRead }
end
end
if resp.content and #resp.content > 0 then
@ -107,6 +114,10 @@ function adapter.transform_stream_chunk(raw_chunk)
local c = u.output_tokens or 0
if p > 0 or c > 0 then
uses = { prompt = p, completion = c, total = p + c }
local cacheRead = u.cache_read_input_tokens or 0
if cacheRead > 0 then
uses.prompt_tokens_details = { cached_tokens = cacheRead }
end
end
end
if uses ~= nil then

View File

@ -49,6 +49,11 @@ function adapter.transform_response(raw_body)
unified.token_usage.prompt = resp.usage.prompt_tokens or 0
unified.token_usage.completion = resp.usage.completion_tokens or 0
unified.token_usage.total = resp.usage.total_tokens or 0
unified.token_usage.prompt_cache_hit_tokens = resp.usage.prompt_cache_hit_tokens or 0
unified.token_usage.prompt_cache_miss_tokens = resp.usage.prompt_cache_miss_tokens or 0
if resp.usage.prompt_tokens_details and resp.usage.prompt_tokens_details.cached_tokens then
unified.token_usage.prompt_tokens_details = { cached_tokens = resp.usage.prompt_tokens_details.cached_tokens }
end
end
if type(resp.choices) == "table" and #resp.choices > 0 then
@ -92,7 +97,12 @@ function adapter.transform_stream_chunk(raw_chunk)
prompt = chunk.usage.prompt_tokens or chunk.usage.prompt or 0,
completion = chunk.usage.completion_tokens or chunk.usage.completion or 0,
total = chunk.usage.total_tokens or chunk.usage.total or 0,
prompt_cache_hit_tokens = chunk.usage.prompt_cache_hit_tokens or 0,
prompt_cache_miss_tokens = chunk.usage.prompt_cache_miss_tokens or 0,
}
if chunk.usage.prompt_tokens_details and chunk.usage.prompt_tokens_details.cached_tokens then
uses.prompt_tokens_details = { cached_tokens = chunk.usage.prompt_tokens_details.cached_tokens }
end
end
if not chunk.choices or #chunk.choices == 0 then

View File

@ -100,6 +100,10 @@ function adapter.transform_stream_chunk(raw_chunk)
local t = chunk.usageMetadata.totalTokenCount or 0
if p > 0 or c > 0 or t > 0 then
uses = { prompt = p, completion = c, total = t }
local cacheRead = chunk.usageMetadata.cachedContentTokenCount or 0
if cacheRead > 0 then
uses.prompt_tokens_details = { cached_tokens = cacheRead }
end
end
end

View File

@ -33,6 +33,22 @@ function adapter.transform_response(raw_body)
unified.token_usage.prompt = resp.usage.prompt_tokens or 0
unified.token_usage.completion = resp.usage.completion_tokens or 0
unified.token_usage.total = resp.usage.total_tokens or 0
-- Cache passthrough: OpenAI v2 prompt_tokens_details.cached_tokens
-- and DeepSeek-legacy prompt_cache_hit/miss_tokens. dsh reads
-- prompt_tokens_details.cached_tokens (falls back to the legacy
-- standalone field), so both shapes reach clients.
local hit = 0
if type(resp.usage.prompt_tokens_details) == "table" and (resp.usage.prompt_tokens_details.cached_tokens or 0) > 0 then
hit = resp.usage.prompt_tokens_details.cached_tokens
unified.token_usage.prompt_tokens_details = { cached_tokens = hit }
end
if (resp.usage.prompt_cache_hit_tokens or 0) > 0 then
unified.token_usage.prompt_cache_hit_tokens = resp.usage.prompt_cache_hit_tokens
unified.token_usage.prompt_cache_miss_tokens = resp.usage.prompt_cache_miss_tokens or 0
if hit == 0 then
unified.token_usage.prompt_tokens_details = { cached_tokens = resp.usage.prompt_cache_hit_tokens }
end
end
end
if type(resp.choices) == "table" and #resp.choices > 0 then
@ -77,6 +93,13 @@ function adapter.transform_stream_chunk(raw_chunk)
completion = chunk.usage.completion_tokens or chunk.usage.completion or 0,
total = chunk.usage.total_tokens or chunk.usage.total or 0,
}
if type(chunk.usage.prompt_tokens_details) == "table" and (chunk.usage.prompt_tokens_details.cached_tokens or 0) > 0 then
uses.prompt_tokens_details = { cached_tokens = chunk.usage.prompt_tokens_details.cached_tokens }
elseif (chunk.usage.prompt_cache_hit_tokens or 0) > 0 then
uses.prompt_cache_hit_tokens = chunk.usage.prompt_cache_hit_tokens
uses.prompt_cache_miss_tokens = chunk.usage.prompt_cache_miss_tokens or 0
uses.prompt_tokens_details = { cached_tokens = chunk.usage.prompt_cache_hit_tokens }
end
end
if not chunk.choices or #chunk.choices == 0 then