mirror of
https://github.com/farion1231/cc-switch.git
synced 2026-07-27 08:14:33 +08:00
fix(proxy): correct usage accounting on format-conversion paths
Audited all proxy format-conversion paths (Chat<->Message, Chat<->Response, Gemini<->Message) for usage/cache metering. Five issues found and fixed. The dedup mechanism (request_id PK, proxy/session source isolation) is untouched, so no double-counting is introduced. - A (Claude + openai_chat, streaming): inject stream_options.include_usage so OpenAI-compatible upstreams emit usage in the SSE tail. Without it the converted Anthropic message_delta was all-zero and the whole request's input/output/cache was dropped. Same root cause as the already-fixed Codex Chat path; the injection is extracted into a shared helper (transform::inject_openai_stream_include_usage) reused by both paths. - C (Claude + gemini_native): subtract cachedContentTokenCount from input_tokens in build_anthropic_usage so input becomes fresh input (Anthropic semantics). Previously the cache-hit tokens were billed twice because this path meters as app_type="claude" (input_includes_cache_read = false) while Gemini's promptTokenCount includes the cache. - D (Codex + openai_chat, streaming): gate log_usage on has_billable_tokens() to skip the synthetic all-zero usage the converter emits when a non-compliant upstream omits usage, preventing empty-row request-count inflation. - P2 (from_claude_stream_events): use has_billable_tokens() for the return gate instead of input>0||output>0, so a fully-cached streamed request (cache_read>0, input==output==0) is still recorded. Affects all Claude-streaming paths, not just Gemini. - P3 (Codex Chat->Responses, non-streaming): apply the same has_billable_tokens() filter the streaming branch got, since the synthesized all-zero usage makes from_codex_response return Some and bypass the `if let Some` guard. Add TokenUsage::has_billable_tokens() as the unified predicate. New tests cover include_usage injection, gemini input subtraction, the gate itself, cache-only stream recording, and synthetic all-zero codex usage. Full lib suite: 1569 passed.
This commit is contained in:
@@ -37,6 +37,18 @@ impl TokenUsage {
|
||||
.map(|mid| format!("{SESSION_REQUEST_ID_PREFIX}{mid}"))
|
||||
.unwrap_or_else(|| uuid::Uuid::new_v4().to_string())
|
||||
}
|
||||
|
||||
/// 是否产生了任一计费维度的 token。
|
||||
///
|
||||
/// 用于在写入前过滤全 0 的空 usage:当 OpenAI 兼容上游在流式下省略 usage 时,
|
||||
/// 转换器会合成一个全 0 的终止事件,若无 message_id 则 `dedup_request_id`
|
||||
/// 退化为随机 UUID,导致每笔请求插入一条无意义的空行、虚增请求数。
|
||||
pub fn has_billable_tokens(&self) -> bool {
|
||||
self.input_tokens > 0
|
||||
|| self.output_tokens > 0
|
||||
|| self.cache_read_tokens > 0
|
||||
|| self.cache_creation_tokens > 0
|
||||
}
|
||||
}
|
||||
|
||||
/// API 类型
|
||||
@@ -185,7 +197,11 @@ impl TokenUsage {
|
||||
}
|
||||
}
|
||||
|
||||
if usage.input_tokens > 0 || usage.output_tokens > 0 {
|
||||
// 用 has_billable_tokens 而非仅看 input/output:完全缓存命中、无输出的流式请求
|
||||
// (input==0 && output==0 但 cache_read>0)是真实的 cache-read 计费,必须保留。
|
||||
// Gemini→Anthropic 路径在 input 改为 fresh(promptTokenCount - cachedContentTokenCount)
|
||||
// 后尤其会出现这种全缓存场景;旧 gate 会把它当成"无 usage"丢弃。
|
||||
if usage.has_billable_tokens() {
|
||||
usage.model = model;
|
||||
usage.message_id = message_id;
|
||||
Some(usage)
|
||||
@@ -522,6 +538,71 @@ mod tests {
|
||||
assert_eq!(usage.model, Some("claude-sonnet-4-20250514".to_string()));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_has_billable_tokens_gates_empty_usage() {
|
||||
// 全 0 usage(如上游省略 usage 时合成的全 0 终止事件)不应计费——
|
||||
// 这是 Codex 流式空行多记修复(D)的闸门依据。
|
||||
assert!(!TokenUsage::default().has_billable_tokens());
|
||||
// 仅有 cache_read 也属于真实计费 token,必须计入。
|
||||
let only_cache = TokenUsage {
|
||||
cache_read_tokens: 100,
|
||||
..Default::default()
|
||||
};
|
||||
assert!(only_cache.has_billable_tokens());
|
||||
let normal = TokenUsage {
|
||||
input_tokens: 10,
|
||||
output_tokens: 5,
|
||||
..Default::default()
|
||||
};
|
||||
assert!(normal.has_billable_tokens());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_claude_stream_cache_only_request_is_recorded() {
|
||||
// P2 回归:完全缓存命中、无输出的流式请求(input==0 && output==0 但 cache_read>0)
|
||||
// 是真实计费,必须保留——旧 gate `input>0 || output>0` 会把它丢弃。
|
||||
let events = vec![
|
||||
json!({
|
||||
"type": "message_start",
|
||||
"message": {
|
||||
"id": "msg_cacheonly",
|
||||
"model": "claude-opus-4-8",
|
||||
"usage": {
|
||||
"input_tokens": 0,
|
||||
"cache_read_input_tokens": 50000,
|
||||
"cache_creation_input_tokens": 0
|
||||
}
|
||||
}
|
||||
}),
|
||||
json!({
|
||||
"type": "message_delta",
|
||||
"usage": { "output_tokens": 0 }
|
||||
}),
|
||||
];
|
||||
let usage = TokenUsage::from_claude_stream_events(&events)
|
||||
.expect("cache-only 流式请求必须被记录,不能被 input/output gate 丢弃");
|
||||
assert_eq!(usage.input_tokens, 0);
|
||||
assert_eq!(usage.output_tokens, 0);
|
||||
assert_eq!(usage.cache_read_tokens, 50000);
|
||||
assert_eq!(usage.message_id, Some("msg_cacheonly".to_string()));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_codex_response_auto_returns_some_for_synthetic_all_zero() {
|
||||
// P3 回归:上游非流式 Chat 省略 usage 时转换器合成的全 0 usage,from_codex_response_auto
|
||||
// 仍返回 Some(字段存在、无 positivity check)——证明 handlers 必须用 has_billable_tokens
|
||||
// 闸门才能挡住空行,单靠 `if let Some` 不够。
|
||||
let synthetic = json!({
|
||||
"usage": { "input_tokens": 0, "output_tokens": 0, "total_tokens": 0 }
|
||||
});
|
||||
let usage = TokenUsage::from_codex_response_auto(&synthetic)
|
||||
.expect("全 0 usage 字段存在时 from_codex_response_auto 返回 Some");
|
||||
assert!(
|
||||
!usage.has_billable_tokens(),
|
||||
"全 0 usage 必须被 has_billable_tokens 判为非计费,由 handlers 闸门跳过"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_claude_response_parsing_no_model() {
|
||||
let response = json!({
|
||||
|
||||
Reference in New Issue
Block a user