fix(proxy): correct usage accounting on format-conversion paths

Audited all proxy format-conversion paths (Chat<->Message, Chat<->Response,
Gemini<->Message) for usage/cache metering. Five issues found and fixed.
The dedup mechanism (request_id PK, proxy/session source isolation) is
untouched, so no double-counting is introduced.

- A (Claude + openai_chat, streaming): inject stream_options.include_usage
  so OpenAI-compatible upstreams emit usage in the SSE tail. Without it the
  converted Anthropic message_delta was all-zero and the whole request's
  input/output/cache was dropped. Same root cause as the already-fixed
  Codex Chat path; the injection is extracted into a shared helper
  (transform::inject_openai_stream_include_usage) reused by both paths.

- C (Claude + gemini_native): subtract cachedContentTokenCount from
  input_tokens in build_anthropic_usage so input becomes fresh input
  (Anthropic semantics). Previously the cache-hit tokens were billed twice
  because this path meters as app_type="claude" (input_includes_cache_read
  = false) while Gemini's promptTokenCount includes the cache.

- D (Codex + openai_chat, streaming): gate log_usage on
  has_billable_tokens() to skip the synthetic all-zero usage the converter
  emits when a non-compliant upstream omits usage, preventing empty-row
  request-count inflation.

- P2 (from_claude_stream_events): use has_billable_tokens() for the return
  gate instead of input>0||output>0, so a fully-cached streamed request
  (cache_read>0, input==output==0) is still recorded. Affects all
  Claude-streaming paths, not just Gemini.

- P3 (Codex Chat->Responses, non-streaming): apply the same
  has_billable_tokens() filter the streaming branch got, since the
  synthesized all-zero usage makes from_codex_response return Some and
  bypass the `if let Some` guard.

Add TokenUsage::has_billable_tokens() as the unified predicate. New tests
cover include_usage injection, gemini input subtraction, the gate itself,
cache-only stream recording, and synthetic all-zero codex usage.
Full lib suite: 1569 passed.
This commit is contained in:
Jason
2026-06-09 13:15:13 +08:00
parent 05bc14e82b
commit 36a103bbe4
6 changed files with 188 additions and 25 deletions
+82 -1
View File
@@ -37,6 +37,18 @@ impl TokenUsage {
.map(|mid| format!("{SESSION_REQUEST_ID_PREFIX}{mid}"))
.unwrap_or_else(|| uuid::Uuid::new_v4().to_string())
}
/// 是否产生了任一计费维度的 token。
///
/// 用于在写入前过滤全 0 的空 usage:当 OpenAI 兼容上游在流式下省略 usage 时,
/// 转换器会合成一个全 0 的终止事件,若无 message_id 则 `dedup_request_id`
/// 退化为随机 UUID,导致每笔请求插入一条无意义的空行、虚增请求数。
pub fn has_billable_tokens(&self) -> bool {
self.input_tokens > 0
|| self.output_tokens > 0
|| self.cache_read_tokens > 0
|| self.cache_creation_tokens > 0
}
}
/// API 类型
@@ -185,7 +197,11 @@ impl TokenUsage {
}
}
if usage.input_tokens > 0 || usage.output_tokens > 0 {
// 用 has_billable_tokens 而非仅看 input/output:完全缓存命中、无输出的流式请求
// input==0 && output==0 但 cache_read>0)是真实的 cache-read 计费,必须保留。
// Gemini→Anthropic 路径在 input 改为 fresh(promptTokenCount - cachedContentTokenCount)
// 后尤其会出现这种全缓存场景;旧 gate 会把它当成"无 usage"丢弃。
if usage.has_billable_tokens() {
usage.model = model;
usage.message_id = message_id;
Some(usage)
@@ -522,6 +538,71 @@ mod tests {
assert_eq!(usage.model, Some("claude-sonnet-4-20250514".to_string()));
}
#[test]
fn test_has_billable_tokens_gates_empty_usage() {
// 全 0 usage(如上游省略 usage 时合成的全 0 终止事件)不应计费——
// 这是 Codex 流式空行多记修复(D)的闸门依据。
assert!(!TokenUsage::default().has_billable_tokens());
// 仅有 cache_read 也属于真实计费 token,必须计入。
let only_cache = TokenUsage {
cache_read_tokens: 100,
..Default::default()
};
assert!(only_cache.has_billable_tokens());
let normal = TokenUsage {
input_tokens: 10,
output_tokens: 5,
..Default::default()
};
assert!(normal.has_billable_tokens());
}
#[test]
fn test_claude_stream_cache_only_request_is_recorded() {
// P2 回归:完全缓存命中、无输出的流式请求(input==0 && output==0 但 cache_read>0
// 是真实计费,必须保留——旧 gate `input>0 || output>0` 会把它丢弃。
let events = vec![
json!({
"type": "message_start",
"message": {
"id": "msg_cacheonly",
"model": "claude-opus-4-8",
"usage": {
"input_tokens": 0,
"cache_read_input_tokens": 50000,
"cache_creation_input_tokens": 0
}
}
}),
json!({
"type": "message_delta",
"usage": { "output_tokens": 0 }
}),
];
let usage = TokenUsage::from_claude_stream_events(&events)
.expect("cache-only 流式请求必须被记录,不能被 input/output gate 丢弃");
assert_eq!(usage.input_tokens, 0);
assert_eq!(usage.output_tokens, 0);
assert_eq!(usage.cache_read_tokens, 50000);
assert_eq!(usage.message_id, Some("msg_cacheonly".to_string()));
}
#[test]
fn test_codex_response_auto_returns_some_for_synthetic_all_zero() {
// P3 回归:上游非流式 Chat 省略 usage 时转换器合成的全 0 usagefrom_codex_response_auto
// 仍返回 Some(字段存在、无 positivity check)——证明 handlers 必须用 has_billable_tokens
// 闸门才能挡住空行,单靠 `if let Some` 不够。
let synthetic = json!({
"usage": { "input_tokens": 0, "output_tokens": 0, "total_tokens": 0 }
});
let usage = TokenUsage::from_codex_response_auto(&synthetic)
.expect("全 0 usage 字段存在时 from_codex_response_auto 返回 Some");
assert!(
!usage.has_billable_tokens(),
"全 0 usage 必须被 has_billable_tokens 判为非计费,由 handlers 闸门跳过"
);
}
#[test]
fn test_claude_response_parsing_no_model() {
let response = json!({