Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
61 changes: 57 additions & 4 deletions crates/aisix-obs/src/metrics.rs
Original file line number Diff line number Diff line change
Expand Up @@ -42,11 +42,16 @@ pub const M_LLM_REQUESTS_TOTAL: &str = "aisix_llm_requests_total";
pub const M_LLM_REQUEST_DURATION: &str = "aisix_llm_request_duration_seconds";
pub const M_LLM_API_LATENCY: &str = "aisix_llm_api_latency_seconds";
pub const M_LLM_TTFT: &str = "aisix_llm_time_to_first_token_seconds";
/// Issue #890 req-4: token volume sliced by inbound client type only — a
/// Issue #890 req-4: token volume sliced by inbound client type — a
/// DEDICATED low-cardinality series so the client dimension never multiplies
/// the per-key `aisix_llm_*_tokens_total` families. `client_type` is
/// normalised to a bounded allowlist by [`client_type_from_user_agent`]; the
/// raw user-agent + client version stay in logs / `UsageEvent`, never here.
/// AISIX-Cloud#1044 adds a `model` label (the requested logical model, same
/// value as the `aisix_llm_*` families' `model`) so the series answers
/// "which models is each client spending tokens on". The label set stays
/// client_type × model × token_type — per-key/team/user dimensions belong to
/// the `aisix_llm_*_tokens_total` families (or UsageEvent/logs), never here.
pub const M_LLM_TOKENS_BY_CLIENT_TOTAL: &str = "aisix_llm_tokens_by_client_total";
pub const M_PROXY_IN_FLIGHT: &str = "aisix_proxy_in_flight_requests";
pub const M_PROXY_REQUESTS_TOTAL: &str = "aisix_proxy_requests_total";
Expand Down Expand Up @@ -512,6 +517,13 @@ impl Metrics {
/// `&'static str` from [`client_type_from_user_agent`] so cardinality is
/// bounded; zero dims are skipped to keep the series sparse.
///
/// `model` (AISIX-Cloud#1044) is the requested logical model — callers
/// MUST pass the same value they put in [`UsageLabels::model`] (or its
/// endpoint's equivalent), never the raw client string of an unresolved
/// request nor the routed `upstream_model`, so the label stays bounded by
/// the configured model set and joins cleanly with the `aisix_llm_*`
/// families.
///
/// `total_tokens` is the caller's canonical cache-inclusive total
/// (`input + output + Anthropic cache_creation/cache_read`), emitted under
/// `token_type="total"` (AISIX-Cloud#1002). It is passed in — not derived
Expand All @@ -522,6 +534,7 @@ impl Metrics {
pub fn record_llm_tokens_by_client(
&self,
client_type: &'static str,
model: &str,
input_tokens: u64,
output_tokens: u64,
total_tokens: u64,
Expand All @@ -534,6 +547,7 @@ impl Metrics {
metrics::counter!(
M_LLM_TOKENS_BY_CLIENT_TOTAL,
"client_type" => client_type,
"model" => model.to_string(),
"token_type" => "input",
)
.increment(input_tokens);
Expand All @@ -542,6 +556,7 @@ impl Metrics {
metrics::counter!(
M_LLM_TOKENS_BY_CLIENT_TOTAL,
"client_type" => client_type,
"model" => model.to_string(),
"token_type" => "output",
)
.increment(output_tokens);
Expand All @@ -550,6 +565,7 @@ impl Metrics {
metrics::counter!(
M_LLM_TOKENS_BY_CLIENT_TOTAL,
"client_type" => client_type,
"model" => model.to_string(),
"token_type" => "total",
)
.increment(total_tokens);
Expand Down Expand Up @@ -1326,10 +1342,10 @@ mod tests {
let m = Metrics::new(false);
// The caller's canonical total is cache-inclusive, so it can exceed
// input+output: 155 = 100 + 40 + 15 cache tokens (#1002).
m.record_llm_tokens_by_client("openai-python", 100, 40, 155);
m.record_llm_tokens_by_client("openai-python", 10, 0, 10);
m.record_llm_tokens_by_client("openai-python", "gpt-4o", 100, 40, 155);
m.record_llm_tokens_by_client("openai-python", "gpt-4o", 10, 0, 10);
// All-zero is a no-op (keeps the series sparse).
m.record_llm_tokens_by_client("curl", 0, 0, 0);
m.record_llm_tokens_by_client("curl", "gpt-4o", 0, 0, 0);
let rendered = m.render();
assert!(rendered.contains(M_LLM_TOKENS_BY_CLIENT_TOTAL));
assert!(rendered.contains("client_type=\"openai-python\""));
Expand All @@ -1342,11 +1358,48 @@ mod tests {
.lines()
.any(|l| l.starts_with("aisix_llm_tokens_by_client_total{")
&& l.contains("token_type=\"total\"")
&& l.contains("model=\"gpt-4o\"")
&& l.trim_end().ends_with(" 165")));
// The all-zero curl call recorded nothing.
assert!(!rendered.contains("client_type=\"curl\""));
}

#[test]
fn tokens_by_client_splits_series_per_model() {
// AISIX-Cloud#1044: one client type spending on two models must
// produce two independent series per token_type, and every series
// must carry the model label.
let m = Metrics::new(false);
m.record_llm_tokens_by_client("claude-code", "claude-sonnet", 100, 60, 160);
m.record_llm_tokens_by_client("claude-code", "claude-haiku", 30, 10, 40);
let rendered = m.render();
let series: Vec<&str> = rendered
.lines()
.filter(|l| l.starts_with("aisix_llm_tokens_by_client_total{"))
.collect();
// 2 models × 3 token types, all under the same client_type.
assert_eq!(series.len(), 6);
assert!(series
.iter()
.all(|l| l.contains("client_type=\"claude-code\"") && l.contains("model=")));
let value_of = |model: &str, token_type: &str| {
series
.iter()
.find(|l| {
l.contains(&format!("model=\"{model}\""))
&& l.contains(&format!("token_type=\"{token_type}\""))
})
.and_then(|l| l.trim_end().rsplit(' ').next())
.map(|v| v.parse::<u64>().unwrap())
};
assert_eq!(value_of("claude-sonnet", "input"), Some(100));
assert_eq!(value_of("claude-sonnet", "output"), Some(60));
assert_eq!(value_of("claude-sonnet", "total"), Some(160));
assert_eq!(value_of("claude-haiku", "input"), Some(30));
assert_eq!(value_of("claude-haiku", "output"), Some(10));
assert_eq!(value_of("claude-haiku", "total"), Some(40));
}

#[test]
fn client_type_from_user_agent_normalises_to_allowlist() {
// Known SDKs/tools normalise to a stable bounded label.
Expand Down
6 changes: 6 additions & 0 deletions crates/aisix-proxy/src/chat.rs
Original file line number Diff line number Diff line change
Expand Up @@ -1523,8 +1523,11 @@ async fn dispatch(
// #1002: comp.total_tokens is the cache-inclusive total (an
// Anthropic upstream bridged to an OpenAI-shape client folds
// cache tokens into total_tokens per #679).
// AISIX-Cloud#1044: same requested logical model as the
// UsageLabels above.
metrics_for_stream.record_llm_tokens_by_client(
client_type_for_metrics,
&model_for_metrics,
u64::from(comp.prompt_tokens),
u64::from(comp.completion_tokens),
comp.total_tokens,
Expand Down Expand Up @@ -3107,8 +3110,11 @@ fn record_success(
// streaming tokens arrive in the SSE on_complete and are recorded there).
// No-op when all counts are zero (e.g. the streaming branch here).
// #1002: s.total_tokens is the cache-inclusive canonical total.
// AISIX-Cloud#1044: `model` is the same requested logical model recorded
// on the UsageLabels above.
metrics.record_llm_tokens_by_client(
client_type,
model,
s.prompt_tokens.unwrap_or(0),
s.completion_tokens.unwrap_or(0),
s.total_tokens.unwrap_or(0),
Expand Down
2 changes: 2 additions & 0 deletions crates/aisix-proxy/src/messages.rs
Original file line number Diff line number Diff line change
Expand Up @@ -2456,8 +2456,10 @@ fn emit_anthropic_usage_event(
// #890 req-4: token volume by inbound client type (covers streaming and
// non-streaming — every /v1/messages usage event flows through here).
// #1002: total_tokens_all folds in the Anthropic cache counters.
// AISIX-Cloud#1044: same requested logical model as the UsageLabels above.
state.metrics.record_llm_tokens_by_client(
aisix_obs::client_type_from_user_agent(&client.user_agent),
model,
u64::from(metrics.prompt_tokens),
u64::from(metrics.completion_tokens),
total_tokens_all,
Expand Down
22 changes: 22 additions & 0 deletions crates/aisix-proxy/src/responses.rs
Original file line number Diff line number Diff line change
Expand Up @@ -2479,6 +2479,28 @@ fn emit_usage_event(
state
.otlp_fan_out
.fan_out(&event, content, exporters.iter().map(|e| &e.value));
// AISIX-Cloud#1044: token volume by inbound client type × model. Codex
// traffic arrives on /v1/responses, so leaving this endpoint out of the
// by-client series made an allowlisted client invisible in it. All three
// usage-bearing paths (non-streaming, verbatim streaming, bridge
// streaming) funnel through here. `requested_model` resolved at dispatch
// on every path that reaches this emit, so the label is bounded by the
// configured model set. The per-key `aisix_llm_*_tokens_total` family
// intentionally stays chat/messages-scoped (cross-API audit #646-652).
// #1002: cache-inclusive total via the shared helper — cache counters are
// non-zero only on the #825 Anthropic bridge path.
state.metrics.record_llm_tokens_by_client(
aisix_obs::client_type_from_user_agent(&client.user_agent),
requested_model,
u64::from(usage.prompt_tokens),
u64::from(usage.completion_tokens),
total_tokens_with_cache(
usage.prompt_tokens,
usage.completion_tokens,
usage.cache_creation_tokens,
usage.cache_read_tokens,
),
);
}

/// Emit a zero-token `UsageEvent` for a failed / pre-dispatch attempt
Expand Down
Loading
Loading