From 440f831228a6f2f4c18a8a01a54585dd1dc958cc Mon Sep 17 00:00:00 2001 From: Ming Wen Date: Wed, 30 Sep 2026 15:02:49 +0800 Subject: [PATCH 01/16] fix(telemetry): price wildcard models by resolved target --- crates/aisix-obs/src/usage.rs | 13 ++ crates/aisix-proxy/src/jobs.rs | 23 ++- crates/aisix-proxy/src/usage_attr.rs | 165 ++++++++++++++++++ .../wildcard-pricing-telemetry-e2e.test.ts | 156 +++++++++++++++++ 4 files changed, 352 insertions(+), 5 deletions(-) create mode 100644 tests/e2e/src/cases/wildcard-pricing-telemetry-e2e.test.ts diff --git a/crates/aisix-obs/src/usage.rs b/crates/aisix-obs/src/usage.rs index 4aeb839f..d97dd28a 100644 --- a/crates/aisix-obs/src/usage.rs +++ b/crates/aisix-obs/src/usage.rs @@ -112,6 +112,17 @@ pub struct UsageEvent { #[serde(default, skip_serializing_if = "String::is_empty")] pub requested_model: String, + /// Concrete upstream model selected through a wildcard upstream template. + /// + /// `model_id` remains the configured wildcard row so policy and + /// attribution remain stable, while this value is the provider model name + /// that was actually dispatched. The control plane uses it only when the + /// configured row's `model_name` contains `*`, to look up its catalog + /// price; it is otherwise absent so an alias over a fixed upstream can + /// never override its configured pricing identity through telemetry. + #[serde(default, skip_serializing_if = "String::is_empty")] + pub resolved_pricing_model: String, + pub prompt_tokens: u32, pub completion_tokens: u32, @@ -1596,6 +1607,7 @@ mod tests { model_id: "mod-uuid".into(), api_key_id: "ak-uuid".into(), requested_model: "smart-group".into(), + resolved_pricing_model: "gpt-4o-2024-08-06".into(), prompt_tokens: 12, completion_tokens: 34, upstream_latency_ms: 56, @@ -1610,6 +1622,7 @@ mod tests { // AISIX-Cloud#790: the client-sent alias rides next to model_id // so the dashboard can show the group a routed request used. assert!(json.contains(r#""requested_model":"smart-group""#)); + assert!(json.contains(r#""resolved_pricing_model":"gpt-4o-2024-08-06""#)); assert!(json.contains(r#""prompt_tokens":12"#)); assert!(json.contains(r#""completion_tokens":34"#)); assert!(json.contains(r#""guardrail_blocked":false"#)); diff --git a/crates/aisix-proxy/src/jobs.rs b/crates/aisix-proxy/src/jobs.rs index dd0740b0..0f414d8d 100644 --- a/crates/aisix-proxy/src/jobs.rs +++ b/crates/aisix-proxy/src/jobs.rs @@ -2028,7 +2028,9 @@ async fn attribute_batch_usage( } let body = resp.bytes().await.map_err(|e| e.to_string())?; - // Aggregate per provider-billed model (`response.body.model`). + // Keep the provider-reported model only as diagnostic version data. + // A batch completion has no persisted per-line dispatch identity, so an + // upstream response field must not select a wildcard catalog price. #[derive(Default)] struct Agg { prompt: u64, @@ -2529,11 +2531,18 @@ mod tests { let snap = AisixSnapshot::new(); snap.provider_keys.insert(openai_pk(PK_A, &upstream.uri())); - snap.models.insert(model("m-a", "jobs-a", PK_A)); + let wildcard: Model = serde_json::from_value(serde_json::json!({ + "display_name": "jobs/*", + "provider": "openai", + "model_name": "*", + "provider_key_id": PK_A, + })) + .unwrap(); + snap.models.insert(ResourceEntry::new("m-a", wildcard, 1)); snap.apikeys.insert(apikey_entry(&["*"])); let (app, mut rx) = build_app_with_sink(snap); - let encoded = encode_routed_id("batch_1", "jobs-a"); + let encoded = encode_routed_id("batch_1", "jobs/*"); let mk_req = || { Request::builder() .method("GET") @@ -2550,7 +2559,7 @@ mod tests { let v: Value = serde_json::from_slice(&bytes).unwrap(); assert_eq!( decode_routed_id(v["output_file_id"].as_str().unwrap()), - Some(("file-out".to_string(), "jobs-a".to_string())) + Some(("file-out".to_string(), "jobs/*".to_string())) ); // Two events expected: the zero-token management event plus ONE @@ -2585,7 +2594,11 @@ mod tests { assert_eq!(agg.completion_tokens, 8); assert_eq!(agg.cached_prompt_tokens, 2); assert_eq!(agg.provider_model_version, "gpt-4o-2024-08-06"); - assert_eq!(agg.requested_model, "jobs-a"); + assert_eq!(agg.requested_model, "jobs/*"); + assert!( + agg.resolved_pricing_model.is_empty(), + "an upstream batch output model must not select wildcard pricing" + ); // Second retrieve: management event only — the attribution is // process-deduped. diff --git a/crates/aisix-proxy/src/usage_attr.rs b/crates/aisix-proxy/src/usage_attr.rs index 004fdbb8..1f62b18b 100644 --- a/crates/aisix-proxy/src/usage_attr.rs +++ b/crates/aisix-proxy/src/usage_attr.rs @@ -439,6 +439,53 @@ pub(crate) fn metric_model_label_pair<'a>( } } +/// The actual upstream model name a wildcard upstream template resolved for +/// this request, if that row still exists in the same snapshot as the emitted +/// event. +/// +/// `model_id` deliberately stays on the configured row: rate limits, +/// policy, and historical attribution all key on it. Pricing is the one +/// consumer that needs the concrete provider model name, and only a row whose +/// configured `model_name` is a template may use it. A wildcard display alias +/// over one fixed upstream still prices by that fixed configuration. The name +/// comes from request-local dispatch attribution, never from the +/// caller-controlled `requested_model` or an upstream response field. +pub(crate) fn wildcard_resolved_pricing_model<'a>( + snap: &AisixSnapshot, + model_id: &str, + upstream_model: &'a str, +) -> Option<&'a str> { + if upstream_model.is_empty() { + return None; + } + let entry = snap.models.get_by_id(model_id)?; + let model = &entry.value; + model + .model_name + .as_deref() + .is_some_and(|name| name.contains('*')) + .then_some(upstream_model) +} + +/// Fill the optional DP-to-CP wildcard-pricing identity at the one usage +/// emission chokepoint. A cache hit has no dispatched target; its +/// `upstream_model` is the cached entry's static mapping, not a concrete +/// pricing identity. Detached gateway work has no caller attribution and +/// therefore cannot accidentally price itself as the parent request. +fn apply_wildcard_pricing_model(snap: &AisixSnapshot, event: &mut UsageEvent) { + let Some(resolved) = crate::attribution::current() else { + return; + }; + if resolved.cache_hit_layer.is_some() { + return; + } + if let Some(model) = + wildcard_resolved_pricing_model(snap, &event.model_id, &resolved.upstream_model) + { + event.resolved_pricing_model = model.to_string(); + } +} + /// Stamp the five per-PK attribution fields onto an in-progress UsageEvent, /// sanitising the operator-controlled tag strings (control-char strip + length /// cap) before they hit the wire. One source of truth for the mapping so the @@ -958,6 +1005,7 @@ pub(crate) fn emit_usage( if terminal && event.guardrail_blocked { state.metrics.record_guardrail_blocked_request(); } + apply_wildcard_pricing_model(snap, &mut event); let emission = trace.map(|bundle| { event.trace_id = bundle.trace_id_hex(); bundle.emission( @@ -1040,6 +1088,123 @@ mod tests { assert_eq!((m.as_ref(), u.as_ref()), ("no-such/model", "raw-upstream")); } + #[test] + fn wildcard_pricing_model_uses_only_the_configured_wildcard_row() { + use aisix_core::resource::ResourceEntry; + use aisix_core::snapshot::ResourceTable; + + let table = ResourceTable::default(); + let wildcard: aisix_core::Model = serde_json::from_value(serde_json::json!({ + "display_name": "openrouter/*", + "provider": "openai", + "model_name": "*", + "provider_key_id": "pk-1", + })) + .unwrap(); + let exact: aisix_core::Model = serde_json::from_value(serde_json::json!({ + "display_name": "exact-model", + "provider": "openai", + "model_name": "gpt-4o", + "provider_key_id": "pk-1", + })) + .unwrap(); + let wildcard_alias_fixed_upstream: aisix_core::Model = + serde_json::from_value(serde_json::json!({ + "display_name": "fixed/*", + "provider": "openai", + "model_name": "gpt-4o", + "provider_key_id": "pk-1", + })) + .unwrap(); + table.insert(ResourceEntry::new("wildcard", wildcard, 1)); + table.insert(ResourceEntry::new("exact", exact, 1)); + table.insert(ResourceEntry::new( + "fixed", + wildcard_alias_fixed_upstream, + 1, + )); + let snap = AisixSnapshot { + models: table, + ..Default::default() + }; + + assert_eq!( + wildcard_resolved_pricing_model(&snap, "wildcard", "gpt-4o-2024-08-06"), + Some("gpt-4o-2024-08-06") + ); + assert_eq!( + wildcard_resolved_pricing_model(&snap, "exact", "forged-price-name"), + None + ); + assert_eq!( + wildcard_resolved_pricing_model(&snap, "fixed", "forged-price-name"), + None + ); + assert_eq!(wildcard_resolved_pricing_model(&snap, "wildcard", ""), None); + } + + #[tokio::test] + async fn request_attribution_stamps_only_concrete_wildcard_pricing_models() { + use aisix_core::resource::ResourceEntry; + use aisix_core::snapshot::ResourceTable; + + let table = ResourceTable::default(); + let configured: aisix_core::Model = serde_json::from_value(serde_json::json!({ + "display_name": "openrouter/*", + "provider": "openai", + "model_name": "*", + "provider_key_id": "pk-1", + })) + .unwrap(); + let cache_entry = configured.clone(); + table.insert(ResourceEntry::new("wildcard", configured, 1)); + let snap = AisixSnapshot { + models: table, + ..Default::default() + }; + let served: aisix_core::Model = serde_json::from_value(serde_json::json!({ + "display_name": "openrouter/*", + "provider": "openai", + "model_name": "gpt-4o-2024-08-06", + "provider_key_id": "pk-1", + })) + .unwrap(); + + crate::attribution::scope( + Arc::new(crate::attribution::RequestAttribution::default()), + async { + crate::attribution::note_target(&served, "pk-1"); + let mut event = UsageEvent { + // NO-GUARDRAIL-CHAIN: this focused unit test constructs + // a synthetic pricing event, not a gateway request. + model_id: "wildcard".to_string(), + guardrail_bypassed_reason: String::new(), + applied_guardrails: Vec::new(), + ..Default::default() + }; + apply_wildcard_pricing_model(&snap, &mut event); + assert_eq!(event.resolved_pricing_model, "gpt-4o-2024-08-06"); + + crate::attribution::note_cache_hit_entry(&cache_entry, "exact"); + let cached_attribution = + crate::attribution::current().expect("in request attribution scope"); + assert_eq!(cached_attribution.cache_hit_layer, Some("exact")); + assert_eq!(cached_attribution.upstream_model, "*"); + let mut cached_event = UsageEvent { + // NO-GUARDRAIL-CHAIN: this focused unit test constructs + // a synthetic pricing event, not a gateway request. + model_id: "wildcard".to_string(), + guardrail_bypassed_reason: String::new(), + applied_guardrails: Vec::new(), + ..Default::default() + }; + apply_wildcard_pricing_model(&snap, &mut cached_event); + assert!(cached_event.resolved_pricing_model.is_empty()); + }, + ) + .await; + } + /// AISIX-Cloud#1289: the id is upstream-controlled and reaches a log line /// and cp-api's `dpmgr_usage_events`. A newline in it would break the /// one-record-per-line shape every log consumer relies on, and an diff --git a/tests/e2e/src/cases/wildcard-pricing-telemetry-e2e.test.ts b/tests/e2e/src/cases/wildcard-pricing-telemetry-e2e.test.ts new file mode 100644 index 00000000..7624d0be --- /dev/null +++ b/tests/e2e/src/cases/wildcard-pricing-telemetry-e2e.test.ts @@ -0,0 +1,156 @@ +import { createHash } from "node:crypto"; +import { afterAll, beforeAll, describe, expect, test } from "vitest"; +import { + EtcdClient, + SeedClient, + spawnApp, + startMockSls, + startOpenAiUpstream, + waitConfigPropagation, + waitForSlsLog, + type MockSls, + type OpenAiUpstream, + type SpawnedApp, +} from "../harness/index.js"; + +// E2E for AISIX-Cloud#1746: the configured wildcard row remains the model +// identity, while the concrete upstream model travels on the usage-export +// wire as `resolved_pricing_model`. AISIX-Cloud's real PostgreSQL receiver +// uses that field only when this row's configured model_name is a template. + +const CALLER_PLAINTEXT = "sk-wildcard-pricing-telemetry-caller"; +const CALLER_KEY_HASH = createHash("sha256").update(CALLER_PLAINTEXT).digest("hex"); +const CREDENTIAL_REF = "wildcardpricing"; +const LOGSTORE = "wildcard-pricing-telemetry"; +const WILDCARD_ALIAS = "openrouter/*"; +const KNOWN_MODEL = "openai/gpt-4o-mini"; +const UNKNOWN_MODEL = "unknown/provider-model"; + +function upstreamResponse() { + return { + id: "chatcmpl-wildcard-pricing", + object: "chat.completion", + created: 0, + // Pricing must come from the dispatch attribution, never a provider + // response field that happens to look like a model identity. + model: "provider-response-model", + choices: [{ index: 0, message: { role: "assistant", content: "ok" }, finish_reason: "stop" }], + usage: { prompt_tokens: 10, completion_tokens: 5, total_tokens: 15 }, + }; +} + +describe("wildcard pricing telemetry e2e", () => { + let app: SpawnedApp | undefined; + let sls: MockSls | undefined; + let upstream: OpenAiUpstream | undefined; + let wildcardID = ""; + let etcdReachable = false; + + beforeAll(async () => { + const etcd = new EtcdClient(); + etcdReachable = await etcd.ping(); + if (!etcdReachable) return; + + sls = await startMockSls(); + upstream = await startOpenAiUpstream({ nonStreamBody: upstreamResponse() }); + app = await spawnApp({ + extraEnv: { + [`SLS_CRED_${CREDENTIAL_REF.toUpperCase()}_AK_ID`]: "mock-akid", + [`SLS_CRED_${CREDENTIAL_REF.toUpperCase()}_AK_SECRET`]: "mock-secret", + }, + }); + const seed = new SeedClient(etcd, app.etcdPrefix); + await seed.createObservabilityExporter({ + name: "wildcard-pricing-sls", + enabled: true, + kind: "aliyun_sls", + endpoint: sls.url, + project: "aisix-e2e-obs", + logstore: LOGSTORE, + credential_ref: CREDENTIAL_REF, + content_mode: "metadata_only", + }); + const providerKey = await seed.createProviderKey({ + display_name: "wildcard-pricing-pk", + provider: "openrouter", + adapter: "openai", + secret: "sk-mock", + api_base: `${upstream.baseUrl}/v1`, + }); + const wildcard = await seed.createModel({ + display_name: WILDCARD_ALIAS, + provider: "openrouter", + model_name: "*", + provider_key_id: providerKey.id, + }); + wildcardID = wildcard.id; + + // Seeded last: a successful models-list gate proves that all preceding + // resources, including the exporter, are in the same gateway snapshot. + await seed.createApiKey({ key_hash: CALLER_KEY_HASH, allowed_models: ["*"] }); + await waitConfigPropagation(async () => { + const res = await fetch(`${app!.proxyUrl}/v1/models`, { + headers: { authorization: `Bearer ${CALLER_PLAINTEXT}` }, + }); + await res.arrayBuffer(); + return res.status === 200; + }); + }, 60_000); + + afterAll(async () => { + await app?.exit(); + await upstream?.close(); + await sls?.close(); + }); + + async function requestModel(model: string): Promise { + if (!app) throw new Error("app not ready"); + const res = await fetch(`${app.proxyUrl}/v1/chat/completions`, { + method: "POST", + headers: { + authorization: `Bearer ${CALLER_PLAINTEXT}`, + "content-type": "application/json", + }, + body: JSON.stringify({ model, messages: [{ role: "user", content: "hi" }] }), + }); + const body = await res.text(); + expect(res.status, body).toBe(200); + } + + test("wildcard dispatch exports the concrete known and unknown pricing identities", async (ctx) => { + if (!etcdReachable || !app || !sls || !upstream || !wildcardID) { + ctx.skip(); + return; + } + + const knownRequest = `openrouter/${KNOWN_MODEL}`; + await requestModel(knownRequest); + expect(upstream.receivedRequests).toHaveLength(1); + expect(JSON.parse(upstream.receivedRequests[0]!.body)).toMatchObject({ model: KNOWN_MODEL }); + + const known = await waitForSlsLog( + sls, + LOGSTORE, + (log) => log.get("requested_model") === knownRequest, + `usage event for ${knownRequest}`, + ); + expect(known.get("model_id")).toBe(wildcardID); + expect(known.get("resolved_pricing_model")).toBe(KNOWN_MODEL); + expect(known.get("prompt_tokens")).toBe("10"); + expect(known.get("completion_tokens")).toBe("5"); + + const unknownRequest = `openrouter/${UNKNOWN_MODEL}`; + await requestModel(unknownRequest); + expect(upstream.receivedRequests).toHaveLength(2); + expect(JSON.parse(upstream.receivedRequests[1]!.body)).toMatchObject({ model: UNKNOWN_MODEL }); + + const unknown = await waitForSlsLog( + sls, + LOGSTORE, + (log) => log.get("requested_model") === unknownRequest, + `usage event for ${unknownRequest}`, + ); + expect(unknown.get("model_id")).toBe(wildcardID); + expect(unknown.get("resolved_pricing_model")).toBe(UNKNOWN_MODEL); + }); +}); From f5cf87b59c94fef1f42d5f74bc6d53c7f460eb44 Mon Sep 17 00:00:00 2001 From: Ming Wen Date: Wed, 30 Sep 2026 16:27:40 +0800 Subject: [PATCH 02/16] fix(telemetry): verify wildcard pricing attribution --- crates/aisix-proxy/src/attribution.rs | 25 +++- crates/aisix-proxy/src/model_resolve.rs | 4 + crates/aisix-proxy/src/usage_attr.rs | 164 +++++++++++++++++++++--- 3 files changed, 172 insertions(+), 21 deletions(-) diff --git a/crates/aisix-proxy/src/attribution.rs b/crates/aisix-proxy/src/attribution.rs index e1f057b5..90711a56 100644 --- a/crates/aisix-proxy/src/attribution.rs +++ b/crates/aisix-proxy/src/attribution.rs @@ -83,6 +83,13 @@ pub(crate) struct Resolved { /// it at read time, so the pair is byte-identical to the one the /// success path emits. pub provider_key_id: String, + /// The concrete upstream model produced by the wildcard-resolution + /// branch, paired with the configured wildcard row that produced it. + /// Empty for exact model resolution, including a request that literally + /// names a wildcard row. Telemetry uses this only for an event that + /// actually dispatched that same row; it is not an access-log identity. + pub wildcard_pricing_model_id: String, + pub wildcard_pricing_model: String, /// Which cache layer answered this request, once one has — `Some` /// exactly when the response came out of the cache. /// @@ -689,6 +696,18 @@ pub(crate) fn note_target(model: &Model, provider_key_id: &str) { }); } +/// Record the concrete model a caller-addressed wildcard row resolved to. +/// +/// This is deliberately separate from [`note_target`]: an exact request for +/// the literal wildcard row also has an upstream model name, but it never +/// passed wildcard capture and must not be used as a pricing identity. +pub(crate) fn note_wildcard_pricing_identity(model_id: &str, concrete_model: &str) { + with(|r| { + r.wildcard_pricing_model_id = model_id.to_string(); + r.wildcard_pricing_model = concrete_model.to_string(); + }); +} + /// Overwrite the target half with what a CACHE HIT may honestly claim. /// /// A hit contacts no upstream, so nothing was dispatched to and the line @@ -707,7 +726,11 @@ pub(crate) fn note_target(model: &Model, provider_key_id: &str) { /// group, whose candidate is no more the producer than any other. pub(crate) fn note_cache_hit_entry(entry: &Model, hit_layer: &'static str) { note_target(entry, entry.provider_key_id.as_deref().unwrap_or_default()); - with(|r| r.cache_hit_layer = Some(hit_layer)); + with(|r| { + r.cache_hit_layer = Some(hit_layer); + r.wildcard_pricing_model_id.clear(); + r.wildcard_pricing_model.clear(); + }); } /// What the current request has resolved, or `None` outside a request. diff --git a/crates/aisix-proxy/src/model_resolve.rs b/crates/aisix-proxy/src/model_resolve.rs index c91d3a79..c0af0cf1 100644 --- a/crates/aisix-proxy/src/model_resolve.rs +++ b/crates/aisix-proxy/src/model_resolve.rs @@ -40,6 +40,10 @@ pub(crate) fn resolve_model( // A wildcard row is a direct model, and attribution stays on the ROW // (see the module docs), so the synthetic clone below inherits its id. note_dispatchable_entry(&entry); + // Keep the concrete value separate from ordinary target attribution: + // an exact request for the literal wildcard row does not run this branch + // and must never make its static template eligible for pricing. + crate::attribution::note_wildcard_pricing_identity(&entry.id, &upstream); let mut model = entry.value.clone(); model.model_name = Some(upstream); Some(Arc::new(ResourceEntry::new( diff --git a/crates/aisix-proxy/src/usage_attr.rs b/crates/aisix-proxy/src/usage_attr.rs index 1f62b18b..eda53c28 100644 --- a/crates/aisix-proxy/src/usage_attr.rs +++ b/crates/aisix-proxy/src/usage_attr.rs @@ -455,7 +455,10 @@ pub(crate) fn wildcard_resolved_pricing_model<'a>( model_id: &str, upstream_model: &'a str, ) -> Option<&'a str> { - if upstream_model.is_empty() { + // A literal wildcard is a configured template, not a concrete upstream + // identity. Sending it to CP would turn a static `*` into a selectable + // price, so only a concrete captured value is eligible. + if upstream_model.is_empty() || upstream_model.contains('*') { return None; } let entry = snap.models.get_by_id(model_id)?; @@ -468,19 +471,24 @@ pub(crate) fn wildcard_resolved_pricing_model<'a>( } /// Fill the optional DP-to-CP wildcard-pricing identity at the one usage -/// emission chokepoint. A cache hit has no dispatched target; its -/// `upstream_model` is the cached entry's static mapping, not a concrete -/// pricing identity. Detached gateway work has no caller attribution and -/// therefore cannot accidentally price itself as the parent request. -fn apply_wildcard_pricing_model(snap: &AisixSnapshot, event: &mut UsageEvent) { +/// emission chokepoint. The identity exists only when model resolution took +/// the wildcard-capture branch; an exact request for the literal wildcard +/// row therefore cannot turn its static template into a price. A cache hit +/// or pre-dispatch failure has no upstream call to price. Detached gateway +/// work has no caller attribution and therefore cannot price itself as the +/// parent request. +fn apply_wildcard_pricing_model(snap: &AisixSnapshot, event: &mut UsageEvent, dispatched: bool) { + if !dispatched { + return; + } let Some(resolved) = crate::attribution::current() else { return; }; - if resolved.cache_hit_layer.is_some() { + if resolved.cache_hit_layer.is_some() || resolved.wildcard_pricing_model_id != event.model_id { return; } if let Some(model) = - wildcard_resolved_pricing_model(snap, &event.model_id, &resolved.upstream_model) + wildcard_resolved_pricing_model(snap, &event.model_id, &resolved.wildcard_pricing_model) { event.resolved_pricing_model = model.to_string(); } @@ -1005,7 +1013,7 @@ pub(crate) fn emit_usage( if terminal && event.guardrail_blocked { state.metrics.record_guardrail_blocked_request(); } - apply_wildcard_pricing_model(snap, &mut event); + apply_wildcard_pricing_model(snap, &mut event, dispatched); let emission = trace.map(|bundle| { event.trace_id = bundle.trace_id_hex(); bundle.emission( @@ -1141,6 +1149,10 @@ mod tests { None ); assert_eq!(wildcard_resolved_pricing_model(&snap, "wildcard", ""), None); + assert_eq!( + wildcard_resolved_pricing_model(&snap, "wildcard", "*"), + None + ); } #[tokio::test] @@ -1162,18 +1174,17 @@ mod tests { models: table, ..Default::default() }; - let served: aisix_core::Model = serde_json::from_value(serde_json::json!({ - "display_name": "openrouter/*", - "provider": "openai", - "model_name": "gpt-4o-2024-08-06", - "provider_key_id": "pk-1", - })) - .unwrap(); - crate::attribution::scope( Arc::new(crate::attribution::RequestAttribution::default()), async { - crate::attribution::note_target(&served, "pk-1"); + let served = + crate::model_resolve::resolve_model(&snap, "openrouter/gpt-4o-2024-08-06") + .expect("wildcard model resolves"); + crate::attribution::note_target(&served.value, "pk-1"); + let attribution = + crate::attribution::current().expect("in request attribution scope"); + assert_eq!(attribution.wildcard_pricing_model_id, "wildcard"); + assert_eq!(attribution.wildcard_pricing_model, "gpt-4o-2024-08-06"); let mut event = UsageEvent { // NO-GUARDRAIL-CHAIN: this focused unit test constructs // a synthetic pricing event, not a gateway request. @@ -1182,7 +1193,7 @@ mod tests { applied_guardrails: Vec::new(), ..Default::default() }; - apply_wildcard_pricing_model(&snap, &mut event); + apply_wildcard_pricing_model(&snap, &mut event, true); assert_eq!(event.resolved_pricing_model, "gpt-4o-2024-08-06"); crate::attribution::note_cache_hit_entry(&cache_entry, "exact"); @@ -1198,11 +1209,124 @@ mod tests { applied_guardrails: Vec::new(), ..Default::default() }; - apply_wildcard_pricing_model(&snap, &mut cached_event); + apply_wildcard_pricing_model(&snap, &mut cached_event, false); assert!(cached_event.resolved_pricing_model.is_empty()); }, ) .await; + + crate::attribution::scope( + Arc::new(crate::attribution::RequestAttribution::default()), + async { + let literal = crate::model_resolve::resolve_model(&snap, "openrouter/*") + .expect("literal configured wildcard row resolves exactly"); + crate::attribution::note_target(&literal.value, "pk-1"); + let attribution = + crate::attribution::current().expect("in request attribution scope"); + assert!(attribution.wildcard_pricing_model_id.is_empty()); + assert!(attribution.wildcard_pricing_model.is_empty()); + + let mut event = UsageEvent { + // This exact alias is dispatchable but its static `*` + // template is never a concrete catalog price. + model_id: "wildcard".to_string(), + guardrail_bypassed_reason: String::new(), + applied_guardrails: Vec::new(), + ..Default::default() + }; + apply_wildcard_pricing_model(&snap, &mut event, true); + assert!(event.resolved_pricing_model.is_empty()); + }, + ) + .await; + } + + /// A failed attempt retains its target row so it can be observed, but a + /// target rate-limit or bridge-preparation refusal never reached a + /// provider and therefore must not claim a concrete wildcard price. + #[tokio::test] + async fn undispatched_wildcard_attempt_has_no_pricing_identity() { + use aisix_core::snapshot::{ResourceTable, SnapshotHandle}; + use aisix_core::ProxyConfig; + use aisix_gateway::Hub; + use aisix_obs::UsageSink; + + let table = ResourceTable::default(); + let configured: aisix_core::Model = serde_json::from_value(serde_json::json!({ + "display_name": "openrouter/*", + "provider": "openai", + "model_name": "*", + "provider_key_id": "pk-1", + })) + .unwrap(); + table.insert(ResourceEntry::new("wildcard", configured, 1)); + let snap = AisixSnapshot { + models: table, + ..Default::default() + }; + let cfg = ProxyConfig { + addr: "127.0.0.1:0".into(), + request_body_limit_bytes: 0, + real_ip: Default::default(), + request_id: Default::default(), + url_rewrites: Vec::new(), + tls: None, + listeners: Vec::new(), + thread_per_core: None, + workers: None, + }; + let (tx, mut rx) = tokio::sync::mpsc::channel(1); + let state = ProxyState::new( + SnapshotHandle::new(snap.clone()), + Arc::new(Hub::new()), + &cfg, + ) + .with_usage_sink(UsageSink::new(tx)); + let client = ClientContext::default(); + let attempts = [crate::attempt::AttemptRecord { + index: 0, + kind: "initial", + target_model: String::new(), + target_model_id: "wildcard".to_string(), + provider_key_id: "pk-1".to_string(), + status: 429, + success: false, + error_class: "rate_limited".to_string(), + error_message: "target rate limit refused before dispatch".to_string(), + latency_ms: 0, + dispatched: false, + }]; + + crate::attribution::scope( + Arc::new(crate::attribution::RequestAttribution::default()), + async { + let served = + crate::model_resolve::resolve_model(&snap, "openrouter/gpt-4o-2024-08-06") + .expect("wildcard model resolves"); + crate::attribution::note_target(&served.value, "pk-1"); + emit_failed_attempts( + &state, + &snap, + crate::operation::CHAT, + "request-1746", + "openrouter/gpt-4o-2024-08-06", + "api-key", + &client, + &[], + &attempts, + true, + false, + Vec::new(), + crate::redact::RedactionCounts::new(), + &None, + ); + }, + ) + .await; + + let event = rx.try_recv().expect("failed attempt emits usage"); + assert_eq!(event.model_id, "wildcard"); + assert!(event.resolved_pricing_model.is_empty()); } /// AISIX-Cloud#1289: the id is upstream-controlled and reaches a log line From 55b0a82dede2fc99df80ac6ddd76bd07095cb896 Mon Sep 17 00:00:00 2001 From: Ming Wen Date: Wed, 30 Sep 2026 16:37:07 +0800 Subject: [PATCH 03/16] test(telemetry): complete wildcard attribution fixture --- crates/aisix-proxy/src/request_metrics.rs | 2 ++ 1 file changed, 2 insertions(+) diff --git a/crates/aisix-proxy/src/request_metrics.rs b/crates/aisix-proxy/src/request_metrics.rs index e0a72d82..e179767f 100644 --- a/crates/aisix-proxy/src/request_metrics.rs +++ b/crates/aisix-proxy/src/request_metrics.rs @@ -702,6 +702,8 @@ mod tests { provider: "OpenAI".to_string(), upstream_model: "gpt-4o-mini".to_string(), provider_key_id: "pk-1".to_string(), + wildcard_pricing_model_id: String::new(), + wildcard_pricing_model: String::new(), cache_hit_layer: None, } } From 4b63f5df75b62cc7af35b956803d3a55c865eabf Mon Sep 17 00:00:00 2001 From: Ming Wen Date: Wed, 30 Sep 2026 16:53:32 +0800 Subject: [PATCH 04/16] fix(telemetry): preserve wildcard pricing eligibility --- crates/aisix-proxy/src/jobs.rs | 18 ++++++-- crates/aisix-proxy/src/request_metrics.rs | 51 ++++++++++++++++++++++- crates/aisix-proxy/src/usage_attr.rs | 48 +++++++++++++++++---- 3 files changed, 104 insertions(+), 13 deletions(-) diff --git a/crates/aisix-proxy/src/jobs.rs b/crates/aisix-proxy/src/jobs.rs index 0f414d8d..2894bcac 100644 --- a/crates/aisix-proxy/src/jobs.rs +++ b/crates/aisix-proxy/src/jobs.rs @@ -2542,7 +2542,10 @@ mod tests { snap.apikeys.insert(apikey_entry(&["*"])); let (app, mut rx) = build_app_with_sink(snap); - let encoded = encode_routed_id("batch_1", "jobs/*"); + // A concrete caller hint enters the wildcard-capture branch. The + // zero-token management event must still stay unpriced even though + // this GET genuinely contacts the upstream batch API. + let encoded = encode_routed_id("batch_1", "jobs/gpt-4o-2024-08-06"); let mk_req = || { Request::builder() .method("GET") @@ -2564,7 +2567,7 @@ mod tests { // Two events expected: the zero-token management event plus ONE // aggregated batch event from the detached attribution task. - let mut mgmt = 0u32; + let mut mgmt: Option = None; let mut agg: Option = None; for _ in 0..2 { let ev = tokio::time::timeout(Duration::from_secs(3), rx.recv()) @@ -2574,10 +2577,17 @@ mod tests { if ev.inbound_protocol == "batch" { agg = Some(ev); } else { - mgmt += 1; + assert!( + mgmt.replace(ev).is_none(), + "only one management event expected" + ); } } - assert_eq!(mgmt, 1); + let mgmt = mgmt.expect("management event must be emitted"); + assert!( + mgmt.resolved_pricing_model.is_empty(), + "a batch-management request must not select wildcard pricing" + ); let agg = agg.expect("aggregated batch event must be emitted"); // cp-api accepts any visible-ASCII request_id since // AISIX-Cloud#1288, so this is no longer a wire constraint — but a diff --git a/crates/aisix-proxy/src/request_metrics.rs b/crates/aisix-proxy/src/request_metrics.rs index e179767f..0aeab00d 100644 --- a/crates/aisix-proxy/src/request_metrics.rs +++ b/crates/aisix-proxy/src/request_metrics.rs @@ -171,7 +171,9 @@ impl Default for Upstream<'_> { /// Whether the caller addressed an ensemble model. See [`LastTarget::new`]. fn is_ensemble(snap: &AisixSnapshot, requested_model: &str) -> bool { !requested_model.is_empty() - && crate::model_resolve::resolve_model(snap, requested_model) + && snap + .models + .get_by_name(requested_model) .is_some_and(|entry| entry.value.is_ensemble()) } @@ -755,6 +757,53 @@ mod tests { assert_eq!(upstream.pk.name(), UNKNOWN); } + /// E2E metric recording loads the latest snapshot after dispatch. Its + /// ensemble check is classification only: resolving a wildcard again + /// there would replace the concrete pricing identity captured from the + /// snapshot that actually dispatched the request. + #[tokio::test] + async fn ensemble_metric_check_does_not_replace_captured_wildcard_identity() { + use std::sync::Arc; + + let dispatched = snapshot_with( + "wildcard", + serde_json::json!({ + "display_name": "openrouter/*", + "provider": "openai", + "model_name": "*", + "provider_key_id": "pk-1", + }), + ); + let refreshed = snapshot_with( + "wildcard", + serde_json::json!({ + "display_name": "openrouter/*", + "provider": "openai", + "model_name": "replacement-*", + "provider_key_id": "pk-1", + }), + ); + + crate::attribution::scope( + Arc::new(crate::attribution::RequestAttribution::default()), + async { + crate::model_resolve::resolve_model(&dispatched, "openrouter/gpt-4o") + .expect("wildcard model resolves at dispatch"); + let captured = + crate::attribution::current().expect("request attribution is installed"); + assert_eq!(captured.wildcard_pricing_model_id, "wildcard"); + assert_eq!(captured.wildcard_pricing_model, "gpt-4o"); + + assert!(!is_ensemble(&refreshed, "openrouter/gpt-4o")); + let after_metrics = + crate::attribution::current().expect("request attribution is installed"); + assert_eq!(after_metrics.wildcard_pricing_model_id, "wildcard"); + assert_eq!(after_metrics.wildcard_pricing_model, "gpt-4o"); + }, + ) + .await; + } + /// A request that never selected a target keeps the placeholder, and the /// ProviderKey id must be `unknown` rather than the empty string an /// unresolved `ResolvedPk` reports verbatim — an empty label value would diff --git a/crates/aisix-proxy/src/usage_attr.rs b/crates/aisix-proxy/src/usage_attr.rs index eda53c28..d41140a8 100644 --- a/crates/aisix-proxy/src/usage_attr.rs +++ b/crates/aisix-proxy/src/usage_attr.rs @@ -477,8 +477,17 @@ pub(crate) fn wildcard_resolved_pricing_model<'a>( /// or pre-dispatch failure has no upstream call to price. Detached gateway /// work has no caller attribution and therefore cannot price itself as the /// parent request. -fn apply_wildcard_pricing_model(snap: &AisixSnapshot, event: &mut UsageEvent, dispatched: bool) { - if !dispatched { +fn apply_wildcard_pricing_model( + snap: &AisixSnapshot, + event: &mut UsageEvent, + surface: Surface, + dispatched: bool, +) { + // Batch-management rows are observable upstream management requests, not + // the batch's model inference. They stay unpriced even when their route + // resolves through a wildcard alias; batch completion accounting is a + // separate surface with its own attribution rules. + if surface == crate::operation::BATCHES || !dispatched { return; } let Some(resolved) = crate::attribution::current() else { @@ -1013,7 +1022,7 @@ pub(crate) fn emit_usage( if terminal && event.guardrail_blocked { state.metrics.record_guardrail_blocked_request(); } - apply_wildcard_pricing_model(snap, &mut event, dispatched); + apply_wildcard_pricing_model(snap, &mut event, surface, dispatched); let emission = trace.map(|bundle| { event.trace_id = bundle.trace_id_hex(); bundle.emission( @@ -1193,9 +1202,25 @@ mod tests { applied_guardrails: Vec::new(), ..Default::default() }; - apply_wildcard_pricing_model(&snap, &mut event, true); + apply_wildcard_pricing_model(&snap, &mut event, crate::operation::CHAT, true); assert_eq!(event.resolved_pricing_model, "gpt-4o-2024-08-06"); + let mut batch_event = UsageEvent { + // NO-GUARDRAIL-CHAIN: this focused unit test constructs + // a synthetic pricing event, not a gateway request. + model_id: "wildcard".to_string(), + guardrail_bypassed_reason: String::new(), + applied_guardrails: Vec::new(), + ..Default::default() + }; + apply_wildcard_pricing_model( + &snap, + &mut batch_event, + crate::operation::BATCHES, + true, + ); + assert!(batch_event.resolved_pricing_model.is_empty()); + crate::attribution::note_cache_hit_entry(&cache_entry, "exact"); let cached_attribution = crate::attribution::current().expect("in request attribution scope"); @@ -1209,7 +1234,12 @@ mod tests { applied_guardrails: Vec::new(), ..Default::default() }; - apply_wildcard_pricing_model(&snap, &mut cached_event, false); + apply_wildcard_pricing_model( + &snap, + &mut cached_event, + crate::operation::CHAT, + false, + ); assert!(cached_event.resolved_pricing_model.is_empty()); }, ) @@ -1227,14 +1257,16 @@ mod tests { assert!(attribution.wildcard_pricing_model.is_empty()); let mut event = UsageEvent { - // This exact alias is dispatchable but its static `*` - // template is never a concrete catalog price. + // NO-GUARDRAIL-CHAIN: this focused unit test constructs + // a synthetic pricing event, not a gateway request. This + // exact alias is dispatchable but its static `*` template + // is never a concrete catalog price. model_id: "wildcard".to_string(), guardrail_bypassed_reason: String::new(), applied_guardrails: Vec::new(), ..Default::default() }; - apply_wildcard_pricing_model(&snap, &mut event, true); + apply_wildcard_pricing_model(&snap, &mut event, crate::operation::CHAT, true); assert!(event.resolved_pricing_model.is_empty()); }, ) From 34d40a63dd6b6cf4a8be69c8aeb2c29cfe8eeeb9 Mon Sep 17 00:00:00 2001 From: Ming Wen Date: Wed, 30 Sep 2026 17:12:18 +0800 Subject: [PATCH 05/16] fix(telemetry): retain dispatched wildcard identity --- crates/aisix-proxy/src/attribution.rs | 13 ++ crates/aisix-proxy/src/model_resolve.rs | 63 ++++++- crates/aisix-proxy/src/realtime.rs | 81 +++++++- crates/aisix-proxy/src/usage_attr.rs | 233 ++++++++++++------------ 4 files changed, 266 insertions(+), 124 deletions(-) diff --git a/crates/aisix-proxy/src/attribution.rs b/crates/aisix-proxy/src/attribution.rs index 90711a56..204a9516 100644 --- a/crates/aisix-proxy/src/attribution.rs +++ b/crates/aisix-proxy/src/attribution.rs @@ -738,6 +738,19 @@ pub(crate) fn current() -> Option { CURRENT.try_with(|a| a.get()).ok() } +/// The current request's attribution cell, for work that continues on a +/// detached task after the HTTP handler returns. +/// +/// A WebSocket upgrade moves its session onto axum's upgrade task. That task +/// does not inherit Tokio task-locals, but it is still the same client request: +/// the terminal session usage row must retain the model identity captured +/// before the upgrade. Callers install this exact cell with [`scope`] around +/// their detached continuation; `None` remains correct outside request +/// middleware (for example, focused unit tests). +pub(crate) fn current_cell() -> Option> { + CURRENT.try_with(Arc::clone).ok() +} + /// The dispatched-target half of an access-log line (AISIX-Cloud#1571). /// /// The line's `model=` is the entry the CALLER addressed — for a routing diff --git a/crates/aisix-proxy/src/model_resolve.rs b/crates/aisix-proxy/src/model_resolve.rs index c0af0cf1..4e99841e 100644 --- a/crates/aisix-proxy/src/model_resolve.rs +++ b/crates/aisix-proxy/src/model_resolve.rs @@ -42,8 +42,13 @@ pub(crate) fn resolve_model( note_dispatchable_entry(&entry); // Keep the concrete value separate from ordinary target attribution: // an exact request for the literal wildcard row does not run this branch - // and must never make its static template eligible for pricing. - crate::attribution::note_wildcard_pricing_identity(&entry.id, &upstream); + // and must never make its static template eligible for pricing. This is + // deliberately decided against the dispatch snapshot: a later terminal + // emitter may see a refreshed configuration where the row changed or was + // deleted, but it must price the concrete model this request dispatched. + if wildcard_pricing_eligible(&entry.value, &upstream) { + crate::attribution::note_wildcard_pricing_identity(&entry.id, &upstream); + } let mut model = entry.value.clone(); model.model_name = Some(upstream); Some(Arc::new(ResourceEntry::new( @@ -163,6 +168,21 @@ fn resolve_upstream_model_name(model: &Model, capture: &str) -> String { } } +/// Whether a wildcard dispatch produced a concrete provider-model identity +/// that is safe to send to CP for pricing. +/// +/// A wildcard display alias with a fixed upstream is already priced by its +/// configured model name. Only an upstream template yields a caller-specific +/// provider-model identity, and a literal `*` is never a concrete catalog key. +fn wildcard_pricing_eligible(model: &Model, upstream: &str) -> bool { + model + .model_name + .as_deref() + .is_some_and(|template| template.contains('*')) + && !upstream.is_empty() + && !upstream.contains('*') +} + #[cfg(test)] mod tests { use super::*; @@ -243,6 +263,45 @@ mod tests { assert_eq!(resolved.value.display_name, "openai/*"); } + /// Pricing eligibility belongs to the snapshot that dispatched the + /// request. In particular, a wildcard display alias over a fixed model, + /// a literal wildcard row, and a capture that is still itself a wildcard + /// must not manufacture a concrete provider-model price identity. + #[tokio::test] + async fn only_concrete_template_capture_sets_wildcard_pricing_identity() { + use std::sync::Arc; + + let snap = snapshot_with(vec![ + ("wildcard", direct_model("openrouter/*", Some("*"))), + ("fixed", direct_model("fixed/*", Some("gpt-4o"))), + ]); + + crate::attribution::scope( + Arc::new(crate::attribution::RequestAttribution::default()), + async { + resolve_model(&snap, "openrouter/gpt-4o-2024-08-06") + .expect("concrete wildcard request resolves"); + let resolved = crate::attribution::current().expect("in request scope"); + assert_eq!(resolved.wildcard_pricing_model_id, "wildcard"); + assert_eq!(resolved.wildcard_pricing_model, "gpt-4o-2024-08-06"); + }, + ) + .await; + + for requested in ["fixed/anything", "openrouter/*", "openrouter/gpt-*"] { + crate::attribution::scope( + Arc::new(crate::attribution::RequestAttribution::default()), + async { + resolve_model(&snap, requested).expect("configured request resolves"); + let resolved = crate::attribution::current().expect("in request scope"); + assert!(resolved.wildcard_pricing_model_id.is_empty(), "{requested}"); + assert!(resolved.wildcard_pricing_model.is_empty(), "{requested}"); + }, + ) + .await; + } + } + #[test] fn wildcard_row_name_bounds_caller_minted_aliases() { let snap = snapshot_with(vec![ diff --git a/crates/aisix-proxy/src/realtime.rs b/crates/aisix-proxy/src/realtime.rs index 248866ef..9dd803ff 100644 --- a/crates/aisix-proxy/src/realtime.rs +++ b/crates/aisix-proxy/src/realtime.rs @@ -270,6 +270,11 @@ pub(crate) async fn realtime( Ok((ws, prep)) => { let state2 = state.clone(); let client2 = client.clone(); + // `on_upgrade` runs on a new Tokio task, which does not inherit + // the request task-local. Keep the same cell so the terminal + // realtime UsageEvent retains the wildcard model identity that + // `prepare` resolved before accepting this upgrade. + let attribution = crate::attribution::current_cell(); // `on_upgrade` runs the session on a detached task, so the // request span has to be attached to the future rather than // inherited — without it the session's guardrail checks log @@ -278,7 +283,16 @@ pub(crate) async fn realtime( ws.protocols(["realtime"]).on_upgrade(move |socket| { use tracing::Instrument as _; async move { - run_session(state2, prep, socket, client2, request_id, started).await; + run_session( + state2, + prep, + socket, + client2, + request_id, + started, + attribution, + ) + .await; } .instrument(span) }) @@ -725,6 +739,22 @@ async fn run_session( client: ClientContext, request_id: String, started: Instant, + attribution: Option>, +) { + let session = run_session_inner(state, prep, client_ws, client, request_id, started); + match attribution { + Some(cell) => crate::attribution::scope(cell, session).await, + None => session.await, + } +} + +async fn run_session_inner( + state: ProxyState, + prep: Prepared, + client_ws: WebSocket, + client: ClientContext, + request_id: String, + started: Instant, ) { let Prepared { auth, @@ -1582,6 +1612,55 @@ mod tests { assert_eq!(ev.api_key_id, "k-1"); } + /// `on_upgrade` moves the session to a task that does not inherit the + /// request task-local. A wildcard model resolved before the upgrade must + /// still reach its terminal session UsageEvent as the concrete provider + /// pricing identity, rather than being lost when the HTTP handler ends. + #[tokio::test] + async fn terminal_realtime_usage_keeps_wildcard_pricing_identity() { + let (up_addr, _handshake, _frames) = spawn_upstream().await; + let snap = snapshot(&format!("http://{up_addr}/v1"), "openai", "openai"); + let wildcard: Model = serde_json::from_value(serde_json::json!({ + "display_name": "rt-*", + "provider": "openai", + "model_name": "gpt-realtime-*", + "provider_key_id": PK_ID, + })) + .unwrap(); + snap.models.insert(ResourceEntry::new("m-rt", wildcard, 2)); + let (addr, _state, mut rx) = serve(snap).await; + + let mut req = format!("ws://{addr}/v1/realtime?model=rt-2026-01-01") + .into_client_request() + .unwrap(); + req.headers_mut() + .insert("authorization", "Bearer sk-caller".parse().unwrap()); + let (ws, _) = tokio_tungstenite::connect_async(req) + .await + .expect("handshake"); + let (mut tx, mut client_rx) = ws.split(); + tx.send(TgMessage::Text( + serde_json::json!({"type": "session.update", "session": {"instructions": "hi"}}) + .to_string(), + )) + .await + .unwrap(); + + while let Some(Ok(message)) = client_rx.next().await { + if matches!(message, TgMessage::Close(_)) { + break; + } + } + + let event = tokio::time::timeout(Duration::from_secs(3), rx.recv()) + .await + .expect("terminal usage event expected") + .expect("usage sink closed"); + assert_eq!(event.model_id, "m-rt"); + assert_eq!(event.requested_model, "rt-2026-01-01"); + assert_eq!(event.resolved_pricing_model, "gpt-realtime-2026-01-01"); + } + /// `/v1/realtime` builds its upstream handshake by hand rather than /// through the shared bridge pipeline, which is exactly where a /// per-request mechanism goes silently missing: the operator diff --git a/crates/aisix-proxy/src/usage_attr.rs b/crates/aisix-proxy/src/usage_attr.rs index d41140a8..d69012f4 100644 --- a/crates/aisix-proxy/src/usage_attr.rs +++ b/crates/aisix-proxy/src/usage_attr.rs @@ -439,50 +439,15 @@ pub(crate) fn metric_model_label_pair<'a>( } } -/// The actual upstream model name a wildcard upstream template resolved for -/// this request, if that row still exists in the same snapshot as the emitted -/// event. -/// -/// `model_id` deliberately stays on the configured row: rate limits, -/// policy, and historical attribution all key on it. Pricing is the one -/// consumer that needs the concrete provider model name, and only a row whose -/// configured `model_name` is a template may use it. A wildcard display alias -/// over one fixed upstream still prices by that fixed configuration. The name -/// comes from request-local dispatch attribution, never from the -/// caller-controlled `requested_model` or an upstream response field. -pub(crate) fn wildcard_resolved_pricing_model<'a>( - snap: &AisixSnapshot, - model_id: &str, - upstream_model: &'a str, -) -> Option<&'a str> { - // A literal wildcard is a configured template, not a concrete upstream - // identity. Sending it to CP would turn a static `*` into a selectable - // price, so only a concrete captured value is eligible. - if upstream_model.is_empty() || upstream_model.contains('*') { - return None; - } - let entry = snap.models.get_by_id(model_id)?; - let model = &entry.value; - model - .model_name - .as_deref() - .is_some_and(|name| name.contains('*')) - .then_some(upstream_model) -} - /// Fill the optional DP-to-CP wildcard-pricing identity at the one usage -/// emission chokepoint. The identity exists only when model resolution took -/// the wildcard-capture branch; an exact request for the literal wildcard -/// row therefore cannot turn its static template into a price. A cache hit -/// or pre-dispatch failure has no upstream call to price. Detached gateway -/// work has no caller attribution and therefore cannot price itself as the -/// parent request. -fn apply_wildcard_pricing_model( - snap: &AisixSnapshot, - event: &mut UsageEvent, - surface: Surface, - dispatched: bool, -) { +/// emission chokepoint. Model resolution establishes eligibility from the +/// dispatch snapshot and records only a concrete provider-model identity, so +/// this path must not consult an emission-time snapshot that may have changed +/// while a stream or realtime session was still running. A cache hit or +/// pre-dispatch failure has no upstream call to price. Detached gateway work +/// has no caller attribution and therefore cannot price itself as the parent +/// request. +fn apply_wildcard_pricing_model(event: &mut UsageEvent, surface: Surface, dispatched: bool) { // Batch-management rows are observable upstream management requests, not // the batch's model inference. They stay unpriced even when their route // resolves through a wildcard alias; batch completion accounting is a @@ -496,10 +461,8 @@ fn apply_wildcard_pricing_model( if resolved.cache_hit_layer.is_some() || resolved.wildcard_pricing_model_id != event.model_id { return; } - if let Some(model) = - wildcard_resolved_pricing_model(snap, &event.model_id, &resolved.wildcard_pricing_model) - { - event.resolved_pricing_model = model.to_string(); + if !resolved.wildcard_pricing_model.is_empty() { + event.resolved_pricing_model = resolved.wildcard_pricing_model; } } @@ -1022,7 +985,7 @@ pub(crate) fn emit_usage( if terminal && event.guardrail_blocked { state.metrics.record_guardrail_blocked_request(); } - apply_wildcard_pricing_model(snap, &mut event, surface, dispatched); + apply_wildcard_pricing_model(&mut event, surface, dispatched); let emission = trace.map(|bundle| { event.trace_id = bundle.trace_id_hex(); bundle.emission( @@ -1105,65 +1068,6 @@ mod tests { assert_eq!((m.as_ref(), u.as_ref()), ("no-such/model", "raw-upstream")); } - #[test] - fn wildcard_pricing_model_uses_only_the_configured_wildcard_row() { - use aisix_core::resource::ResourceEntry; - use aisix_core::snapshot::ResourceTable; - - let table = ResourceTable::default(); - let wildcard: aisix_core::Model = serde_json::from_value(serde_json::json!({ - "display_name": "openrouter/*", - "provider": "openai", - "model_name": "*", - "provider_key_id": "pk-1", - })) - .unwrap(); - let exact: aisix_core::Model = serde_json::from_value(serde_json::json!({ - "display_name": "exact-model", - "provider": "openai", - "model_name": "gpt-4o", - "provider_key_id": "pk-1", - })) - .unwrap(); - let wildcard_alias_fixed_upstream: aisix_core::Model = - serde_json::from_value(serde_json::json!({ - "display_name": "fixed/*", - "provider": "openai", - "model_name": "gpt-4o", - "provider_key_id": "pk-1", - })) - .unwrap(); - table.insert(ResourceEntry::new("wildcard", wildcard, 1)); - table.insert(ResourceEntry::new("exact", exact, 1)); - table.insert(ResourceEntry::new( - "fixed", - wildcard_alias_fixed_upstream, - 1, - )); - let snap = AisixSnapshot { - models: table, - ..Default::default() - }; - - assert_eq!( - wildcard_resolved_pricing_model(&snap, "wildcard", "gpt-4o-2024-08-06"), - Some("gpt-4o-2024-08-06") - ); - assert_eq!( - wildcard_resolved_pricing_model(&snap, "exact", "forged-price-name"), - None - ); - assert_eq!( - wildcard_resolved_pricing_model(&snap, "fixed", "forged-price-name"), - None - ); - assert_eq!(wildcard_resolved_pricing_model(&snap, "wildcard", ""), None); - assert_eq!( - wildcard_resolved_pricing_model(&snap, "wildcard", "*"), - None - ); - } - #[tokio::test] async fn request_attribution_stamps_only_concrete_wildcard_pricing_models() { use aisix_core::resource::ResourceEntry; @@ -1202,7 +1106,7 @@ mod tests { applied_guardrails: Vec::new(), ..Default::default() }; - apply_wildcard_pricing_model(&snap, &mut event, crate::operation::CHAT, true); + apply_wildcard_pricing_model(&mut event, crate::operation::CHAT, true); assert_eq!(event.resolved_pricing_model, "gpt-4o-2024-08-06"); let mut batch_event = UsageEvent { @@ -1213,12 +1117,7 @@ mod tests { applied_guardrails: Vec::new(), ..Default::default() }; - apply_wildcard_pricing_model( - &snap, - &mut batch_event, - crate::operation::BATCHES, - true, - ); + apply_wildcard_pricing_model(&mut batch_event, crate::operation::BATCHES, true); assert!(batch_event.resolved_pricing_model.is_empty()); crate::attribution::note_cache_hit_entry(&cache_entry, "exact"); @@ -1226,6 +1125,10 @@ mod tests { crate::attribution::current().expect("in request attribution scope"); assert_eq!(cached_attribution.cache_hit_layer, Some("exact")); assert_eq!(cached_attribution.upstream_model, "*"); + // `note_cache_hit_entry` clears the identity above. Restore + // one here to pin the separate emission gate too: a cache + // hit is never billable as an upstream wildcard dispatch. + crate::attribution::note_wildcard_pricing_identity("wildcard", "gpt-4o-2024-08-06"); let mut cached_event = UsageEvent { // NO-GUARDRAIL-CHAIN: this focused unit test constructs // a synthetic pricing event, not a gateway request. @@ -1234,12 +1137,7 @@ mod tests { applied_guardrails: Vec::new(), ..Default::default() }; - apply_wildcard_pricing_model( - &snap, - &mut cached_event, - crate::operation::CHAT, - false, - ); + apply_wildcard_pricing_model(&mut cached_event, crate::operation::CHAT, true); assert!(cached_event.resolved_pricing_model.is_empty()); }, ) @@ -1266,13 +1164,106 @@ mod tests { applied_guardrails: Vec::new(), ..Default::default() }; - apply_wildcard_pricing_model(&snap, &mut event, crate::operation::CHAT, true); + apply_wildcard_pricing_model(&mut event, crate::operation::CHAT, true); assert!(event.resolved_pricing_model.is_empty()); }, ) .await; } + /// A stream can outlive the configuration generation that dispatched it. + /// The terminal event must retain the concrete provider model captured at + /// dispatch, even if the same row becomes a fixed alias or disappears + /// before the terminal emit runs. + #[tokio::test] + async fn terminal_usage_keeps_dispatch_time_wildcard_identity_after_refresh_or_delete() { + use aisix_core::snapshot::{ResourceTable, SnapshotHandle}; + use aisix_core::ProxyConfig; + use aisix_gateway::Hub; + use aisix_obs::UsageSink; + + fn wildcard_snapshot(template: &str) -> AisixSnapshot { + let table = ResourceTable::default(); + let model: aisix_core::Model = serde_json::from_value(serde_json::json!({ + "display_name": "openrouter/*", + "provider": "openai", + "model_name": template, + "provider_key_id": "pk-1", + })) + .unwrap(); + table.insert(ResourceEntry::new("wildcard", model, 1)); + AisixSnapshot { + models: table, + ..Default::default() + } + } + + let dispatched = wildcard_snapshot("*"); + let refreshed_fixed = wildcard_snapshot("gpt-4o"); + let deleted = AisixSnapshot::new(); + let cfg = ProxyConfig { + addr: "127.0.0.1:0".into(), + request_body_limit_bytes: 0, + real_ip: Default::default(), + request_id: Default::default(), + url_rewrites: Vec::new(), + tls: None, + listeners: Vec::new(), + thread_per_core: None, + workers: None, + }; + + for (case, terminal_snapshot) in [ + ("the row changed to a fixed alias", refreshed_fixed), + ("the row was deleted", deleted), + ] { + let (tx, mut rx) = tokio::sync::mpsc::channel(1); + let state = ProxyState::new( + SnapshotHandle::new(terminal_snapshot.clone()), + Arc::new(Hub::new()), + &cfg, + ) + .with_usage_sink(UsageSink::new(tx)); + + crate::attribution::scope( + Arc::new(crate::attribution::RequestAttribution::default()), + async { + crate::model_resolve::resolve_model( + &dispatched, + "openrouter/gpt-4o-2024-08-06", + ) + .expect("wildcard model resolves at dispatch"); + let pk = ResolvedPk::unresolved(); + emit_usage( + &state, + &terminal_snapshot, + crate::operation::CHAT, + UsageEvent { + // NO-GUARDRAIL-CHAIN: this focused test emits a + // synthetic terminal usage event after dispatch. + model_id: "wildcard".to_string(), + guardrail_bypassed_reason: String::new(), + applied_guardrails: Vec::new(), + ..Default::default() + }, + usage_event_labels("openrouter/*", &pk), + None, + None, + /* terminal */ true, + /* dispatched */ true, + ); + }, + ) + .await; + + let event = rx.try_recv().expect("terminal usage event emitted"); + assert_eq!( + event.resolved_pricing_model, "gpt-4o-2024-08-06", + "{case} must not rewrite the dispatched wildcard identity" + ); + } + } + /// A failed attempt retains its target row so it can be observed, but a /// target rate-limit or bridge-preparation refusal never reached a /// provider and therefore must not claim a concrete wildcard price. From 33c9c18e9aa928f3b7e51ebf7af02d49997d7fcb Mon Sep 17 00:00:00 2001 From: Ming Wen Date: Wed, 30 Sep 2026 19:34:41 +0800 Subject: [PATCH 06/16] fix(telemetry): retain wildcard pricing authority --- crates/aisix-core/src/models/model.rs | 81 +++++++++++++++++--- crates/aisix-core/src/models/schema.rs | 53 ++++++++++++- crates/aisix-obs/src/usage.rs | 9 +++ crates/aisix-proxy/src/attribution.rs | 48 +++++++++++- crates/aisix-proxy/src/messages.rs | 21 +++++- crates/aisix-proxy/src/model_resolve.rs | 75 +++++++++++++++--- crates/aisix-proxy/src/realtime.rs | 5 ++ crates/aisix-proxy/src/request_metrics.rs | 10 +++ crates/aisix-proxy/src/usage_attr.rs | 84 ++++++++++++++++++--- schemas/resources-lenient/model.schema.json | 7 ++ schemas/resources/model.schema.json | 25 ++++++ 11 files changed, 381 insertions(+), 37 deletions(-) diff --git a/crates/aisix-core/src/models/model.rs b/crates/aisix-core/src/models/model.rs index 286f429d..b954a82a 100644 --- a/crates/aisix-core/src/models/model.rs +++ b/crates/aisix-core/src/models/model.rs @@ -310,6 +310,19 @@ pub struct Model { #[schemars(length(min = 1, max = 255))] pub pricing_key: Option, + /// Opaque control-plane-issued canonical, non-nil UUID that authorizes a + /// concrete wildcard-upstream model name for pricing. It is relevant only + /// to a direct-shaped wildcard model (chat or embedding); without it the + /// data plane deliberately omits the resolved model from terminal + /// telemetry so the control plane can leave the call unpriced rather than + /// trust mutable configuration. + #[serde(default, skip_serializing_if = "Option::is_none")] + #[schemars( + regex(pattern = "^[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}$"), + length(min = 36, max = 64) + )] + pub pricing_authority_id: Option, + /// Direct-model-only background health-check configuration. #[serde(default, skip_serializing_if = "Option::is_none")] pub background_model_check: Option, @@ -421,6 +434,9 @@ impl Model { if self.effort_mapping.take().is_some() { stripped.push("effort_mapping"); } + if self.pricing_authority_id.take().is_some() { + stripped.push("pricing_authority_id"); + } if self.pricing_key.take().is_some() { stripped.push("pricing_key"); } @@ -533,8 +549,8 @@ pub fn model_one_of() -> Value { /// [`Model::strip_kind_inapplicable`]). Kind policy (project decision): /// generic call knobs (`timeout`/`stream_timeout`/`retries`) resolve /// member → group → deployment default wherever a group slot exists; -/// model-specific knobs (`auto_prompt_caching`, `cost`, `pricing_key`) are -/// direct-only. +/// model-specific knobs (`auto_prompt_caching`, `cost`, `pricing_key`, +/// `pricing_authority_id`) are direct-shaped-only. pub fn model_one_of_strict() -> Value { model_one_of_variant(true) } @@ -559,6 +575,7 @@ fn model_one_of_variant(strict: bool) -> Value { "auto_prompt_caching", "cost", "pricing_key", + "pricing_authority_id", "effort_mapping", ], ); @@ -580,6 +597,7 @@ fn model_one_of_variant(strict: bool) -> Value { "auto_prompt_caching", "cost", "pricing_key", + "pricing_authority_id", "effort_mapping", ], ); @@ -592,6 +610,7 @@ fn model_one_of_variant(strict: bool) -> Value { "auto_prompt_caching", "cost", "pricing_key", + "pricing_authority_id", "effort_mapping", ], ); @@ -694,6 +713,23 @@ mod tests { assert_eq!(m.rate_limit.as_ref().unwrap().rpm, Some(100)); } + #[test] + fn pricing_authority_id_round_trips_only_when_set() { + let mut model: Model = serde_json::from_str(sample_json()).unwrap(); + assert!(model.pricing_authority_id.is_none()); + assert!(serde_json::to_value(&model) + .unwrap() + .get("pricing_authority_id") + .is_none()); + + model.pricing_authority_id = Some("a3ebdc63-e921-4323-a75c-3b911f950046".to_string()); + let encoded = serde_json::to_value(&model).unwrap(); + assert_eq!( + encoded["pricing_authority_id"], + serde_json::json!("a3ebdc63-e921-4323-a75c-3b911f950046") + ); + } + #[test] fn deserialises_stream_timeout_and_helpers_fold_zero() { let m: Model = serde_json::from_str( @@ -762,11 +798,20 @@ mod tests { "retries": 2, "timeout": 1000, "cost": {"input_per_1k": 0.0, "output_per_1k": 0.0}, - "auto_prompt_caching": {"enabled": true} + "auto_prompt_caching": {"enabled": true}, + "pricing_authority_id": "a3ebdc63-e921-4323-a75c-3b911f950046" })); let mut stripped = group.strip_kind_inapplicable(); stripped.sort_unstable(); - assert_eq!(stripped, ["auto_prompt_caching", "cost", "retries"]); + assert_eq!( + stripped, + [ + "auto_prompt_caching", + "cost", + "pricing_authority_id", + "retries" + ] + ); assert!(group.retries.is_none() && group.cost.is_none()); assert_eq!(group.timeout, Some(1000)); // Semantic parent: timeout/retries are the group slots and stay. @@ -780,9 +825,13 @@ mod tests { }, "retries": 2, "timeout": 1000, - "cost": {"input_per_1k": 0.0, "output_per_1k": 0.0} + "cost": {"input_per_1k": 0.0, "output_per_1k": 0.0}, + "pricing_authority_id": "a3ebdc63-e921-4323-a75c-3b911f950046" })); - assert_eq!(sem.strip_kind_inapplicable(), ["cost"]); + assert_eq!( + sem.strip_kind_inapplicable(), + ["cost", "pricing_authority_id"] + ); assert_eq!(sem.retries, Some(2)); assert_eq!(sem.timeout, Some(1000)); // Direct: nothing strips. @@ -792,10 +841,15 @@ mod tests { "model_name": "gpt-4o", "provider_key_id": "pk-1", "retries": 2, - "cost": {"input_per_1k": 0.0, "output_per_1k": 0.0} + "cost": {"input_per_1k": 0.0, "output_per_1k": 0.0}, + "pricing_authority_id": "a3ebdc63-e921-4323-a75c-3b911f950046" })); assert!(direct.strip_kind_inapplicable().is_empty()); assert_eq!(direct.retries, Some(2)); + assert_eq!( + direct.pricing_authority_id.as_deref(), + Some("a3ebdc63-e921-4323-a75c-3b911f950046") + ); // Ensemble parent: the whole generic set strips (its own // deadline knob is `ensemble.timeout_ms`). let mut ens = load(serde_json::json!({ @@ -804,7 +858,9 @@ mod tests { "timeout": 1000, "stream_timeout": 500, "retries": 1, - "cost": {"input_per_1k": 0.0, "output_per_1k": 0.0} + "cost": {"input_per_1k": 0.0, "output_per_1k": 0.0}, + "pricing_authority_id": "a3ebdc63-e921-4323-a75c-3b911f950046", + "pricing_key": "catalog-gpt" })); // Asserted WITHOUT a pre-sort: the strip output is already // lexicographic (a pure-strip loader row keeps the fields @@ -812,7 +868,14 @@ mod tests { let ens_stripped = ens.strip_kind_inapplicable(); assert_eq!( ens_stripped, - ["cost", "retries", "stream_timeout", "timeout"] + [ + "cost", + "pricing_authority_id", + "pricing_key", + "retries", + "stream_timeout", + "timeout" + ] ); } diff --git a/crates/aisix-core/src/models/schema.rs b/crates/aisix-core/src/models/schema.rs index 3b394ab0..fa0a52c0 100644 --- a/crates/aisix-core/src/models/schema.rs +++ b/crates/aisix-core/src/models/schema.rs @@ -845,6 +845,19 @@ pub fn model_root_schema(strict: bool) -> Value { .as_object_mut() .expect("model root schema is a JSON object") .insert("oneOf".to_string(), one_of); + if strict { + // CP-issued pricing authorities are canonical UUIDs, but UUID nil is + // not an authority. Keep this write-only so an already-projected + // legacy row still loads and the DP can safely emit it unpriced. + schema + .pointer_mut("/properties/pricing_authority_id") + .and_then(Value::as_object_mut) + .expect("model schema declares pricing_authority_id") + .insert( + "not".to_string(), + json!({"const": "00000000-0000-0000-0000-000000000000"}), + ); + } // `OnEmbeddingFailure` is `#[serde(untagged)]` with an object variant // (`{ "target": … }`): serde buffers untagged content and silently // swallows unknown fields inside it, invisible to the write path's @@ -6065,10 +6078,12 @@ mod tests { "display_name": "g", "routing": {"strategy": "failover", "targets": [{"model": "m"}]}, "retries": 3, - "cost": {"input_per_1k": 0.5, "output_per_1k": 1.5} + "cost": {"input_per_1k": 0.5, "output_per_1k": 1.5}, + "pricing_authority_id": "a3ebdc63-e921-4323-a75c-3b911f950046" }); let msg = validate_model(&group).unwrap_err().message; assert!(msg.contains("`cost`"), "{msg}"); + assert!(msg.contains("`pricing_authority_id`"), "{msg}"); assert!(msg.contains("`retries`"), "{msg}"); assert!(msg.contains("model group"), "{msg}"); assert!( @@ -6114,6 +6129,42 @@ mod tests { assert!(!msg.contains("semantic router"), "{msg}"); } + #[test] + fn pricing_authority_id_is_canonical_non_nil_on_write_and_lenient_on_read() { + let base = json!({ + "display_name": "catalog/*", + "provider": "openai", + "model_name": "*", + "provider_key_id": "pk" + }); + let mut valid = base.clone(); + valid["pricing_authority_id"] = json!("a3ebdc63-e921-4323-a75c-3b911f950046"); + validate_model(&valid).expect("canonical authority UUID writes"); + + for invalid in [ + "00000000-0000-0000-0000-000000000000", + "A3EBDC63-E921-4323-A75C-3B911F950046", + "not-a-uuid", + ] { + let mut model = base.clone(); + model["pricing_authority_id"] = json!(invalid); + assert!( + validate_model(&model).is_err(), + "strict write must reject {invalid:?}" + ); + } + + let mut legacy = base; + legacy["pricing_authority_id"] = json!("00000000-0000-0000-0000-000000000000"); + validate_model_lenient(&legacy).expect("legacy nil authority still loads unpriced"); + + let schema = model_root_schema(true); + assert_eq!( + schema["properties"]["pricing_authority_id"]["maxLength"], + json!(64) + ); + } + #[test] fn model_non_dead_knob_failures_keep_the_generic_message() { // A failure that is NOT a dead knob must not be relabelled: the diff --git a/crates/aisix-obs/src/usage.rs b/crates/aisix-obs/src/usage.rs index d97dd28a..2ec0b2c3 100644 --- a/crates/aisix-obs/src/usage.rs +++ b/crates/aisix-obs/src/usage.rs @@ -123,6 +123,12 @@ pub struct UsageEvent { #[serde(default, skip_serializing_if = "String::is_empty")] pub resolved_pricing_model: String, + /// Canonical non-nil CP-issued UUID paired with `resolved_pricing_model`. + /// Both values are absent for older DP configuration or any request that + /// did not dispatch a concrete wildcard-template model. + #[serde(default, skip_serializing_if = "String::is_empty")] + pub pricing_authority_id: String, + pub prompt_tokens: u32, pub completion_tokens: u32, @@ -1608,6 +1614,7 @@ mod tests { api_key_id: "ak-uuid".into(), requested_model: "smart-group".into(), resolved_pricing_model: "gpt-4o-2024-08-06".into(), + pricing_authority_id: "a3ebdc63-e921-4323-a75c-3b911f950046".into(), prompt_tokens: 12, completion_tokens: 34, upstream_latency_ms: 56, @@ -1623,6 +1630,7 @@ mod tests { // so the dashboard can show the group a routed request used. assert!(json.contains(r#""requested_model":"smart-group""#)); assert!(json.contains(r#""resolved_pricing_model":"gpt-4o-2024-08-06""#)); + assert!(json.contains(r#""pricing_authority_id":"a3ebdc63-e921-4323-a75c-3b911f950046""#)); assert!(json.contains(r#""prompt_tokens":12"#)); assert!(json.contains(r#""completion_tokens":34"#)); assert!(json.contains(r#""guardrail_blocked":false"#)); @@ -1644,6 +1652,7 @@ mod tests { assert!(!json.contains("reasoning_tokens")); assert!(!json.contains("cache_creation_tokens")); assert!(!json.contains("cache_read_tokens")); + assert!(!json.contains("pricing_authority_id")); assert!(!json.contains("provider_request_id")); assert!(!json.contains("provider_model_version")); assert!(!json.contains("finish_reason")); diff --git a/crates/aisix-proxy/src/attribution.rs b/crates/aisix-proxy/src/attribution.rs index 204a9516..0c5f7157 100644 --- a/crates/aisix-proxy/src/attribution.rs +++ b/crates/aisix-proxy/src/attribution.rs @@ -89,6 +89,10 @@ pub(crate) struct Resolved { /// names a wildcard row. Telemetry uses this only for an event that /// actually dispatched that same row; it is not an access-log identity. pub wildcard_pricing_model_id: String, + /// Canonical non-nil UUID issued by CP that makes the concrete model below + /// billable. This stays beside the captured model id so a terminal emitter + /// can either send the complete authority tuple or omit it entirely. + pub wildcard_pricing_authority_id: String, pub wildcard_pricing_model: String, /// Which cache layer answered this request, once one has — `Some` /// exactly when the response came out of the cache. @@ -696,18 +700,55 @@ pub(crate) fn note_target(model: &Model, provider_key_id: &str) { }); } -/// Record the concrete model a caller-addressed wildcard row resolved to. +/// Record the complete pricing authority a caller-addressed wildcard row +/// resolved to. /// /// This is deliberately separate from [`note_target`]: an exact request for /// the literal wildcard row also has an upstream model name, but it never -/// passed wildcard capture and must not be used as a pricing identity. -pub(crate) fn note_wildcard_pricing_identity(model_id: &str, concrete_model: &str) { +/// passed wildcard capture and must not be used as a pricing identity. The +/// authority is optional for rolling upgrades; without it, retain nothing so +/// the terminal event cannot assert a concrete wildcard price. +pub(crate) fn note_wildcard_pricing_identity( + model_id: &str, + pricing_authority_id: Option<&str>, + concrete_model: &str, +) { + let Some(pricing_authority_id) = pricing_authority_id.filter(|id| !id.is_empty()) else { + return; + }; + if model_id.is_empty() + || !valid_wildcard_pricing_model(concrete_model) + || !valid_pricing_authority_id(pricing_authority_id) + { + return; + } with(|r| { r.wildcard_pricing_model_id = model_id.to_string(); + r.wildcard_pricing_authority_id = pricing_authority_id.to_string(); r.wildcard_pricing_model = concrete_model.to_string(); }); } +// Keep this in step with AISIX Cloud's model-pricing name bound. A wildcard +// capture can be a valid upstream model name while still being too long to +// become a safe CP pricing lookup key; omit the entire optional tuple so its +// terminal parent event remains observable and explicitly unpriced. +const MAX_WILDCARD_PRICING_MODEL_CHARS: usize = 120; + +fn valid_wildcard_pricing_model(model: &str) -> bool { + !model.is_empty() + && !model.contains('*') + && !model.contains('\0') + && model.chars().count() <= MAX_WILDCARD_PRICING_MODEL_CHARS +} + +fn valid_pricing_authority_id(id: &str) -> bool { + match uuid::Uuid::parse_str(id) { + Ok(parsed) => !parsed.is_nil() && parsed.to_string() == id, + Err(_) => false, + } +} + /// Overwrite the target half with what a CACHE HIT may honestly claim. /// /// A hit contacts no upstream, so nothing was dispatched to and the line @@ -729,6 +770,7 @@ pub(crate) fn note_cache_hit_entry(entry: &Model, hit_layer: &'static str) { with(|r| { r.cache_hit_layer = Some(hit_layer); r.wildcard_pricing_model_id.clear(); + r.wildcard_pricing_authority_id.clear(); r.wildcard_pricing_model.clear(); }); } diff --git a/crates/aisix-proxy/src/messages.rs b/crates/aisix-proxy/src/messages.rs index 8f3cad94..7e96bb56 100644 --- a/crates/aisix-proxy/src/messages.rs +++ b/crates/aisix-proxy/src/messages.rs @@ -5309,7 +5309,7 @@ data: [DONE]\n\n"; } #[tokio::test] - async fn non_anthropic_streaming_records_anthropic_usage_event_with_ttft() { + async fn wildcard_streaming_records_anthropic_usage_event_with_pricing_authority_and_ttft() { use aisix_obs::UsageSink; use aisix_provider_openai::OpenAiBridge; @@ -5331,7 +5331,16 @@ data: [DONE]\n\n"; let (tx, mut rx) = tokio::sync::mpsc::channel(4); let snap = new_snap_openai(&upstream.uri()); - snap.models.insert(openai_model("my-claude-alias")); + let wildcard: Model = serde_json::from_value(serde_json::json!({ + "display_name": "my-claude/*", + "provider": "openai", + "model_name": "*", + "provider_key_id": OPENAI_PK_ID, + "pricing_authority_id": "a3ebdc63-e921-4323-a75c-3b911f950046", + })) + .expect("wildcard model parses"); + snap.models + .insert(ResourceEntry::new("m-wildcard", wildcard, 1)); snap.apikeys.insert(apikey_entry(&["*"])); let hub = Arc::new(Hub::new()); @@ -5344,7 +5353,7 @@ data: [DONE]\n\n"; let app = crate::build_router(state); let body = serde_json::json!({ - "model": "my-claude-alias", + "model": "my-claude/gpt-4o-2024-08-06", "messages": [{"role": "user", "content": "hi"}], "max_tokens": 100, "stream": true, @@ -5361,6 +5370,12 @@ data: [DONE]\n\n"; .expect("usage event was never emitted") .expect("usage event sender dropped"); assert_eq!(event.inbound_protocol, "anthropic"); + assert_eq!(event.model_id, "m-wildcard"); + assert_eq!( + event.pricing_authority_id, + "a3ebdc63-e921-4323-a75c-3b911f950046" + ); + assert_eq!(event.resolved_pricing_model, "gpt-4o-2024-08-06"); assert_eq!(event.prompt_tokens, 13); assert_eq!(event.completion_tokens, 4); assert_eq!(event.provider_request_id, "cmpl-359"); diff --git a/crates/aisix-proxy/src/model_resolve.rs b/crates/aisix-proxy/src/model_resolve.rs index 4e99841e..14dc4858 100644 --- a/crates/aisix-proxy/src/model_resolve.rs +++ b/crates/aisix-proxy/src/model_resolve.rs @@ -37,7 +37,7 @@ pub(crate) fn resolve_model( } let (entry, upstream) = best_wildcard_row(snapshot, requested)?; crate::attribution::note_requested_model(requested); - // A wildcard row is a direct model, and attribution stays on the ROW + // A wildcard row is direct-shaped, and attribution stays on the ROW // (see the module docs), so the synthetic clone below inherits its id. note_dispatchable_entry(&entry); // Keep the concrete value separate from ordinary target attribution: @@ -47,7 +47,11 @@ pub(crate) fn resolve_model( // emitter may see a refreshed configuration where the row changed or was // deleted, but it must price the concrete model this request dispatched. if wildcard_pricing_eligible(&entry.value, &upstream) { - crate::attribution::note_wildcard_pricing_identity(&entry.id, &upstream); + crate::attribution::note_wildcard_pricing_identity( + &entry.id, + entry.value.pricing_authority_id.as_deref(), + &upstream, + ); } let mut model = entry.value.clone(); model.model_name = Some(upstream); @@ -85,8 +89,9 @@ fn best_wildcard_row( if !model.display_name.contains('*') { continue; } - // Only direct Models can serve a wildcard alias — routers / ensembles / - // semantic routers have no upstream `model_name` to dispatch. + // Only direct-shaped Models can serve a wildcard alias — routers / + // ensembles / semantic routers have no upstream `model_name` to + // dispatch. if model.is_routing() || model.is_ensemble() || model.is_semantic() { continue; } @@ -109,7 +114,7 @@ fn best_wildcard_row( /// body (a client-supplied video id) and must not be echoed back as though /// the gateway had attested it. pub(crate) fn row_serves_name(model: &Model, requested: &str) -> bool { - // The same kind gate `best_wildcard_row` applies: only a direct row can + // The same kind gate `best_wildcard_row` applies: only a direct-shaped row can // serve a caller-minted alias. Unreachable today on the one surface that // calls this — `dispatch::require_provider` rejects those kinds first — // but this sits beside the function it mirrors, and it judges an entry @@ -198,6 +203,12 @@ mod tests { .unwrap() } + fn priced_direct_model(display_name: &str, model_name: Option<&str>) -> Model { + let mut model = direct_model(display_name, model_name); + model.pricing_authority_id = Some("a3ebdc63-e921-4323-a75c-3b911f950046".to_string()); + model + } + /// `row_serves_name` gates a name that did NOT arrive on a live request /// body — it decides whether the gateway will echo a caller-supplied /// string back as its own `model`. A row must accept every name it @@ -265,16 +276,30 @@ mod tests { /// Pricing eligibility belongs to the snapshot that dispatched the /// request. In particular, a wildcard display alias over a fixed model, - /// a literal wildcard row, and a capture that is still itself a wildcard - /// must not manufacture a concrete provider-model price identity. + /// a literal wildcard row, a capture that is still itself a wildcard, or + /// an absent, nil, or noncanonical pricing authority must not manufacture + /// a concrete provider-model price identity. #[tokio::test] - async fn only_concrete_template_capture_sets_wildcard_pricing_identity() { + async fn only_concrete_template_capture_with_authority_sets_wildcard_pricing_identity() { use std::sync::Arc; let snap = snapshot_with(vec![ - ("wildcard", direct_model("openrouter/*", Some("*"))), - ("fixed", direct_model("fixed/*", Some("gpt-4o"))), + ("wildcard", priced_direct_model("openrouter/*", Some("*"))), + ("fixed", priced_direct_model("fixed/*", Some("gpt-4o"))), + ("unpriced", direct_model("unpriced/*", Some("*"))), ]); + let mut nil_authority = direct_model("nil/*", Some("*")); + nil_authority.pricing_authority_id = Some("00000000-0000-0000-0000-000000000000".into()); + snap.models + .insert(ResourceEntry::new("nil", nil_authority, 1)); + let mut noncanonical_authority = direct_model("noncanonical/*", Some("*")); + noncanonical_authority.pricing_authority_id = + Some("A3EBDC63-E921-4323-A75C-3B911F950046".into()); + snap.models.insert(ResourceEntry::new( + "noncanonical", + noncanonical_authority, + 1, + )); crate::attribution::scope( Arc::new(crate::attribution::RequestAttribution::default()), @@ -283,23 +308,51 @@ mod tests { .expect("concrete wildcard request resolves"); let resolved = crate::attribution::current().expect("in request scope"); assert_eq!(resolved.wildcard_pricing_model_id, "wildcard"); + assert_eq!( + resolved.wildcard_pricing_authority_id, + "a3ebdc63-e921-4323-a75c-3b911f950046" + ); assert_eq!(resolved.wildcard_pricing_model, "gpt-4o-2024-08-06"); }, ) .await; - for requested in ["fixed/anything", "openrouter/*", "openrouter/gpt-*"] { + for requested in [ + "fixed/anything", + "openrouter/*", + "openrouter/gpt-*", + "unpriced/gpt-4o", + "nil/gpt-4o", + "noncanonical/gpt-4o", + ] { crate::attribution::scope( Arc::new(crate::attribution::RequestAttribution::default()), async { resolve_model(&snap, requested).expect("configured request resolves"); let resolved = crate::attribution::current().expect("in request scope"); assert!(resolved.wildcard_pricing_model_id.is_empty(), "{requested}"); + assert!( + resolved.wildcard_pricing_authority_id.is_empty(), + "{requested}" + ); assert!(resolved.wildcard_pricing_model.is_empty(), "{requested}"); }, ) .await; } + + let overlong = format!("openrouter/{}", "界".repeat(121)); + crate::attribution::scope( + Arc::new(crate::attribution::RequestAttribution::default()), + async { + resolve_model(&snap, &overlong).expect("overlong wildcard request still resolves"); + let resolved = crate::attribution::current().expect("in request scope"); + assert!(resolved.wildcard_pricing_model_id.is_empty()); + assert!(resolved.wildcard_pricing_authority_id.is_empty()); + assert!(resolved.wildcard_pricing_model.is_empty()); + }, + ) + .await; } #[test] diff --git a/crates/aisix-proxy/src/realtime.rs b/crates/aisix-proxy/src/realtime.rs index 9dd803ff..ccd59c77 100644 --- a/crates/aisix-proxy/src/realtime.rs +++ b/crates/aisix-proxy/src/realtime.rs @@ -1625,6 +1625,7 @@ mod tests { "provider": "openai", "model_name": "gpt-realtime-*", "provider_key_id": PK_ID, + "pricing_authority_id": "a3ebdc63-e921-4323-a75c-3b911f950046", })) .unwrap(); snap.models.insert(ResourceEntry::new("m-rt", wildcard, 2)); @@ -1658,6 +1659,10 @@ mod tests { .expect("usage sink closed"); assert_eq!(event.model_id, "m-rt"); assert_eq!(event.requested_model, "rt-2026-01-01"); + assert_eq!( + event.pricing_authority_id, + "a3ebdc63-e921-4323-a75c-3b911f950046" + ); assert_eq!(event.resolved_pricing_model, "gpt-realtime-2026-01-01"); } diff --git a/crates/aisix-proxy/src/request_metrics.rs b/crates/aisix-proxy/src/request_metrics.rs index 0aeab00d..09925344 100644 --- a/crates/aisix-proxy/src/request_metrics.rs +++ b/crates/aisix-proxy/src/request_metrics.rs @@ -705,6 +705,7 @@ mod tests { upstream_model: "gpt-4o-mini".to_string(), provider_key_id: "pk-1".to_string(), wildcard_pricing_model_id: String::new(), + wildcard_pricing_authority_id: String::new(), wildcard_pricing_model: String::new(), cache_hit_layer: None, } @@ -772,6 +773,7 @@ mod tests { "provider": "openai", "model_name": "*", "provider_key_id": "pk-1", + "pricing_authority_id": "a3ebdc63-e921-4323-a75c-3b911f950046", }), ); let refreshed = snapshot_with( @@ -792,12 +794,20 @@ mod tests { let captured = crate::attribution::current().expect("request attribution is installed"); assert_eq!(captured.wildcard_pricing_model_id, "wildcard"); + assert_eq!( + captured.wildcard_pricing_authority_id, + "a3ebdc63-e921-4323-a75c-3b911f950046" + ); assert_eq!(captured.wildcard_pricing_model, "gpt-4o"); assert!(!is_ensemble(&refreshed, "openrouter/gpt-4o")); let after_metrics = crate::attribution::current().expect("request attribution is installed"); assert_eq!(after_metrics.wildcard_pricing_model_id, "wildcard"); + assert_eq!( + after_metrics.wildcard_pricing_authority_id, + "a3ebdc63-e921-4323-a75c-3b911f950046" + ); assert_eq!(after_metrics.wildcard_pricing_model, "gpt-4o"); }, ) diff --git a/crates/aisix-proxy/src/usage_attr.rs b/crates/aisix-proxy/src/usage_attr.rs index d69012f4..bb186e11 100644 --- a/crates/aisix-proxy/src/usage_attr.rs +++ b/crates/aisix-proxy/src/usage_attr.rs @@ -439,14 +439,13 @@ pub(crate) fn metric_model_label_pair<'a>( } } -/// Fill the optional DP-to-CP wildcard-pricing identity at the one usage +/// Fill the optional DP-to-CP wildcard-pricing authority at the one usage /// emission chokepoint. Model resolution establishes eligibility from the -/// dispatch snapshot and records only a concrete provider-model identity, so -/// this path must not consult an emission-time snapshot that may have changed -/// while a stream or realtime session was still running. A cache hit or -/// pre-dispatch failure has no upstream call to price. Detached gateway work -/// has no caller attribution and therefore cannot price itself as the parent -/// request. +/// dispatch snapshot and records a complete authority tuple, so this path must +/// not consult an emission-time snapshot that may have changed while a stream +/// or realtime session was still running. A cache hit or pre-dispatch failure +/// has no upstream call to price. Detached gateway work has no caller +/// attribution and therefore cannot price itself as the parent request. fn apply_wildcard_pricing_model(event: &mut UsageEvent, surface: Surface, dispatched: bool) { // Batch-management rows are observable upstream management requests, not // the batch's model inference. They stay unpriced even when their route @@ -461,7 +460,10 @@ fn apply_wildcard_pricing_model(event: &mut UsageEvent, surface: Surface, dispat if resolved.cache_hit_layer.is_some() || resolved.wildcard_pricing_model_id != event.model_id { return; } - if !resolved.wildcard_pricing_model.is_empty() { + if !resolved.wildcard_pricing_authority_id.is_empty() + && !resolved.wildcard_pricing_model.is_empty() + { + event.pricing_authority_id = resolved.wildcard_pricing_authority_id; event.resolved_pricing_model = resolved.wildcard_pricing_model; } } @@ -1069,7 +1071,7 @@ mod tests { } #[tokio::test] - async fn request_attribution_stamps_only_concrete_wildcard_pricing_models() { + async fn request_attribution_stamps_only_concrete_wildcard_pricing_authorities() { use aisix_core::resource::ResourceEntry; use aisix_core::snapshot::ResourceTable; @@ -1079,6 +1081,7 @@ mod tests { "provider": "openai", "model_name": "*", "provider_key_id": "pk-1", + "pricing_authority_id": "a3ebdc63-e921-4323-a75c-3b911f950046", })) .unwrap(); let cache_entry = configured.clone(); @@ -1097,6 +1100,10 @@ mod tests { let attribution = crate::attribution::current().expect("in request attribution scope"); assert_eq!(attribution.wildcard_pricing_model_id, "wildcard"); + assert_eq!( + attribution.wildcard_pricing_authority_id, + "a3ebdc63-e921-4323-a75c-3b911f950046" + ); assert_eq!(attribution.wildcard_pricing_model, "gpt-4o-2024-08-06"); let mut event = UsageEvent { // NO-GUARDRAIL-CHAIN: this focused unit test constructs @@ -1107,6 +1114,10 @@ mod tests { ..Default::default() }; apply_wildcard_pricing_model(&mut event, crate::operation::CHAT, true); + assert_eq!( + event.pricing_authority_id, + "a3ebdc63-e921-4323-a75c-3b911f950046" + ); assert_eq!(event.resolved_pricing_model, "gpt-4o-2024-08-06"); let mut batch_event = UsageEvent { @@ -1118,6 +1129,7 @@ mod tests { ..Default::default() }; apply_wildcard_pricing_model(&mut batch_event, crate::operation::BATCHES, true); + assert!(batch_event.pricing_authority_id.is_empty()); assert!(batch_event.resolved_pricing_model.is_empty()); crate::attribution::note_cache_hit_entry(&cache_entry, "exact"); @@ -1128,7 +1140,11 @@ mod tests { // `note_cache_hit_entry` clears the identity above. Restore // one here to pin the separate emission gate too: a cache // hit is never billable as an upstream wildcard dispatch. - crate::attribution::note_wildcard_pricing_identity("wildcard", "gpt-4o-2024-08-06"); + crate::attribution::note_wildcard_pricing_identity( + "wildcard", + Some("a3ebdc63-e921-4323-a75c-3b911f950046"), + "gpt-4o-2024-08-06", + ); let mut cached_event = UsageEvent { // NO-GUARDRAIL-CHAIN: this focused unit test constructs // a synthetic pricing event, not a gateway request. @@ -1138,6 +1154,7 @@ mod tests { ..Default::default() }; apply_wildcard_pricing_model(&mut cached_event, crate::operation::CHAT, true); + assert!(cached_event.pricing_authority_id.is_empty()); assert!(cached_event.resolved_pricing_model.is_empty()); }, ) @@ -1152,6 +1169,7 @@ mod tests { let attribution = crate::attribution::current().expect("in request attribution scope"); assert!(attribution.wildcard_pricing_model_id.is_empty()); + assert!(attribution.wildcard_pricing_authority_id.is_empty()); assert!(attribution.wildcard_pricing_model.is_empty()); let mut event = UsageEvent { @@ -1165,10 +1183,49 @@ mod tests { ..Default::default() }; apply_wildcard_pricing_model(&mut event, crate::operation::CHAT, true); + assert!(event.pricing_authority_id.is_empty()); assert!(event.resolved_pricing_model.is_empty()); }, ) .await; + + // A configuration projected by an older control plane has the + // wildcard template but no authority UUID. It may dispatch, but its + // terminal event must not claim a concrete catalog price. + let legacy_table = ResourceTable::default(); + let mut legacy_configured = cache_entry; + legacy_configured.pricing_authority_id = None; + legacy_table.insert(ResourceEntry::new("legacy", legacy_configured, 1)); + let legacy_snap = AisixSnapshot { + models: legacy_table, + ..Default::default() + }; + crate::attribution::scope( + Arc::new(crate::attribution::RequestAttribution::default()), + async { + crate::model_resolve::resolve_model(&legacy_snap, "openrouter/gpt-4o-2024-08-06") + .expect("legacy wildcard model resolves"); + let attribution = + crate::attribution::current().expect("in request attribution scope"); + assert!(attribution.wildcard_pricing_model_id.is_empty()); + assert!(attribution.wildcard_pricing_authority_id.is_empty()); + assert!(attribution.wildcard_pricing_model.is_empty()); + + let mut event = UsageEvent { + // NO-GUARDRAIL-CHAIN: this focused test constructs a + // synthetic terminal usage event after dispatch. + model_id: "legacy".to_string(), + guardrail_bypassed_reason: String::new(), + applied_guardrails: Vec::new(), + ..Default::default() + }; + apply_wildcard_pricing_model(&mut event, crate::operation::CHAT, true); + let wire = serde_json::to_value(event).expect("usage event serialises"); + assert!(wire.get("pricing_authority_id").is_none()); + assert!(wire.get("resolved_pricing_model").is_none()); + }, + ) + .await; } /// A stream can outlive the configuration generation that dispatched it. @@ -1189,6 +1246,7 @@ mod tests { "provider": "openai", "model_name": template, "provider_key_id": "pk-1", + "pricing_authority_id": "a3ebdc63-e921-4323-a75c-3b911f950046", })) .unwrap(); table.insert(ResourceEntry::new("wildcard", model, 1)); @@ -1261,6 +1319,10 @@ mod tests { event.resolved_pricing_model, "gpt-4o-2024-08-06", "{case} must not rewrite the dispatched wildcard identity" ); + assert_eq!( + event.pricing_authority_id, "a3ebdc63-e921-4323-a75c-3b911f950046", + "{case} must not rewrite the dispatched wildcard authority" + ); } } @@ -1280,6 +1342,7 @@ mod tests { "provider": "openai", "model_name": "*", "provider_key_id": "pk-1", + "pricing_authority_id": "a3ebdc63-e921-4323-a75c-3b911f950046", })) .unwrap(); table.insert(ResourceEntry::new("wildcard", configured, 1)); @@ -1349,6 +1412,7 @@ mod tests { let event = rx.try_recv().expect("failed attempt emits usage"); assert_eq!(event.model_id, "wildcard"); + assert!(event.pricing_authority_id.is_empty()); assert!(event.resolved_pricing_model.is_empty()); } diff --git a/schemas/resources-lenient/model.schema.json b/schemas/resources-lenient/model.schema.json index b256dbe2..a956a9f1 100644 --- a/schemas/resources-lenient/model.schema.json +++ b/schemas/resources-lenient/model.schema.json @@ -1365,6 +1365,13 @@ "minLength": 1, "type": "string" }, + "pricing_authority_id": { + "description": "Opaque control-plane-issued canonical, non-nil UUID that authorizes a concrete wildcard-upstream model name for pricing. It is relevant only to a direct-shaped wildcard model (chat or embedding); without it the data plane deliberately omits the resolved model from terminal telemetry so the control plane can leave the call unpriced rather than trust mutable configuration.", + "maxLength": 64, + "minLength": 36, + "pattern": "^[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}$", + "type": "string" + }, "pricing_key": { "description": "Name of a shared `pricing` document to take the per-token cost from, instead of setting `cost` on this model. The environment's own pricing documents are searched first and the shared catalog second; when the key matches neither, `cost` applies. Editing the pricing document repricies every model naming it, with no change to the models themselves.", "maxLength": 255, diff --git a/schemas/resources/model.schema.json b/schemas/resources/model.schema.json index bfe32d9d..c4fdf649 100644 --- a/schemas/resources/model.schema.json +++ b/schemas/resources/model.schema.json @@ -1189,6 +1189,11 @@ "pricing_key" ] }, + { + "required": [ + "pricing_authority_id" + ] + }, { "required": [ "effort_mapping" @@ -1305,6 +1310,11 @@ "pricing_key" ] }, + { + "required": [ + "pricing_authority_id" + ] + }, { "required": [ "effort_mapping" @@ -1374,6 +1384,11 @@ "pricing_key" ] }, + { + "required": [ + "pricing_authority_id" + ] + }, { "required": [ "effort_mapping" @@ -1471,6 +1486,16 @@ "minLength": 1, "type": "string" }, + "pricing_authority_id": { + "description": "Opaque control-plane-issued canonical, non-nil UUID that authorizes a concrete wildcard-upstream model name for pricing. It is relevant only to a direct-shaped wildcard model (chat or embedding); without it the data plane deliberately omits the resolved model from terminal telemetry so the control plane can leave the call unpriced rather than trust mutable configuration.", + "maxLength": 64, + "minLength": 36, + "not": { + "const": "00000000-0000-0000-0000-000000000000" + }, + "pattern": "^[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}$", + "type": "string" + }, "pricing_key": { "description": "Name of a shared `pricing` document to take the per-token cost from, instead of setting `cost` on this model. The environment's own pricing documents are searched first and the shared catalog second; when the key matches neither, `cost` applies. Editing the pricing document repricies every model naming it, with no change to the models themselves.", "maxLength": 255, From cae0ccfd48b978f4d90a79f5ba49ded1816234fa Mon Sep 17 00:00:00 2001 From: Ming Wen Date: Wed, 30 Sep 2026 20:33:17 +0800 Subject: [PATCH 07/16] fix(telemetry): emit fixed dispatch pricing authority --- crates/aisix-core/src/models/model.rs | 12 +- .../tests/resource_schema_characterization.rs | 4 + crates/aisix-obs/src/usage.rs | 2 +- crates/aisix-proxy/src/attribution.rs | 57 +++++---- crates/aisix-proxy/src/model_resolve.rs | 87 ++++++++------ crates/aisix-proxy/src/request_metrics.rs | 18 +-- crates/aisix-proxy/src/usage_attr.rs | 110 ++++++++++++++---- schemas/README.md | 2 +- schemas/resources-lenient/model.schema.json | 2 +- schemas/resources/model.schema.json | 2 +- 10 files changed, 189 insertions(+), 107 deletions(-) diff --git a/crates/aisix-core/src/models/model.rs b/crates/aisix-core/src/models/model.rs index b954a82a..e8ee55c1 100644 --- a/crates/aisix-core/src/models/model.rs +++ b/crates/aisix-core/src/models/model.rs @@ -310,12 +310,12 @@ pub struct Model { #[schemars(length(min = 1, max = 255))] pub pricing_key: Option, - /// Opaque control-plane-issued canonical, non-nil UUID that authorizes a - /// concrete wildcard-upstream model name for pricing. It is relevant only - /// to a direct-shaped wildcard model (chat or embedding); without it the - /// data plane deliberately omits the resolved model from terminal - /// telemetry so the control plane can leave the call unpriced rather than - /// trust mutable configuration. + /// Opaque control-plane-issued canonical, non-nil UUID that authorizes the + /// configured exact or wildcard-resolved upstream model name for pricing. + /// It is relevant only to a direct-shaped model (chat or embedding); + /// without it the data plane deliberately omits the resolved model from + /// terminal telemetry so the control plane can leave the call unpriced + /// rather than trust mutable configuration. #[serde(default, skip_serializing_if = "Option::is_none")] #[schemars( regex(pattern = "^[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}$"), diff --git a/crates/aisix-core/tests/resource_schema_characterization.rs b/crates/aisix-core/tests/resource_schema_characterization.rs index 20bc05bf..b90e33c1 100644 --- a/crates/aisix-core/tests/resource_schema_characterization.rs +++ b/crates/aisix-core/tests/resource_schema_characterization.rs @@ -1061,6 +1061,10 @@ const EXTRA_RELAXATIONS: &[(&str, &[&str])] = &[ "/oneOf/3/not/anyOf", "/properties/effort_mapping/additionalProperties/minLength", "/properties/effort_mapping/properties//minLength", + // CP may have projected an older nil authority. The strict contract + // rejects it, while a DP must keep loading that row and emit its + // usage unpriced during a rolling upgrade. + "/properties/pricing_authority_id/not", ], ), ]; diff --git a/crates/aisix-obs/src/usage.rs b/crates/aisix-obs/src/usage.rs index 2ec0b2c3..95eefa23 100644 --- a/crates/aisix-obs/src/usage.rs +++ b/crates/aisix-obs/src/usage.rs @@ -125,7 +125,7 @@ pub struct UsageEvent { /// Canonical non-nil CP-issued UUID paired with `resolved_pricing_model`. /// Both values are absent for older DP configuration or any request that - /// did not dispatch a concrete wildcard-template model. + /// did not dispatch a concrete direct or embedding model. #[serde(default, skip_serializing_if = "String::is_empty")] pub pricing_authority_id: String, diff --git a/crates/aisix-proxy/src/attribution.rs b/crates/aisix-proxy/src/attribution.rs index 0c5f7157..022dfe7f 100644 --- a/crates/aisix-proxy/src/attribution.rs +++ b/crates/aisix-proxy/src/attribution.rs @@ -83,17 +83,16 @@ pub(crate) struct Resolved { /// it at read time, so the pair is byte-identical to the one the /// success path emits. pub provider_key_id: String, - /// The concrete upstream model produced by the wildcard-resolution - /// branch, paired with the configured wildcard row that produced it. - /// Empty for exact model resolution, including a request that literally - /// names a wildcard row. Telemetry uses this only for an event that + /// The concrete upstream model paired with the immutable CP-issued + /// authority that existed at dispatch. Empty means no valid authority was + /// present in this snapshot. Telemetry uses this only for an event that /// actually dispatched that same row; it is not an access-log identity. - pub wildcard_pricing_model_id: String, + pub pricing_model_id: String, /// Canonical non-nil UUID issued by CP that makes the concrete model below /// billable. This stays beside the captured model id so a terminal emitter /// can either send the complete authority tuple or omit it entirely. - pub wildcard_pricing_authority_id: String, - pub wildcard_pricing_model: String, + pub pricing_authority_id: String, + pub pricing_model: String, /// Which cache layer answered this request, once one has — `Some` /// exactly when the response came out of the cache. /// @@ -700,15 +699,12 @@ pub(crate) fn note_target(model: &Model, provider_key_id: &str) { }); } -/// Record the complete pricing authority a caller-addressed wildcard row -/// resolved to. -/// -/// This is deliberately separate from [`note_target`]: an exact request for -/// the literal wildcard row also has an upstream model name, but it never -/// passed wildcard capture and must not be used as a pricing identity. The -/// authority is optional for rolling upgrades; without it, retain nothing so -/// the terminal event cannot assert a concrete wildcard price. -pub(crate) fn note_wildcard_pricing_identity( +/// Record the complete pricing authority a direct or embedding dispatch +/// resolved to. This is deliberately separate from [`note_target`]: it is +/// accounting provenance rather than an access-log identity. The authority is +/// optional for rolling upgrades; without it, retain nothing so the terminal +/// event cannot assert a mutable current-model price. +pub(crate) fn note_pricing_identity( model_id: &str, pricing_authority_id: Option<&str>, concrete_model: &str, @@ -717,29 +713,30 @@ pub(crate) fn note_wildcard_pricing_identity( return; }; if model_id.is_empty() - || !valid_wildcard_pricing_model(concrete_model) + || !valid_pricing_model(concrete_model) || !valid_pricing_authority_id(pricing_authority_id) { return; } with(|r| { - r.wildcard_pricing_model_id = model_id.to_string(); - r.wildcard_pricing_authority_id = pricing_authority_id.to_string(); - r.wildcard_pricing_model = concrete_model.to_string(); + r.pricing_model_id = model_id.to_string(); + r.pricing_authority_id = pricing_authority_id.to_string(); + r.pricing_model = concrete_model.to_string(); }); } -// Keep this in step with AISIX Cloud's model-pricing name bound. A wildcard -// capture can be a valid upstream model name while still being too long to -// become a safe CP pricing lookup key; omit the entire optional tuple so its -// terminal parent event remains observable and explicitly unpriced. -const MAX_WILDCARD_PRICING_MODEL_CHARS: usize = 120; +// Keep this in step with AISIX Cloud's model-pricing name bound. A configured +// or wildcard-resolved upstream name can be valid for dispatch while still +// being too long to become a safe CP pricing lookup key; omit the entire +// optional tuple so its terminal parent event remains observable and +// explicitly unpriced. +const MAX_PRICING_MODEL_CHARS: usize = 120; -fn valid_wildcard_pricing_model(model: &str) -> bool { +fn valid_pricing_model(model: &str) -> bool { !model.is_empty() && !model.contains('*') && !model.contains('\0') - && model.chars().count() <= MAX_WILDCARD_PRICING_MODEL_CHARS + && model.chars().count() <= MAX_PRICING_MODEL_CHARS } fn valid_pricing_authority_id(id: &str) -> bool { @@ -769,9 +766,9 @@ pub(crate) fn note_cache_hit_entry(entry: &Model, hit_layer: &'static str) { note_target(entry, entry.provider_key_id.as_deref().unwrap_or_default()); with(|r| { r.cache_hit_layer = Some(hit_layer); - r.wildcard_pricing_model_id.clear(); - r.wildcard_pricing_authority_id.clear(); - r.wildcard_pricing_model.clear(); + r.pricing_model_id.clear(); + r.pricing_authority_id.clear(); + r.pricing_model.clear(); }); } diff --git a/crates/aisix-proxy/src/model_resolve.rs b/crates/aisix-proxy/src/model_resolve.rs index 14dc4858..69a8b95c 100644 --- a/crates/aisix-proxy/src/model_resolve.rs +++ b/crates/aisix-proxy/src/model_resolve.rs @@ -46,8 +46,13 @@ pub(crate) fn resolve_model( // deliberately decided against the dispatch snapshot: a later terminal // emitter may see a refreshed configuration where the row changed or was // deleted, but it must price the concrete model this request dispatched. - if wildcard_pricing_eligible(&entry.value, &upstream) { - crate::attribution::note_wildcard_pricing_identity( + if entry + .value + .model_name + .as_deref() + .is_some_and(|template| template.contains('*')) + { + crate::attribution::note_pricing_identity( &entry.id, entry.value.pricing_authority_id.as_deref(), &upstream, @@ -74,6 +79,13 @@ fn note_dispatchable_entry(entry: &ResourceEntry) { return; } crate::attribution::note_resolved_entry(&entry.id); + if let Some(upstream) = model.model_name.as_deref() { + crate::attribution::note_pricing_identity( + &entry.id, + model.pricing_authority_id.as_deref(), + upstream, + ); + } } /// Wildcard fallback: the most specific direct Model whose `*`-glob @@ -173,21 +185,6 @@ fn resolve_upstream_model_name(model: &Model, capture: &str) -> String { } } -/// Whether a wildcard dispatch produced a concrete provider-model identity -/// that is safe to send to CP for pricing. -/// -/// A wildcard display alias with a fixed upstream is already priced by its -/// configured model name. Only an upstream template yields a caller-specific -/// provider-model identity, and a literal `*` is never a concrete catalog key. -fn wildcard_pricing_eligible(model: &Model, upstream: &str) -> bool { - model - .model_name - .as_deref() - .is_some_and(|template| template.contains('*')) - && !upstream.is_empty() - && !upstream.contains('*') -} - #[cfg(test)] mod tests { use super::*; @@ -275,17 +272,24 @@ mod tests { } /// Pricing eligibility belongs to the snapshot that dispatched the - /// request. In particular, a wildcard display alias over a fixed model, - /// a literal wildcard row, a capture that is still itself a wildcard, or - /// an absent, nil, or noncanonical pricing authority must not manufacture - /// a concrete provider-model price identity. + /// request. Exact models and wildcard aliases over a fixed upstream carry + /// their exact authority; a literal wildcard row, a capture that is still + /// itself a wildcard, or an absent, nil, or noncanonical authority must + /// not manufacture a concrete provider-model price identity. #[tokio::test] - async fn only_concrete_template_capture_with_authority_sets_wildcard_pricing_identity() { + async fn dispatchable_models_with_authority_set_pricing_identity() { use std::sync::Arc; + let mut embedding = priced_direct_model("embedding", Some("text-embedding-3-small")); + embedding.embedding = Some(aisix_core::EmbeddingConfig { + dimensions: 4, + normalize: true, + }); let snap = snapshot_with(vec![ ("wildcard", priced_direct_model("openrouter/*", Some("*"))), ("fixed", priced_direct_model("fixed/*", Some("gpt-4o"))), + ("exact", priced_direct_model("exact", Some("gpt-4o-mini"))), + ("embedding", embedding), ("unpriced", direct_model("unpriced/*", Some("*"))), ]); let mut nil_authority = direct_model("nil/*", Some("*")); @@ -307,18 +311,17 @@ mod tests { resolve_model(&snap, "openrouter/gpt-4o-2024-08-06") .expect("concrete wildcard request resolves"); let resolved = crate::attribution::current().expect("in request scope"); - assert_eq!(resolved.wildcard_pricing_model_id, "wildcard"); + assert_eq!(resolved.pricing_model_id, "wildcard"); assert_eq!( - resolved.wildcard_pricing_authority_id, + resolved.pricing_authority_id, "a3ebdc63-e921-4323-a75c-3b911f950046" ); - assert_eq!(resolved.wildcard_pricing_model, "gpt-4o-2024-08-06"); + assert_eq!(resolved.pricing_model, "gpt-4o-2024-08-06"); }, ) .await; for requested in [ - "fixed/anything", "openrouter/*", "openrouter/gpt-*", "unpriced/gpt-4o", @@ -330,12 +333,30 @@ mod tests { async { resolve_model(&snap, requested).expect("configured request resolves"); let resolved = crate::attribution::current().expect("in request scope"); - assert!(resolved.wildcard_pricing_model_id.is_empty(), "{requested}"); - assert!( - resolved.wildcard_pricing_authority_id.is_empty(), + assert!(resolved.pricing_model_id.is_empty(), "{requested}"); + assert!(resolved.pricing_authority_id.is_empty(), "{requested}"); + assert!(resolved.pricing_model.is_empty(), "{requested}"); + }, + ) + .await; + } + + for (requested, model_id, model_name) in [ + ("fixed/anything", "fixed", "gpt-4o"), + ("exact", "exact", "gpt-4o-mini"), + ("embedding", "embedding", "text-embedding-3-small"), + ] { + crate::attribution::scope( + Arc::new(crate::attribution::RequestAttribution::default()), + async { + resolve_model(&snap, requested).expect("configured request resolves"); + let resolved = crate::attribution::current().expect("in request scope"); + assert_eq!(resolved.pricing_model_id, model_id, "{requested}"); + assert_eq!( + resolved.pricing_authority_id, "a3ebdc63-e921-4323-a75c-3b911f950046", "{requested}" ); - assert!(resolved.wildcard_pricing_model.is_empty(), "{requested}"); + assert_eq!(resolved.pricing_model, model_name, "{requested}"); }, ) .await; @@ -347,9 +368,9 @@ mod tests { async { resolve_model(&snap, &overlong).expect("overlong wildcard request still resolves"); let resolved = crate::attribution::current().expect("in request scope"); - assert!(resolved.wildcard_pricing_model_id.is_empty()); - assert!(resolved.wildcard_pricing_authority_id.is_empty()); - assert!(resolved.wildcard_pricing_model.is_empty()); + assert!(resolved.pricing_model_id.is_empty()); + assert!(resolved.pricing_authority_id.is_empty()); + assert!(resolved.pricing_model.is_empty()); }, ) .await; diff --git a/crates/aisix-proxy/src/request_metrics.rs b/crates/aisix-proxy/src/request_metrics.rs index 09925344..916ae992 100644 --- a/crates/aisix-proxy/src/request_metrics.rs +++ b/crates/aisix-proxy/src/request_metrics.rs @@ -704,9 +704,9 @@ mod tests { provider: "OpenAI".to_string(), upstream_model: "gpt-4o-mini".to_string(), provider_key_id: "pk-1".to_string(), - wildcard_pricing_model_id: String::new(), - wildcard_pricing_authority_id: String::new(), - wildcard_pricing_model: String::new(), + pricing_model_id: String::new(), + pricing_authority_id: String::new(), + pricing_model: String::new(), cache_hit_layer: None, } } @@ -793,22 +793,22 @@ mod tests { .expect("wildcard model resolves at dispatch"); let captured = crate::attribution::current().expect("request attribution is installed"); - assert_eq!(captured.wildcard_pricing_model_id, "wildcard"); + assert_eq!(captured.pricing_model_id, "wildcard"); assert_eq!( - captured.wildcard_pricing_authority_id, + captured.pricing_authority_id, "a3ebdc63-e921-4323-a75c-3b911f950046" ); - assert_eq!(captured.wildcard_pricing_model, "gpt-4o"); + assert_eq!(captured.pricing_model, "gpt-4o"); assert!(!is_ensemble(&refreshed, "openrouter/gpt-4o")); let after_metrics = crate::attribution::current().expect("request attribution is installed"); - assert_eq!(after_metrics.wildcard_pricing_model_id, "wildcard"); + assert_eq!(after_metrics.pricing_model_id, "wildcard"); assert_eq!( - after_metrics.wildcard_pricing_authority_id, + after_metrics.pricing_authority_id, "a3ebdc63-e921-4323-a75c-3b911f950046" ); - assert_eq!(after_metrics.wildcard_pricing_model, "gpt-4o"); + assert_eq!(after_metrics.pricing_model, "gpt-4o"); }, ) .await; diff --git a/crates/aisix-proxy/src/usage_attr.rs b/crates/aisix-proxy/src/usage_attr.rs index bb186e11..af1f914c 100644 --- a/crates/aisix-proxy/src/usage_attr.rs +++ b/crates/aisix-proxy/src/usage_attr.rs @@ -439,14 +439,14 @@ pub(crate) fn metric_model_label_pair<'a>( } } -/// Fill the optional DP-to-CP wildcard-pricing authority at the one usage +/// Fill the optional DP-to-CP dispatch-pricing authority at the one usage /// emission chokepoint. Model resolution establishes eligibility from the /// dispatch snapshot and records a complete authority tuple, so this path must /// not consult an emission-time snapshot that may have changed while a stream /// or realtime session was still running. A cache hit or pre-dispatch failure /// has no upstream call to price. Detached gateway work has no caller /// attribution and therefore cannot price itself as the parent request. -fn apply_wildcard_pricing_model(event: &mut UsageEvent, surface: Surface, dispatched: bool) { +fn apply_pricing_identity(event: &mut UsageEvent, surface: Surface, dispatched: bool) { // Batch-management rows are observable upstream management requests, not // the batch's model inference. They stay unpriced even when their route // resolves through a wildcard alias; batch completion accounting is a @@ -457,14 +457,12 @@ fn apply_wildcard_pricing_model(event: &mut UsageEvent, surface: Surface, dispat let Some(resolved) = crate::attribution::current() else { return; }; - if resolved.cache_hit_layer.is_some() || resolved.wildcard_pricing_model_id != event.model_id { + if resolved.cache_hit_layer.is_some() || resolved.pricing_model_id != event.model_id { return; } - if !resolved.wildcard_pricing_authority_id.is_empty() - && !resolved.wildcard_pricing_model.is_empty() - { - event.pricing_authority_id = resolved.wildcard_pricing_authority_id; - event.resolved_pricing_model = resolved.wildcard_pricing_model; + if !resolved.pricing_authority_id.is_empty() && !resolved.pricing_model.is_empty() { + event.pricing_authority_id = resolved.pricing_authority_id; + event.resolved_pricing_model = resolved.pricing_model; } } @@ -987,7 +985,7 @@ pub(crate) fn emit_usage( if terminal && event.guardrail_blocked { state.metrics.record_guardrail_blocked_request(); } - apply_wildcard_pricing_model(&mut event, surface, dispatched); + apply_pricing_identity(&mut event, surface, dispatched); let emission = trace.map(|bundle| { event.trace_id = bundle.trace_id_hex(); bundle.emission( @@ -1071,7 +1069,7 @@ mod tests { } #[tokio::test] - async fn request_attribution_stamps_only_concrete_wildcard_pricing_authorities() { + async fn request_attribution_stamps_concrete_dispatch_pricing_authorities() { use aisix_core::resource::ResourceEntry; use aisix_core::snapshot::ResourceTable; @@ -1099,12 +1097,12 @@ mod tests { crate::attribution::note_target(&served.value, "pk-1"); let attribution = crate::attribution::current().expect("in request attribution scope"); - assert_eq!(attribution.wildcard_pricing_model_id, "wildcard"); + assert_eq!(attribution.pricing_model_id, "wildcard"); assert_eq!( - attribution.wildcard_pricing_authority_id, + attribution.pricing_authority_id, "a3ebdc63-e921-4323-a75c-3b911f950046" ); - assert_eq!(attribution.wildcard_pricing_model, "gpt-4o-2024-08-06"); + assert_eq!(attribution.pricing_model, "gpt-4o-2024-08-06"); let mut event = UsageEvent { // NO-GUARDRAIL-CHAIN: this focused unit test constructs // a synthetic pricing event, not a gateway request. @@ -1113,7 +1111,7 @@ mod tests { applied_guardrails: Vec::new(), ..Default::default() }; - apply_wildcard_pricing_model(&mut event, crate::operation::CHAT, true); + apply_pricing_identity(&mut event, crate::operation::CHAT, true); assert_eq!( event.pricing_authority_id, "a3ebdc63-e921-4323-a75c-3b911f950046" @@ -1128,7 +1126,7 @@ mod tests { applied_guardrails: Vec::new(), ..Default::default() }; - apply_wildcard_pricing_model(&mut batch_event, crate::operation::BATCHES, true); + apply_pricing_identity(&mut batch_event, crate::operation::BATCHES, true); assert!(batch_event.pricing_authority_id.is_empty()); assert!(batch_event.resolved_pricing_model.is_empty()); @@ -1140,7 +1138,7 @@ mod tests { // `note_cache_hit_entry` clears the identity above. Restore // one here to pin the separate emission gate too: a cache // hit is never billable as an upstream wildcard dispatch. - crate::attribution::note_wildcard_pricing_identity( + crate::attribution::note_pricing_identity( "wildcard", Some("a3ebdc63-e921-4323-a75c-3b911f950046"), "gpt-4o-2024-08-06", @@ -1153,7 +1151,7 @@ mod tests { applied_guardrails: Vec::new(), ..Default::default() }; - apply_wildcard_pricing_model(&mut cached_event, crate::operation::CHAT, true); + apply_pricing_identity(&mut cached_event, crate::operation::CHAT, true); assert!(cached_event.pricing_authority_id.is_empty()); assert!(cached_event.resolved_pricing_model.is_empty()); }, @@ -1168,9 +1166,9 @@ mod tests { crate::attribution::note_target(&literal.value, "pk-1"); let attribution = crate::attribution::current().expect("in request attribution scope"); - assert!(attribution.wildcard_pricing_model_id.is_empty()); - assert!(attribution.wildcard_pricing_authority_id.is_empty()); - assert!(attribution.wildcard_pricing_model.is_empty()); + assert!(attribution.pricing_model_id.is_empty()); + assert!(attribution.pricing_authority_id.is_empty()); + assert!(attribution.pricing_model.is_empty()); let mut event = UsageEvent { // NO-GUARDRAIL-CHAIN: this focused unit test constructs @@ -1182,7 +1180,7 @@ mod tests { applied_guardrails: Vec::new(), ..Default::default() }; - apply_wildcard_pricing_model(&mut event, crate::operation::CHAT, true); + apply_pricing_identity(&mut event, crate::operation::CHAT, true); assert!(event.pricing_authority_id.is_empty()); assert!(event.resolved_pricing_model.is_empty()); }, @@ -1207,9 +1205,9 @@ mod tests { .expect("legacy wildcard model resolves"); let attribution = crate::attribution::current().expect("in request attribution scope"); - assert!(attribution.wildcard_pricing_model_id.is_empty()); - assert!(attribution.wildcard_pricing_authority_id.is_empty()); - assert!(attribution.wildcard_pricing_model.is_empty()); + assert!(attribution.pricing_model_id.is_empty()); + assert!(attribution.pricing_authority_id.is_empty()); + assert!(attribution.pricing_model.is_empty()); let mut event = UsageEvent { // NO-GUARDRAIL-CHAIN: this focused test constructs a @@ -1219,13 +1217,75 @@ mod tests { applied_guardrails: Vec::new(), ..Default::default() }; - apply_wildcard_pricing_model(&mut event, crate::operation::CHAT, true); + apply_pricing_identity(&mut event, crate::operation::CHAT, true); let wire = serde_json::to_value(event).expect("usage event serialises"); assert!(wire.get("pricing_authority_id").is_none()); assert!(wire.get("resolved_pricing_model").is_none()); }, ) .await; + + // Exact direct and embedding rows take the same immutable authority + // path as a wildcard's captured upstream value. This pins both + // modalities at the terminal-event chokepoint without changing their + // parent model id, target, cost, or budget attribution. + let fixed_table = ResourceTable::default(); + let direct: aisix_core::Model = serde_json::from_value(serde_json::json!({ + "display_name": "fixed-chat", + "provider": "openai", + "model_name": "gpt-4o-mini", + "provider_key_id": "pk-1", + "pricing_authority_id": "a3ebdc63-e921-4323-a75c-3b911f950046", + })) + .unwrap(); + let embedding: aisix_core::Model = serde_json::from_value(serde_json::json!({ + "display_name": "fixed-embedding", + "provider": "openai", + "model_name": "text-embedding-3-small", + "provider_key_id": "pk-1", + "pricing_authority_id": "a3ebdc63-e921-4323-a75c-3b911f950046", + "embedding": {"dimensions": 4}, + })) + .unwrap(); + fixed_table.insert(ResourceEntry::new("fixed-chat", direct, 1)); + fixed_table.insert(ResourceEntry::new("fixed-embedding", embedding, 1)); + let fixed_snap = AisixSnapshot { + models: fixed_table, + ..Default::default() + }; + for (model_id, upstream, surface) in [ + ("fixed-chat", "gpt-4o-mini", crate::operation::CHAT), + ( + "fixed-embedding", + "text-embedding-3-small", + crate::operation::EMBEDDINGS, + ), + ] { + crate::attribution::scope( + Arc::new(crate::attribution::RequestAttribution::default()), + async { + let served = crate::model_resolve::resolve_model(&fixed_snap, model_id) + .expect("fixed direct-shaped model resolves"); + crate::attribution::note_target(&served.value, "pk-1"); + let mut event = UsageEvent { + // NO-GUARDRAIL-CHAIN: this focused test constructs a + // synthetic terminal usage event after dispatch. + model_id: model_id.to_string(), + guardrail_bypassed_reason: String::new(), + applied_guardrails: Vec::new(), + ..Default::default() + }; + apply_pricing_identity(&mut event, surface, true); + assert_eq!(event.model_id, model_id); + assert_eq!( + event.pricing_authority_id, + "a3ebdc63-e921-4323-a75c-3b911f950046" + ); + assert_eq!(event.resolved_pricing_model, upstream); + }, + ) + .await; + } } /// A stream can outlive the configuration generation that dispatched it. diff --git a/schemas/README.md b/schemas/README.md index 5b3938b9..ef66d9cd 100644 --- a/schemas/README.md +++ b/schemas/README.md @@ -116,7 +116,7 @@ so a consumer that models the lenient set as "the strict set with | `guardrail` | the `semantic` kind requires neither an embedding model (under either spelling) nor a threshold beside each example list | | `mcp_policy` | `allow` is not required; the `McpToolRef` relaxations above apply here too, as does the absent `deny`-beside-`deny_ids` guard (its team-scope guard is on both sets) | | `mcp_server` | the label pattern (`name`, and its former spelling `display_name`) still forbids `__` and a trailing `_`, but not a `*` | -| `model` | the per-kind `not`/`anyOf` lists that forbid a knob a kind never resolves are shorter — a stored row keeps loading and `Model::strip_kind_inapplicable` drops the dead knob; and an `effort_mapping` target value may be empty, which the write path refuses | +| `model` | the per-kind `not`/`anyOf` lists that forbid a knob a kind never resolves are shorter — a stored row keeps loading and `Model::strip_kind_inapplicable` drops the dead knob; an `effort_mapping` target value may be empty; and a legacy nil `pricing_authority_id` still loads so its usage can remain explicitly unpriced during a rolling upgrade | Three of those are worth spelling out. A half-written `McpToolRef` entry has to keep DESERIALIZING, not merely validating: the loader skips a row diff --git a/schemas/resources-lenient/model.schema.json b/schemas/resources-lenient/model.schema.json index a956a9f1..333cabf3 100644 --- a/schemas/resources-lenient/model.schema.json +++ b/schemas/resources-lenient/model.schema.json @@ -1366,7 +1366,7 @@ "type": "string" }, "pricing_authority_id": { - "description": "Opaque control-plane-issued canonical, non-nil UUID that authorizes a concrete wildcard-upstream model name for pricing. It is relevant only to a direct-shaped wildcard model (chat or embedding); without it the data plane deliberately omits the resolved model from terminal telemetry so the control plane can leave the call unpriced rather than trust mutable configuration.", + "description": "Opaque control-plane-issued canonical, non-nil UUID that authorizes the configured exact or wildcard-resolved upstream model name for pricing. It is relevant only to a direct-shaped model (chat or embedding); without it the data plane deliberately omits the resolved model from terminal telemetry so the control plane can leave the call unpriced rather than trust mutable configuration.", "maxLength": 64, "minLength": 36, "pattern": "^[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}$", diff --git a/schemas/resources/model.schema.json b/schemas/resources/model.schema.json index c4fdf649..6622a024 100644 --- a/schemas/resources/model.schema.json +++ b/schemas/resources/model.schema.json @@ -1487,7 +1487,7 @@ "type": "string" }, "pricing_authority_id": { - "description": "Opaque control-plane-issued canonical, non-nil UUID that authorizes a concrete wildcard-upstream model name for pricing. It is relevant only to a direct-shaped wildcard model (chat or embedding); without it the data plane deliberately omits the resolved model from terminal telemetry so the control plane can leave the call unpriced rather than trust mutable configuration.", + "description": "Opaque control-plane-issued canonical, non-nil UUID that authorizes the configured exact or wildcard-resolved upstream model name for pricing. It is relevant only to a direct-shaped model (chat or embedding); without it the data plane deliberately omits the resolved model from terminal telemetry so the control plane can leave the call unpriced rather than trust mutable configuration.", "maxLength": 64, "minLength": 36, "not": { From c5fd1de60863e12c12c439f67de76e1d448cb212 Mon Sep 17 00:00:00 2001 From: Ming Wen Date: Thu, 1 Oct 2026 09:46:29 +0800 Subject: [PATCH 08/16] fix: restrict wildcard pricing attribution --- crates/aisix-core/src/models/model.rs | 12 +-- crates/aisix-obs/src/usage.rs | 2 +- crates/aisix-proxy/src/attribution.rs | 57 +++++++------- crates/aisix-proxy/src/model_resolve.rs | 87 ++++++++++----------- crates/aisix-proxy/src/request_metrics.rs | 18 ++--- crates/aisix-proxy/src/usage_attr.rs | 73 ++++++++--------- schemas/resources-lenient/model.schema.json | 2 +- schemas/resources/model.schema.json | 2 +- 8 files changed, 123 insertions(+), 130 deletions(-) diff --git a/crates/aisix-core/src/models/model.rs b/crates/aisix-core/src/models/model.rs index e8ee55c1..b954a82a 100644 --- a/crates/aisix-core/src/models/model.rs +++ b/crates/aisix-core/src/models/model.rs @@ -310,12 +310,12 @@ pub struct Model { #[schemars(length(min = 1, max = 255))] pub pricing_key: Option, - /// Opaque control-plane-issued canonical, non-nil UUID that authorizes the - /// configured exact or wildcard-resolved upstream model name for pricing. - /// It is relevant only to a direct-shaped model (chat or embedding); - /// without it the data plane deliberately omits the resolved model from - /// terminal telemetry so the control plane can leave the call unpriced - /// rather than trust mutable configuration. + /// Opaque control-plane-issued canonical, non-nil UUID that authorizes a + /// concrete wildcard-upstream model name for pricing. It is relevant only + /// to a direct-shaped wildcard model (chat or embedding); without it the + /// data plane deliberately omits the resolved model from terminal + /// telemetry so the control plane can leave the call unpriced rather than + /// trust mutable configuration. #[serde(default, skip_serializing_if = "Option::is_none")] #[schemars( regex(pattern = "^[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}$"), diff --git a/crates/aisix-obs/src/usage.rs b/crates/aisix-obs/src/usage.rs index 95eefa23..2ec0b2c3 100644 --- a/crates/aisix-obs/src/usage.rs +++ b/crates/aisix-obs/src/usage.rs @@ -125,7 +125,7 @@ pub struct UsageEvent { /// Canonical non-nil CP-issued UUID paired with `resolved_pricing_model`. /// Both values are absent for older DP configuration or any request that - /// did not dispatch a concrete direct or embedding model. + /// did not dispatch a concrete wildcard-template model. #[serde(default, skip_serializing_if = "String::is_empty")] pub pricing_authority_id: String, diff --git a/crates/aisix-proxy/src/attribution.rs b/crates/aisix-proxy/src/attribution.rs index 022dfe7f..0c5f7157 100644 --- a/crates/aisix-proxy/src/attribution.rs +++ b/crates/aisix-proxy/src/attribution.rs @@ -83,16 +83,17 @@ pub(crate) struct Resolved { /// it at read time, so the pair is byte-identical to the one the /// success path emits. pub provider_key_id: String, - /// The concrete upstream model paired with the immutable CP-issued - /// authority that existed at dispatch. Empty means no valid authority was - /// present in this snapshot. Telemetry uses this only for an event that + /// The concrete upstream model produced by the wildcard-resolution + /// branch, paired with the configured wildcard row that produced it. + /// Empty for exact model resolution, including a request that literally + /// names a wildcard row. Telemetry uses this only for an event that /// actually dispatched that same row; it is not an access-log identity. - pub pricing_model_id: String, + pub wildcard_pricing_model_id: String, /// Canonical non-nil UUID issued by CP that makes the concrete model below /// billable. This stays beside the captured model id so a terminal emitter /// can either send the complete authority tuple or omit it entirely. - pub pricing_authority_id: String, - pub pricing_model: String, + pub wildcard_pricing_authority_id: String, + pub wildcard_pricing_model: String, /// Which cache layer answered this request, once one has — `Some` /// exactly when the response came out of the cache. /// @@ -699,12 +700,15 @@ pub(crate) fn note_target(model: &Model, provider_key_id: &str) { }); } -/// Record the complete pricing authority a direct or embedding dispatch -/// resolved to. This is deliberately separate from [`note_target`]: it is -/// accounting provenance rather than an access-log identity. The authority is -/// optional for rolling upgrades; without it, retain nothing so the terminal -/// event cannot assert a mutable current-model price. -pub(crate) fn note_pricing_identity( +/// Record the complete pricing authority a caller-addressed wildcard row +/// resolved to. +/// +/// This is deliberately separate from [`note_target`]: an exact request for +/// the literal wildcard row also has an upstream model name, but it never +/// passed wildcard capture and must not be used as a pricing identity. The +/// authority is optional for rolling upgrades; without it, retain nothing so +/// the terminal event cannot assert a concrete wildcard price. +pub(crate) fn note_wildcard_pricing_identity( model_id: &str, pricing_authority_id: Option<&str>, concrete_model: &str, @@ -713,30 +717,29 @@ pub(crate) fn note_pricing_identity( return; }; if model_id.is_empty() - || !valid_pricing_model(concrete_model) + || !valid_wildcard_pricing_model(concrete_model) || !valid_pricing_authority_id(pricing_authority_id) { return; } with(|r| { - r.pricing_model_id = model_id.to_string(); - r.pricing_authority_id = pricing_authority_id.to_string(); - r.pricing_model = concrete_model.to_string(); + r.wildcard_pricing_model_id = model_id.to_string(); + r.wildcard_pricing_authority_id = pricing_authority_id.to_string(); + r.wildcard_pricing_model = concrete_model.to_string(); }); } -// Keep this in step with AISIX Cloud's model-pricing name bound. A configured -// or wildcard-resolved upstream name can be valid for dispatch while still -// being too long to become a safe CP pricing lookup key; omit the entire -// optional tuple so its terminal parent event remains observable and -// explicitly unpriced. -const MAX_PRICING_MODEL_CHARS: usize = 120; +// Keep this in step with AISIX Cloud's model-pricing name bound. A wildcard +// capture can be a valid upstream model name while still being too long to +// become a safe CP pricing lookup key; omit the entire optional tuple so its +// terminal parent event remains observable and explicitly unpriced. +const MAX_WILDCARD_PRICING_MODEL_CHARS: usize = 120; -fn valid_pricing_model(model: &str) -> bool { +fn valid_wildcard_pricing_model(model: &str) -> bool { !model.is_empty() && !model.contains('*') && !model.contains('\0') - && model.chars().count() <= MAX_PRICING_MODEL_CHARS + && model.chars().count() <= MAX_WILDCARD_PRICING_MODEL_CHARS } fn valid_pricing_authority_id(id: &str) -> bool { @@ -766,9 +769,9 @@ pub(crate) fn note_cache_hit_entry(entry: &Model, hit_layer: &'static str) { note_target(entry, entry.provider_key_id.as_deref().unwrap_or_default()); with(|r| { r.cache_hit_layer = Some(hit_layer); - r.pricing_model_id.clear(); - r.pricing_authority_id.clear(); - r.pricing_model.clear(); + r.wildcard_pricing_model_id.clear(); + r.wildcard_pricing_authority_id.clear(); + r.wildcard_pricing_model.clear(); }); } diff --git a/crates/aisix-proxy/src/model_resolve.rs b/crates/aisix-proxy/src/model_resolve.rs index 69a8b95c..75033dcd 100644 --- a/crates/aisix-proxy/src/model_resolve.rs +++ b/crates/aisix-proxy/src/model_resolve.rs @@ -46,13 +46,8 @@ pub(crate) fn resolve_model( // deliberately decided against the dispatch snapshot: a later terminal // emitter may see a refreshed configuration where the row changed or was // deleted, but it must price the concrete model this request dispatched. - if entry - .value - .model_name - .as_deref() - .is_some_and(|template| template.contains('*')) - { - crate::attribution::note_pricing_identity( + if wildcard_pricing_eligible(&entry.value, &upstream) { + crate::attribution::note_wildcard_pricing_identity( &entry.id, entry.value.pricing_authority_id.as_deref(), &upstream, @@ -79,13 +74,6 @@ fn note_dispatchable_entry(entry: &ResourceEntry) { return; } crate::attribution::note_resolved_entry(&entry.id); - if let Some(upstream) = model.model_name.as_deref() { - crate::attribution::note_pricing_identity( - &entry.id, - model.pricing_authority_id.as_deref(), - upstream, - ); - } } /// Wildcard fallback: the most specific direct Model whose `*`-glob @@ -185,6 +173,22 @@ fn resolve_upstream_model_name(model: &Model, capture: &str) -> String { } } +/// Whether a wildcard dispatch produced a concrete provider-model identity +/// that is safe to send to CP for pricing. +/// +/// A wildcard display alias with a fixed upstream is already priced by its +/// configured model name. Only an upstream template with exactly one `*` +/// yields a caller-specific provider-model identity, and a literal `*` is +/// never a concrete catalog key. +fn wildcard_pricing_eligible(model: &Model, upstream: &str) -> bool { + model + .model_name + .as_deref() + .is_some_and(|template| template.bytes().filter(|&byte| byte == b'*').count() == 1) + && !upstream.is_empty() + && !upstream.contains('*') +} + #[cfg(test)] mod tests { use super::*; @@ -272,12 +276,13 @@ mod tests { } /// Pricing eligibility belongs to the snapshot that dispatched the - /// request. Exact models and wildcard aliases over a fixed upstream carry - /// their exact authority; a literal wildcard row, a capture that is still + /// request. In particular, a wildcard display alias over a fixed model, + /// a fixed direct or embedding model with a stale authority, a stale + /// multiple-star template, a literal wildcard row, a capture that is still /// itself a wildcard, or an absent, nil, or noncanonical authority must /// not manufacture a concrete provider-model price identity. #[tokio::test] - async fn dispatchable_models_with_authority_set_pricing_identity() { + async fn only_concrete_template_capture_with_authority_sets_wildcard_pricing_identity() { use std::sync::Arc; let mut embedding = priced_direct_model("embedding", Some("text-embedding-3-small")); @@ -285,11 +290,17 @@ mod tests { dimensions: 4, normalize: true, }); + let multiple_stars = priced_direct_model("multiple/*", Some("gpt-*-*")); + assert!( + !wildcard_pricing_eligible(&multiple_stars, "gpt-4o"), + "an authority left on an unsupported multiple-star template must not mint pricing" + ); let snap = snapshot_with(vec![ ("wildcard", priced_direct_model("openrouter/*", Some("*"))), ("fixed", priced_direct_model("fixed/*", Some("gpt-4o"))), ("exact", priced_direct_model("exact", Some("gpt-4o-mini"))), ("embedding", embedding), + ("multiple", multiple_stars), ("unpriced", direct_model("unpriced/*", Some("*"))), ]); let mut nil_authority = direct_model("nil/*", Some("*")); @@ -311,17 +322,21 @@ mod tests { resolve_model(&snap, "openrouter/gpt-4o-2024-08-06") .expect("concrete wildcard request resolves"); let resolved = crate::attribution::current().expect("in request scope"); - assert_eq!(resolved.pricing_model_id, "wildcard"); + assert_eq!(resolved.wildcard_pricing_model_id, "wildcard"); assert_eq!( - resolved.pricing_authority_id, + resolved.wildcard_pricing_authority_id, "a3ebdc63-e921-4323-a75c-3b911f950046" ); - assert_eq!(resolved.pricing_model, "gpt-4o-2024-08-06"); + assert_eq!(resolved.wildcard_pricing_model, "gpt-4o-2024-08-06"); }, ) .await; for requested in [ + "fixed/anything", + "exact", + "embedding", + "multiple/gpt-4o", "openrouter/*", "openrouter/gpt-*", "unpriced/gpt-4o", @@ -333,30 +348,12 @@ mod tests { async { resolve_model(&snap, requested).expect("configured request resolves"); let resolved = crate::attribution::current().expect("in request scope"); - assert!(resolved.pricing_model_id.is_empty(), "{requested}"); - assert!(resolved.pricing_authority_id.is_empty(), "{requested}"); - assert!(resolved.pricing_model.is_empty(), "{requested}"); - }, - ) - .await; - } - - for (requested, model_id, model_name) in [ - ("fixed/anything", "fixed", "gpt-4o"), - ("exact", "exact", "gpt-4o-mini"), - ("embedding", "embedding", "text-embedding-3-small"), - ] { - crate::attribution::scope( - Arc::new(crate::attribution::RequestAttribution::default()), - async { - resolve_model(&snap, requested).expect("configured request resolves"); - let resolved = crate::attribution::current().expect("in request scope"); - assert_eq!(resolved.pricing_model_id, model_id, "{requested}"); - assert_eq!( - resolved.pricing_authority_id, "a3ebdc63-e921-4323-a75c-3b911f950046", + assert!(resolved.wildcard_pricing_model_id.is_empty(), "{requested}"); + assert!( + resolved.wildcard_pricing_authority_id.is_empty(), "{requested}" ); - assert_eq!(resolved.pricing_model, model_name, "{requested}"); + assert!(resolved.wildcard_pricing_model.is_empty(), "{requested}"); }, ) .await; @@ -368,9 +365,9 @@ mod tests { async { resolve_model(&snap, &overlong).expect("overlong wildcard request still resolves"); let resolved = crate::attribution::current().expect("in request scope"); - assert!(resolved.pricing_model_id.is_empty()); - assert!(resolved.pricing_authority_id.is_empty()); - assert!(resolved.pricing_model.is_empty()); + assert!(resolved.wildcard_pricing_model_id.is_empty()); + assert!(resolved.wildcard_pricing_authority_id.is_empty()); + assert!(resolved.wildcard_pricing_model.is_empty()); }, ) .await; diff --git a/crates/aisix-proxy/src/request_metrics.rs b/crates/aisix-proxy/src/request_metrics.rs index 916ae992..09925344 100644 --- a/crates/aisix-proxy/src/request_metrics.rs +++ b/crates/aisix-proxy/src/request_metrics.rs @@ -704,9 +704,9 @@ mod tests { provider: "OpenAI".to_string(), upstream_model: "gpt-4o-mini".to_string(), provider_key_id: "pk-1".to_string(), - pricing_model_id: String::new(), - pricing_authority_id: String::new(), - pricing_model: String::new(), + wildcard_pricing_model_id: String::new(), + wildcard_pricing_authority_id: String::new(), + wildcard_pricing_model: String::new(), cache_hit_layer: None, } } @@ -793,22 +793,22 @@ mod tests { .expect("wildcard model resolves at dispatch"); let captured = crate::attribution::current().expect("request attribution is installed"); - assert_eq!(captured.pricing_model_id, "wildcard"); + assert_eq!(captured.wildcard_pricing_model_id, "wildcard"); assert_eq!( - captured.pricing_authority_id, + captured.wildcard_pricing_authority_id, "a3ebdc63-e921-4323-a75c-3b911f950046" ); - assert_eq!(captured.pricing_model, "gpt-4o"); + assert_eq!(captured.wildcard_pricing_model, "gpt-4o"); assert!(!is_ensemble(&refreshed, "openrouter/gpt-4o")); let after_metrics = crate::attribution::current().expect("request attribution is installed"); - assert_eq!(after_metrics.pricing_model_id, "wildcard"); + assert_eq!(after_metrics.wildcard_pricing_model_id, "wildcard"); assert_eq!( - after_metrics.pricing_authority_id, + after_metrics.wildcard_pricing_authority_id, "a3ebdc63-e921-4323-a75c-3b911f950046" ); - assert_eq!(after_metrics.pricing_model, "gpt-4o"); + assert_eq!(after_metrics.wildcard_pricing_model, "gpt-4o"); }, ) .await; diff --git a/crates/aisix-proxy/src/usage_attr.rs b/crates/aisix-proxy/src/usage_attr.rs index af1f914c..c3216f06 100644 --- a/crates/aisix-proxy/src/usage_attr.rs +++ b/crates/aisix-proxy/src/usage_attr.rs @@ -439,14 +439,14 @@ pub(crate) fn metric_model_label_pair<'a>( } } -/// Fill the optional DP-to-CP dispatch-pricing authority at the one usage +/// Fill the optional DP-to-CP wildcard-pricing authority at the one usage /// emission chokepoint. Model resolution establishes eligibility from the /// dispatch snapshot and records a complete authority tuple, so this path must /// not consult an emission-time snapshot that may have changed while a stream /// or realtime session was still running. A cache hit or pre-dispatch failure /// has no upstream call to price. Detached gateway work has no caller /// attribution and therefore cannot price itself as the parent request. -fn apply_pricing_identity(event: &mut UsageEvent, surface: Surface, dispatched: bool) { +fn apply_wildcard_pricing_model(event: &mut UsageEvent, surface: Surface, dispatched: bool) { // Batch-management rows are observable upstream management requests, not // the batch's model inference. They stay unpriced even when their route // resolves through a wildcard alias; batch completion accounting is a @@ -457,12 +457,14 @@ fn apply_pricing_identity(event: &mut UsageEvent, surface: Surface, dispatched: let Some(resolved) = crate::attribution::current() else { return; }; - if resolved.cache_hit_layer.is_some() || resolved.pricing_model_id != event.model_id { + if resolved.cache_hit_layer.is_some() || resolved.wildcard_pricing_model_id != event.model_id { return; } - if !resolved.pricing_authority_id.is_empty() && !resolved.pricing_model.is_empty() { - event.pricing_authority_id = resolved.pricing_authority_id; - event.resolved_pricing_model = resolved.pricing_model; + if !resolved.wildcard_pricing_authority_id.is_empty() + && !resolved.wildcard_pricing_model.is_empty() + { + event.pricing_authority_id = resolved.wildcard_pricing_authority_id; + event.resolved_pricing_model = resolved.wildcard_pricing_model; } } @@ -985,7 +987,7 @@ pub(crate) fn emit_usage( if terminal && event.guardrail_blocked { state.metrics.record_guardrail_blocked_request(); } - apply_pricing_identity(&mut event, surface, dispatched); + apply_wildcard_pricing_model(&mut event, surface, dispatched); let emission = trace.map(|bundle| { event.trace_id = bundle.trace_id_hex(); bundle.emission( @@ -1069,7 +1071,7 @@ mod tests { } #[tokio::test] - async fn request_attribution_stamps_concrete_dispatch_pricing_authorities() { + async fn request_attribution_stamps_only_concrete_wildcard_pricing_authorities() { use aisix_core::resource::ResourceEntry; use aisix_core::snapshot::ResourceTable; @@ -1097,12 +1099,12 @@ mod tests { crate::attribution::note_target(&served.value, "pk-1"); let attribution = crate::attribution::current().expect("in request attribution scope"); - assert_eq!(attribution.pricing_model_id, "wildcard"); + assert_eq!(attribution.wildcard_pricing_model_id, "wildcard"); assert_eq!( - attribution.pricing_authority_id, + attribution.wildcard_pricing_authority_id, "a3ebdc63-e921-4323-a75c-3b911f950046" ); - assert_eq!(attribution.pricing_model, "gpt-4o-2024-08-06"); + assert_eq!(attribution.wildcard_pricing_model, "gpt-4o-2024-08-06"); let mut event = UsageEvent { // NO-GUARDRAIL-CHAIN: this focused unit test constructs // a synthetic pricing event, not a gateway request. @@ -1111,7 +1113,7 @@ mod tests { applied_guardrails: Vec::new(), ..Default::default() }; - apply_pricing_identity(&mut event, crate::operation::CHAT, true); + apply_wildcard_pricing_model(&mut event, crate::operation::CHAT, true); assert_eq!( event.pricing_authority_id, "a3ebdc63-e921-4323-a75c-3b911f950046" @@ -1126,7 +1128,7 @@ mod tests { applied_guardrails: Vec::new(), ..Default::default() }; - apply_pricing_identity(&mut batch_event, crate::operation::BATCHES, true); + apply_wildcard_pricing_model(&mut batch_event, crate::operation::BATCHES, true); assert!(batch_event.pricing_authority_id.is_empty()); assert!(batch_event.resolved_pricing_model.is_empty()); @@ -1138,7 +1140,7 @@ mod tests { // `note_cache_hit_entry` clears the identity above. Restore // one here to pin the separate emission gate too: a cache // hit is never billable as an upstream wildcard dispatch. - crate::attribution::note_pricing_identity( + crate::attribution::note_wildcard_pricing_identity( "wildcard", Some("a3ebdc63-e921-4323-a75c-3b911f950046"), "gpt-4o-2024-08-06", @@ -1151,7 +1153,7 @@ mod tests { applied_guardrails: Vec::new(), ..Default::default() }; - apply_pricing_identity(&mut cached_event, crate::operation::CHAT, true); + apply_wildcard_pricing_model(&mut cached_event, crate::operation::CHAT, true); assert!(cached_event.pricing_authority_id.is_empty()); assert!(cached_event.resolved_pricing_model.is_empty()); }, @@ -1166,9 +1168,9 @@ mod tests { crate::attribution::note_target(&literal.value, "pk-1"); let attribution = crate::attribution::current().expect("in request attribution scope"); - assert!(attribution.pricing_model_id.is_empty()); - assert!(attribution.pricing_authority_id.is_empty()); - assert!(attribution.pricing_model.is_empty()); + assert!(attribution.wildcard_pricing_model_id.is_empty()); + assert!(attribution.wildcard_pricing_authority_id.is_empty()); + assert!(attribution.wildcard_pricing_model.is_empty()); let mut event = UsageEvent { // NO-GUARDRAIL-CHAIN: this focused unit test constructs @@ -1180,7 +1182,7 @@ mod tests { applied_guardrails: Vec::new(), ..Default::default() }; - apply_pricing_identity(&mut event, crate::operation::CHAT, true); + apply_wildcard_pricing_model(&mut event, crate::operation::CHAT, true); assert!(event.pricing_authority_id.is_empty()); assert!(event.resolved_pricing_model.is_empty()); }, @@ -1205,9 +1207,9 @@ mod tests { .expect("legacy wildcard model resolves"); let attribution = crate::attribution::current().expect("in request attribution scope"); - assert!(attribution.pricing_model_id.is_empty()); - assert!(attribution.pricing_authority_id.is_empty()); - assert!(attribution.pricing_model.is_empty()); + assert!(attribution.wildcard_pricing_model_id.is_empty()); + assert!(attribution.wildcard_pricing_authority_id.is_empty()); + assert!(attribution.wildcard_pricing_model.is_empty()); let mut event = UsageEvent { // NO-GUARDRAIL-CHAIN: this focused test constructs a @@ -1217,7 +1219,7 @@ mod tests { applied_guardrails: Vec::new(), ..Default::default() }; - apply_pricing_identity(&mut event, crate::operation::CHAT, true); + apply_wildcard_pricing_model(&mut event, crate::operation::CHAT, true); let wire = serde_json::to_value(event).expect("usage event serialises"); assert!(wire.get("pricing_authority_id").is_none()); assert!(wire.get("resolved_pricing_model").is_none()); @@ -1225,10 +1227,8 @@ mod tests { ) .await; - // Exact direct and embedding rows take the same immutable authority - // path as a wildcard's captured upstream value. This pins both - // modalities at the terminal-event chokepoint without changing their - // parent model id, target, cost, or budget attribution. + // A pointer left on a fixed direct or embedding row after a wildcard + // alias is changed must not broaden the wildcard-only contract. let fixed_table = ResourceTable::default(); let direct: aisix_core::Model = serde_json::from_value(serde_json::json!({ "display_name": "fixed-chat", @@ -1253,13 +1253,9 @@ mod tests { models: fixed_table, ..Default::default() }; - for (model_id, upstream, surface) in [ - ("fixed-chat", "gpt-4o-mini", crate::operation::CHAT), - ( - "fixed-embedding", - "text-embedding-3-small", - crate::operation::EMBEDDINGS, - ), + for (model_id, surface) in [ + ("fixed-chat", crate::operation::CHAT), + ("fixed-embedding", crate::operation::EMBEDDINGS), ] { crate::attribution::scope( Arc::new(crate::attribution::RequestAttribution::default()), @@ -1275,13 +1271,10 @@ mod tests { applied_guardrails: Vec::new(), ..Default::default() }; - apply_pricing_identity(&mut event, surface, true); + apply_wildcard_pricing_model(&mut event, surface, true); assert_eq!(event.model_id, model_id); - assert_eq!( - event.pricing_authority_id, - "a3ebdc63-e921-4323-a75c-3b911f950046" - ); - assert_eq!(event.resolved_pricing_model, upstream); + assert!(event.pricing_authority_id.is_empty()); + assert!(event.resolved_pricing_model.is_empty()); }, ) .await; diff --git a/schemas/resources-lenient/model.schema.json b/schemas/resources-lenient/model.schema.json index 333cabf3..a956a9f1 100644 --- a/schemas/resources-lenient/model.schema.json +++ b/schemas/resources-lenient/model.schema.json @@ -1366,7 +1366,7 @@ "type": "string" }, "pricing_authority_id": { - "description": "Opaque control-plane-issued canonical, non-nil UUID that authorizes the configured exact or wildcard-resolved upstream model name for pricing. It is relevant only to a direct-shaped model (chat or embedding); without it the data plane deliberately omits the resolved model from terminal telemetry so the control plane can leave the call unpriced rather than trust mutable configuration.", + "description": "Opaque control-plane-issued canonical, non-nil UUID that authorizes a concrete wildcard-upstream model name for pricing. It is relevant only to a direct-shaped wildcard model (chat or embedding); without it the data plane deliberately omits the resolved model from terminal telemetry so the control plane can leave the call unpriced rather than trust mutable configuration.", "maxLength": 64, "minLength": 36, "pattern": "^[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}$", diff --git a/schemas/resources/model.schema.json b/schemas/resources/model.schema.json index 6622a024..c4fdf649 100644 --- a/schemas/resources/model.schema.json +++ b/schemas/resources/model.schema.json @@ -1487,7 +1487,7 @@ "type": "string" }, "pricing_authority_id": { - "description": "Opaque control-plane-issued canonical, non-nil UUID that authorizes the configured exact or wildcard-resolved upstream model name for pricing. It is relevant only to a direct-shaped model (chat or embedding); without it the data plane deliberately omits the resolved model from terminal telemetry so the control plane can leave the call unpriced rather than trust mutable configuration.", + "description": "Opaque control-plane-issued canonical, non-nil UUID that authorizes a concrete wildcard-upstream model name for pricing. It is relevant only to a direct-shaped wildcard model (chat or embedding); without it the data plane deliberately omits the resolved model from terminal telemetry so the control plane can leave the call unpriced rather than trust mutable configuration.", "maxLength": 64, "minLength": 36, "not": { From db9f80c4df0052abd39026e12c5bf473578cb4de Mon Sep 17 00:00:00 2001 From: Ming Wen Date: Thu, 1 Oct 2026 09:52:20 +0800 Subject: [PATCH 09/16] test: import embedding config in pricing coverage --- crates/aisix-proxy/src/model_resolve.rs | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/crates/aisix-proxy/src/model_resolve.rs b/crates/aisix-proxy/src/model_resolve.rs index 75033dcd..ae11ec14 100644 --- a/crates/aisix-proxy/src/model_resolve.rs +++ b/crates/aisix-proxy/src/model_resolve.rs @@ -192,6 +192,7 @@ fn wildcard_pricing_eligible(model: &Model, upstream: &str) -> bool { #[cfg(test)] mod tests { use super::*; + use aisix_core::models::EmbeddingConfig; use aisix_core::snapshot::ResourceTable; fn direct_model(display_name: &str, model_name: Option<&str>) -> Model { @@ -286,7 +287,7 @@ mod tests { use std::sync::Arc; let mut embedding = priced_direct_model("embedding", Some("text-embedding-3-small")); - embedding.embedding = Some(aisix_core::EmbeddingConfig { + embedding.embedding = Some(EmbeddingConfig { dimensions: 4, normalize: true, }); From 2bd26f189c574d38cffd4455241f02eaa1273ebe Mon Sep 17 00:00:00 2001 From: Ming Wen Date: Thu, 1 Oct 2026 10:08:37 +0800 Subject: [PATCH 10/16] fix: preserve wildcard pricing across realtime upgrades --- crates/aisix-proxy/src/attribution.rs | 34 +++++++++++++------ crates/aisix-proxy/src/realtime.rs | 8 ++--- .../wildcard-pricing-telemetry-e2e.test.ts | 4 +++ 3 files changed, 32 insertions(+), 14 deletions(-) diff --git a/crates/aisix-proxy/src/attribution.rs b/crates/aisix-proxy/src/attribution.rs index 254b253d..5b9e060e 100644 --- a/crates/aisix-proxy/src/attribution.rs +++ b/crates/aisix-proxy/src/attribution.rs @@ -615,6 +615,21 @@ pub(crate) struct RequestAttribution { } impl RequestAttribution { + /// A continuation for a detached realtime session. It deliberately copies + /// only the wildcard pricing tuple: the HTTP request's body counters and + /// pending access log are finalized when its upgrade response completes, + /// while the session owns a separate terminal event and access-log line. + fn wildcard_pricing_continuation(&self) -> Self { + let resolved = self.get(); + let continuation = Self::default(); + let mut cell = continuation.lock(); + cell.resolved.wildcard_pricing_model_id = resolved.wildcard_pricing_model_id; + cell.resolved.wildcard_pricing_authority_id = resolved.wildcard_pricing_authority_id; + cell.resolved.wildcard_pricing_model = resolved.wildcard_pricing_model; + drop(cell); + continuation + } + /// A sub-call scope that cannot change the parent request's target /// attribution but can still account for gateway-initiated embeddings on /// the parent's eventual terminal usage event. @@ -931,17 +946,16 @@ pub(crate) fn current() -> Option { CURRENT.try_with(|a| a.get()).ok() } -/// The current request's attribution cell, for work that continues on a -/// detached task after the HTTP handler returns. +/// A fresh attribution cell for a detached realtime session. /// -/// A WebSocket upgrade moves its session onto axum's upgrade task. That task -/// does not inherit Tokio task-locals, but it is still the same client request: -/// the terminal session usage row must retain the model identity captured -/// before the upgrade. Callers install this exact cell with [`scope`] around -/// their detached continuation; `None` remains correct outside request -/// middleware (for example, focused unit tests). -pub(crate) fn current_cell() -> Option> { - CURRENT.try_with(Arc::clone).ok() +/// A WebSocket upgrade moves its session onto axum's upgrade task, which does +/// not inherit Tokio task-locals. The session needs a wildcard pricing tuple +/// captured before the upgrade, but must not reuse the HTTP cell: the latter +/// owns the completed upgrade response's body counters and access-log line. +pub(crate) fn current_wildcard_pricing_continuation() -> Option> { + CURRENT + .try_with(|parent| Arc::new(parent.wildcard_pricing_continuation())) + .ok() } /// Record one actual gateway-initiated embedding bridge call. This is safe diff --git a/crates/aisix-proxy/src/realtime.rs b/crates/aisix-proxy/src/realtime.rs index ccd59c77..1fc9eb83 100644 --- a/crates/aisix-proxy/src/realtime.rs +++ b/crates/aisix-proxy/src/realtime.rs @@ -271,10 +271,10 @@ pub(crate) async fn realtime( let state2 = state.clone(); let client2 = client.clone(); // `on_upgrade` runs on a new Tokio task, which does not inherit - // the request task-local. Keep the same cell so the terminal - // realtime UsageEvent retains the wildcard model identity that - // `prepare` resolved before accepting this upgrade. - let attribution = crate::attribution::current_cell(); + // the request task-local. Copy only the wildcard pricing tuple: + // the HTTP cell is already owned by the completed upgrade + // response and cannot also own the session's terminal log. + let attribution = crate::attribution::current_wildcard_pricing_continuation(); // `on_upgrade` runs the session on a detached task, so the // request span has to be attached to the future rather than // inherited — without it the session's guardrail checks log diff --git a/tests/e2e/src/cases/wildcard-pricing-telemetry-e2e.test.ts b/tests/e2e/src/cases/wildcard-pricing-telemetry-e2e.test.ts index 7624d0be..eab567c4 100644 --- a/tests/e2e/src/cases/wildcard-pricing-telemetry-e2e.test.ts +++ b/tests/e2e/src/cases/wildcard-pricing-telemetry-e2e.test.ts @@ -25,6 +25,7 @@ const LOGSTORE = "wildcard-pricing-telemetry"; const WILDCARD_ALIAS = "openrouter/*"; const KNOWN_MODEL = "openai/gpt-4o-mini"; const UNKNOWN_MODEL = "unknown/provider-model"; +const PRICING_AUTHORITY_ID = "a3ebdc63-e921-4323-a75c-3b911f950046"; function upstreamResponse() { return { @@ -82,6 +83,7 @@ describe("wildcard pricing telemetry e2e", () => { provider: "openrouter", model_name: "*", provider_key_id: providerKey.id, + pricing_authority_id: PRICING_AUTHORITY_ID, }); wildcardID = wildcard.id; @@ -135,6 +137,7 @@ describe("wildcard pricing telemetry e2e", () => { `usage event for ${knownRequest}`, ); expect(known.get("model_id")).toBe(wildcardID); + expect(known.get("pricing_authority_id")).toBe(PRICING_AUTHORITY_ID); expect(known.get("resolved_pricing_model")).toBe(KNOWN_MODEL); expect(known.get("prompt_tokens")).toBe("10"); expect(known.get("completion_tokens")).toBe("5"); @@ -151,6 +154,7 @@ describe("wildcard pricing telemetry e2e", () => { `usage event for ${unknownRequest}`, ); expect(unknown.get("model_id")).toBe(wildcardID); + expect(unknown.get("pricing_authority_id")).toBe(PRICING_AUTHORITY_ID); expect(unknown.get("resolved_pricing_model")).toBe(UNKNOWN_MODEL); }); }); From be4848a6d77358d1ad259575b2a7f73aef0187d9 Mon Sep 17 00:00:00 2001 From: Ming Wen Date: Thu, 1 Oct 2026 10:36:08 +0800 Subject: [PATCH 11/16] test: cover wildcard embedding pricing attribution --- crates/aisix-proxy/src/attribution.rs | 92 +++++++++++++++++++ .../wildcard-pricing-telemetry-e2e.test.ts | 81 ++++++++++++++++ 2 files changed, 173 insertions(+) diff --git a/crates/aisix-proxy/src/attribution.rs b/crates/aisix-proxy/src/attribution.rs index 5b9e060e..33a502fa 100644 --- a/crates/aisix-proxy/src/attribution.rs +++ b/crates/aisix-proxy/src/attribution.rs @@ -1384,6 +1384,98 @@ mod tests { assert!(current().is_none()); } + #[test] + fn wildcard_pricing_continuation_keeps_only_the_pricing_tuple() { + let parent = RequestAttribution::default(); + parent.track(tracing::info_span!("parent_request")); + parent.note_head_written(); + parent.add_request_bytes(41); + parent.note_request_complete(); + parent.add_response_bytes(17); + parent.park_access_log( + &aisix_obs::AccessLog { + method: "POST", + path: "/v1/realtime", + status: 101, + latency: Duration::from_secs(0), + duration: Duration::from_secs(0), + provider: Some("openai"), + model: Some("realtime/*"), + upstream_model: Some("gpt-realtime"), + provider_key_id: Some("pk-1"), + api_key_id: Some("api-key"), + prompt_tokens: Some(1), + completion_tokens: Some(2), + total_tokens: Some(3), + request_id: "req-parent", + provider_request_id: Some("resp-parent"), + served_by_model: None, + routing_attempt_count: None, + routing_fallback_count: None, + error_kind: None, + error: None, + mcp: None, + cache: None, + request_body_bytes: None, + response_body_bytes: None, + }, + None, + ); + { + let mut cell = parent.lock(); + cell.resolved = Resolved { + requested_model: "realtime/customer-model".into(), + provider: "openai".into(), + upstream_model: "gpt-realtime".into(), + provider_key_id: "pk-1".into(), + wildcard_pricing_model_id: "wildcard-row".into(), + wildcard_pricing_authority_id: "a3ebdc63-e921-4323-a75c-3b911f950046".into(), + wildcard_pricing_model: "gpt-realtime".into(), + cache_hit_layer: Some("exact"), + }; + cell.cancel.api_key_id = "api-key".into(); + cell.cancel.emitted_terminal = true; + cell.pending_log = Some(PendingAccessLog::new( + "POST", + "/v1/realtime", + "req-parent", + "api-key", + Instant::now(), + )); + cell.stream_owns_log = true; + cell.finished = true; + assert!( + cell.ready_line.is_some(), + "premise: parent owns a pending access log" + ); + } + + let continuation = parent.wildcard_pricing_continuation(); + let resolved = continuation.get(); + assert_eq!(resolved.wildcard_pricing_model_id, "wildcard-row"); + assert_eq!( + resolved.wildcard_pricing_authority_id, + "a3ebdc63-e921-4323-a75c-3b911f950046" + ); + assert_eq!(resolved.wildcard_pricing_model, "gpt-realtime"); + assert!(resolved.requested_model.is_empty()); + assert!(resolved.provider.is_empty()); + assert!(resolved.upstream_model.is_empty()); + assert!(resolved.provider_key_id.is_empty()); + assert!(resolved.cache_hit_layer.is_none()); + assert_eq!(continuation.body_sizes(true), (None, Some(0))); + + let cell = continuation.lock(); + assert!(cell.cancel.api_key_id.is_empty()); + assert!(!cell.cancel.emitted_terminal); + assert!(cell.pending_log.is_none()); + assert!(!cell.stream_owns_log); + assert!(cell.request_span.is_none()); + assert!(!cell.head_written); + assert!(!cell.finished); + assert!(cell.ready_line.is_none()); + } + /// A gateway-owned embedding must retain the caller's actual target while /// still reaching that caller's one terminal usage event. This is the /// reason the detached cell shares only the child-call ledger, not its diff --git a/tests/e2e/src/cases/wildcard-pricing-telemetry-e2e.test.ts b/tests/e2e/src/cases/wildcard-pricing-telemetry-e2e.test.ts index eab567c4..247cde20 100644 --- a/tests/e2e/src/cases/wildcard-pricing-telemetry-e2e.test.ts +++ b/tests/e2e/src/cases/wildcard-pricing-telemetry-e2e.test.ts @@ -26,6 +26,11 @@ const WILDCARD_ALIAS = "openrouter/*"; const KNOWN_MODEL = "openai/gpt-4o-mini"; const UNKNOWN_MODEL = "unknown/provider-model"; const PRICING_AUTHORITY_ID = "a3ebdc63-e921-4323-a75c-3b911f950046"; +const EMBEDDING_WILDCARD_ALIAS = "embedding/*"; +const EMBEDDING_REQUEST_MODEL = "embedding/embedding-3-small"; +const EMBEDDING_UPSTREAM_MODEL = "text-embedding-3-small"; +const EMBEDDING_INPUT = "price this embedding"; +const EMBEDDING_VECTOR = [0.1, 0.2, 0.3]; function upstreamResponse() { return { @@ -40,11 +45,24 @@ function upstreamResponse() { }; } +function embeddingUpstreamResponse() { + return { + object: "list", + // Pricing must come from wildcard dispatch attribution rather than the + // provider response's optional model field. + model: "provider-response-embedding-model", + data: [{ object: "embedding", index: 0, embedding: EMBEDDING_VECTOR }], + usage: { prompt_tokens: 7, total_tokens: 7 }, + }; +} + describe("wildcard pricing telemetry e2e", () => { let app: SpawnedApp | undefined; let sls: MockSls | undefined; let upstream: OpenAiUpstream | undefined; + let embeddingUpstream: OpenAiUpstream | undefined; let wildcardID = ""; + let embeddingWildcardID = ""; let etcdReachable = false; beforeAll(async () => { @@ -54,6 +72,7 @@ describe("wildcard pricing telemetry e2e", () => { sls = await startMockSls(); upstream = await startOpenAiUpstream({ nonStreamBody: upstreamResponse() }); + embeddingUpstream = await startOpenAiUpstream({ nonStreamBody: embeddingUpstreamResponse() }); app = await spawnApp({ extraEnv: { [`SLS_CRED_${CREDENTIAL_REF.toUpperCase()}_AK_ID`]: "mock-akid", @@ -86,6 +105,22 @@ describe("wildcard pricing telemetry e2e", () => { pricing_authority_id: PRICING_AUTHORITY_ID, }); wildcardID = wildcard.id; + const embeddingProviderKey = await seed.createProviderKey({ + display_name: "wildcard-pricing-embedding-pk", + provider: "openai", + adapter: "openai", + secret: "sk-mock", + api_base: `${embeddingUpstream.baseUrl}/v1`, + }); + const embeddingWildcard = await seed.createModel({ + display_name: EMBEDDING_WILDCARD_ALIAS, + provider: "openai", + model_name: "text-*", + provider_key_id: embeddingProviderKey.id, + pricing_authority_id: PRICING_AUTHORITY_ID, + embedding: { dimensions: EMBEDDING_VECTOR.length }, + }); + embeddingWildcardID = embeddingWildcard.id; // Seeded last: a successful models-list gate proves that all preceding // resources, including the exporter, are in the same gateway snapshot. @@ -102,6 +137,7 @@ describe("wildcard pricing telemetry e2e", () => { afterAll(async () => { await app?.exit(); await upstream?.close(); + await embeddingUpstream?.close(); await sls?.close(); }); @@ -157,4 +193,49 @@ describe("wildcard pricing telemetry e2e", () => { expect(unknown.get("pricing_authority_id")).toBe(PRICING_AUTHORITY_ID); expect(unknown.get("resolved_pricing_model")).toBe(UNKNOWN_MODEL); }); + + test("direct embedding wildcard dispatch exports its concrete pricing identity", async (ctx) => { + if (!etcdReachable || !app || !sls || !embeddingUpstream || !embeddingWildcardID) { + ctx.skip(); + return; + } + + const baseline = embeddingUpstream.receivedRequests.length; + const res = await fetch(`${app.proxyUrl}/v1/embeddings`, { + method: "POST", + headers: { + authorization: `Bearer ${CALLER_PLAINTEXT}`, + "content-type": "application/json", + }, + body: JSON.stringify({ model: EMBEDDING_REQUEST_MODEL, input: EMBEDDING_INPUT }), + }); + const body = await res.text(); + expect(res.status, body).toBe(200); + expect(JSON.parse(body)).toMatchObject({ + object: "list", + model: EMBEDDING_REQUEST_MODEL, + data: [{ object: "embedding", index: 0, embedding: EMBEDDING_VECTOR }], + usage: { prompt_tokens: 7, total_tokens: 7 }, + }); + + const calls = embeddingUpstream.receivedRequests.slice(baseline); + expect(calls).toHaveLength(1); + expect(calls[0]!.method).toBe("POST"); + expect(calls[0]!.path).toBe("/v1/embeddings"); + expect(JSON.parse(calls[0]!.body)).toMatchObject({ + model: EMBEDDING_UPSTREAM_MODEL, + input: EMBEDDING_INPUT, + }); + + const event = await waitForSlsLog( + sls, + LOGSTORE, + (log) => log.get("requested_model") === EMBEDDING_REQUEST_MODEL, + `usage event for ${EMBEDDING_REQUEST_MODEL}`, + ); + expect(event.get("model_id")).toBe(embeddingWildcardID); + expect(event.get("pricing_authority_id")).toBe(PRICING_AUTHORITY_ID); + expect(event.get("resolved_pricing_model")).toBe(EMBEDDING_UPSTREAM_MODEL); + expect(event.get("prompt_tokens")).toBe("7"); + }); }); From 19ac5f7f3d4f6bb141e1af8a582cdf9a8b861bf0 Mon Sep 17 00:00:00 2001 From: Ming Wen Date: Thu, 1 Oct 2026 10:50:57 +0800 Subject: [PATCH 12/16] fix: keep management wildcard events unpriced --- crates/aisix-proxy/src/jobs.rs | 84 ++++++++++++++++++++++++++++ crates/aisix-proxy/src/usage_attr.rs | 48 ++++++++++------ 2 files changed, 116 insertions(+), 16 deletions(-) diff --git a/crates/aisix-proxy/src/jobs.rs b/crates/aisix-proxy/src/jobs.rs index 2894bcac..7510fc21 100644 --- a/crates/aisix-proxy/src/jobs.rs +++ b/crates/aisix-proxy/src/jobs.rs @@ -2421,6 +2421,90 @@ mod tests { assert_eq!(&bytes[..], b"{\"custom_id\":\"r1\"}\n"); } + #[tokio::test] + async fn wildcard_selected_file_and_fine_tuning_management_events_are_unpriced() { + let upstream = MockServer::start().await; + Mock::given(wm_method("GET")) + .and(path("/v1/files/file-abc")) + .respond_with(ResponseTemplate::new(200).set_body_json(serde_json::json!({ + "id": "file-abc", + "object": "file" + }))) + .expect(1) + .mount(&upstream) + .await; + Mock::given(wm_method("GET")) + .and(path("/v1/fine_tuning/jobs/ftjob-9")) + .respond_with(ResponseTemplate::new(200).set_body_json(serde_json::json!({ + "id": "ftjob-9", + "object": "fine_tuning.job" + }))) + .expect(1) + .mount(&upstream) + .await; + + let snap = AisixSnapshot::new(); + snap.provider_keys.insert(openai_pk(PK_A, &upstream.uri())); + let wildcard: Model = serde_json::from_value(serde_json::json!({ + "display_name": "jobs/*", + "provider": "openai", + "model_name": "*", + "provider_key_id": PK_A, + "pricing_authority_id": "a3ebdc63-e921-4323-a75c-3b911f950046", + })) + .unwrap(); + snap.models + .insert(ResourceEntry::new("wildcard", wildcard, 1)); + snap.apikeys.insert(apikey_entry(&["*"])); + let (app, mut rx) = build_app_with_sink(snap); + + for (management_kind, path, operation) in [ + ( + "file", + format!( + "/v1/files/{}", + encode_routed_id("file-abc", "jobs/gpt-4o-2024-08-06") + ), + "files", + ), + ( + "fine-tuning", + format!( + "/v1/fine_tuning/jobs/{}", + encode_routed_id("ftjob-9", "jobs/gpt-4o-2024-08-06") + ), + "fine_tuning", + ), + ] { + let req = Request::builder() + .method("GET") + .uri(path) + .header("authorization", "Bearer sk-caller") + .body(axum::body::Body::empty()) + .unwrap(); + let resp = app.clone().oneshot(req).await.unwrap(); + assert_eq!(resp.status(), StatusCode::OK, "{management_kind}"); + + let event = tokio::time::timeout(Duration::from_secs(2), rx.recv()) + .await + .expect("management usage event must arrive") + .expect("usage sink must remain open"); + assert_eq!(event.model_id, "wildcard", "{management_kind}"); + assert_eq!(event.requested_model, "jobs/*", "{management_kind}"); + assert_eq!(event.operation, operation, "{management_kind}"); + assert_eq!(event.prompt_tokens, 0, "{management_kind}"); + assert_eq!(event.completion_tokens, 0, "{management_kind}"); + assert!( + event.pricing_authority_id.is_empty(), + "{management_kind} management event must not select wildcard pricing" + ); + assert!( + event.resolved_pricing_model.is_empty(), + "{management_kind} management event must not select wildcard pricing" + ); + } + } + // ---- batches ---- #[tokio::test] diff --git a/crates/aisix-proxy/src/usage_attr.rs b/crates/aisix-proxy/src/usage_attr.rs index 5550762a..8683fe4a 100644 --- a/crates/aisix-proxy/src/usage_attr.rs +++ b/crates/aisix-proxy/src/usage_attr.rs @@ -447,11 +447,15 @@ pub(crate) fn metric_model_label_pair<'a>( /// has no upstream call to price. Detached gateway work has no caller /// attribution and therefore cannot price itself as the parent request. fn apply_wildcard_pricing_model(event: &mut UsageEvent, surface: Surface, dispatched: bool) { - // Batch-management rows are observable upstream management requests, not - // the batch's model inference. They stay unpriced even when their route - // resolves through a wildcard alias; batch completion accounting is a - // separate surface with its own attribution rules. - if surface == crate::operation::BATCHES || !dispatched { + // Files, batches, and fine-tuning rows are observable zero-token + // management requests, not model inference. They stay unpriced even when + // their route resolves through a wildcard alias; batch completion + // accounting is a separate surface with its own attribution rules. + if !dispatched + || surface == crate::operation::FILES + || surface == crate::operation::BATCHES + || surface == crate::operation::FINE_TUNING + { return; } let Some(resolved) = crate::attribution::current() else { @@ -1198,17 +1202,29 @@ mod tests { ); assert_eq!(event.resolved_pricing_model, "gpt-4o-2024-08-06"); - let mut batch_event = UsageEvent { - // NO-GUARDRAIL-CHAIN: this focused unit test constructs - // a synthetic pricing event, not a gateway request. - model_id: "wildcard".to_string(), - guardrail_bypassed_reason: String::new(), - applied_guardrails: Vec::new(), - ..Default::default() - }; - apply_wildcard_pricing_model(&mut batch_event, crate::operation::BATCHES, true); - assert!(batch_event.pricing_authority_id.is_empty()); - assert!(batch_event.resolved_pricing_model.is_empty()); + for (surface, management_kind) in [ + (crate::operation::FILES, "file"), + (crate::operation::BATCHES, "batch"), + (crate::operation::FINE_TUNING, "fine-tuning"), + ] { + let mut management_event = UsageEvent { + // NO-GUARDRAIL-CHAIN: this focused unit test constructs + // a synthetic pricing event, not a gateway request. + model_id: "wildcard".to_string(), + guardrail_bypassed_reason: String::new(), + applied_guardrails: Vec::new(), + ..Default::default() + }; + apply_wildcard_pricing_model(&mut management_event, surface, true); + assert!( + management_event.pricing_authority_id.is_empty(), + "{management_kind} management event must not select wildcard pricing" + ); + assert!( + management_event.resolved_pricing_model.is_empty(), + "{management_kind} management event must not select wildcard pricing" + ); + } crate::attribution::note_cache_hit_entry(&cache_entry, "exact"); let cached_attribution = From 9a5fbc2af71a18d3a7e58fa9996e40d087d06fe5 Mon Sep 17 00:00:00 2001 From: Ming Wen Date: Thu, 1 Oct 2026 11:06:58 +0800 Subject: [PATCH 13/16] fix: price only wildcard inference usage --- crates/aisix-proxy/src/count_tokens.rs | 73 +++++++++++++++++++ crates/aisix-proxy/src/jobs.rs | 11 ++- crates/aisix-proxy/src/usage_attr.rs | 45 +++++++++--- .../wildcard-pricing-telemetry-e2e.test.ts | 59 +++++++++++++++ 4 files changed, 175 insertions(+), 13 deletions(-) diff --git a/crates/aisix-proxy/src/count_tokens.rs b/crates/aisix-proxy/src/count_tokens.rs index f15f4d06..37b54e80 100644 --- a/crates/aisix-proxy/src/count_tokens.rs +++ b/crates/aisix-proxy/src/count_tokens.rs @@ -1242,6 +1242,79 @@ mod tests { ); } + /// A wildcard alias still resolves and dispatches on the counting route, + /// but token counting measures a later inference request rather than + /// consuming model tokens itself. Its zero-token row therefore must not + /// carry the wildcard row's concrete pricing authority. + #[tokio::test] + async fn wildcard_count_tokens_routes_upstream_without_pricing_identity() { + use aisix_obs::UsageSink; + + let upstream = MockServer::start().await; + Mock::given(method("POST")) + .and(path("/v1/messages/count_tokens")) + .respond_with( + ResponseTemplate::new(200).set_body_json(serde_json::json!({"input_tokens": 42})), + ) + .expect(1) + .mount(&upstream) + .await; + + let snap = new_snap(&upstream.uri()); + let wildcard: Model = serde_json::from_value(serde_json::json!({ + "display_name": "anthropic/*", + "provider": "anthropic", + "model_name": "*", + "provider_key_id": PK_ID, + "pricing_authority_id": "a3ebdc63-e921-4323-a75c-3b911f950046", + })) + .unwrap(); + snap.models + .insert(ResourceEntry::new("wildcard", wildcard, 1)); + snap.apikeys.insert(apikey_entry(&["*"])); + + let hub = Arc::new(Hub::new()); + hub.register_specialized( + "anthropic", + Arc::new(aisix_provider_anthropic::AnthropicBridge::new()), + ); + let handle = SnapshotHandle::new(snap); + let (tx, mut rx) = tokio::sync::mpsc::channel(8); + let app = crate::build_router( + crate::ProxyState::new(handle, hub, &cfg()) + .without_cache() + .with_usage_sink(UsageSink::new(tx)), + ); + + let requested_model = "anthropic/claude-haiku-4-5-20251001"; + let resp = app + .oneshot(make_req(serde_json::json!({ + "model": requested_model, + "messages": [{"role": "user", "content": "hello"}], + }))) + .await + .unwrap(); + assert_eq!(resp.status(), StatusCode::OK); + + let received = upstream.received_requests().await.unwrap(); + assert_eq!(received.len(), 1); + assert_eq!(received[0].url.path(), "/v1/messages/count_tokens"); + let sent: serde_json::Value = serde_json::from_slice(&received[0].body).unwrap(); + assert_eq!(sent["model"], "claude-haiku-4-5-20251001"); + + let event = tokio::time::timeout(std::time::Duration::from_secs(2), rx.recv()) + .await + .expect("count_tokens must emit a usage event") + .expect("channel open"); + assert_eq!(event.operation, "count_tokens"); + assert_eq!(event.model_id, "wildcard"); + assert_eq!(event.requested_model, requested_model); + assert_eq!(event.prompt_tokens, 0); + assert_eq!(event.completion_tokens, 0); + assert!(event.pricing_authority_id.is_empty()); + assert!(event.resolved_pricing_model.is_empty()); + } + /// Same gate as `/v1/messages`: a body the scan parser rejects is /// refused only when a guardrail would have read it. An output-hook-only /// row resolves into the chain but never sees the request, so the body diff --git a/crates/aisix-proxy/src/jobs.rs b/crates/aisix-proxy/src/jobs.rs index 7510fc21..fde7df9d 100644 --- a/crates/aisix-proxy/src/jobs.rs +++ b/crates/aisix-proxy/src/jobs.rs @@ -2620,6 +2620,7 @@ mod tests { "provider": "openai", "model_name": "*", "provider_key_id": PK_A, + "pricing_authority_id": "a3ebdc63-e921-4323-a75c-3b911f950046", })) .unwrap(); snap.models.insert(ResourceEntry::new("m-a", wildcard, 1)); @@ -2668,6 +2669,10 @@ mod tests { } } let mgmt = mgmt.expect("management event must be emitted"); + assert!( + mgmt.pricing_authority_id.is_empty(), + "a batch-management request must not select wildcard pricing" + ); assert!( mgmt.resolved_pricing_model.is_empty(), "a batch-management request must not select wildcard pricing" @@ -2689,9 +2694,13 @@ mod tests { assert_eq!(agg.cached_prompt_tokens, 2); assert_eq!(agg.provider_model_version, "gpt-4o-2024-08-06"); assert_eq!(agg.requested_model, "jobs/*"); + assert!( + agg.pricing_authority_id.is_empty(), + "a detached batch aggregate must not select wildcard pricing" + ); assert!( agg.resolved_pricing_model.is_empty(), - "an upstream batch output model must not select wildcard pricing" + "a detached batch aggregate must not select wildcard pricing" ); // Second retrieve: management event only — the attribution is diff --git a/crates/aisix-proxy/src/usage_attr.rs b/crates/aisix-proxy/src/usage_attr.rs index 8683fe4a..8140e87f 100644 --- a/crates/aisix-proxy/src/usage_attr.rs +++ b/crates/aisix-proxy/src/usage_attr.rs @@ -439,6 +439,34 @@ pub(crate) fn metric_model_label_pair<'a>( } } +/// Whether a usage surface is a model-inference request whose dispatch +/// identity can name a concrete wildcard price. +/// +/// This is intentionally an allowlist rather than a list of the management +/// surfaces we currently know about. A new zero-token or control-plane route +/// must remain unpriced until it explicitly establishes the same billing +/// contract as a model-inference surface. Batch completion is also absent: +/// it represents detached, after-the-fact output aggregation and has no live +/// request attribution from which to take a wildcard identity. +fn is_billable_inference_surface(surface: Surface) -> bool { + [ + crate::operation::CHAT, + crate::operation::COMPLETIONS, + crate::operation::MESSAGES, + crate::operation::RESPONSES, + crate::operation::EMBEDDINGS, + crate::operation::RERANK, + crate::operation::REALTIME, + crate::operation::IMAGE_GENERATION, + crate::operation::IMAGE_EDIT, + crate::operation::TRANSCRIPTION, + crate::operation::TRANSLATION, + crate::operation::SPEECH, + crate::operation::VIDEO_GENERATION, + ] + .contains(&surface) +} + /// Fill the optional DP-to-CP wildcard-pricing authority at the one usage /// emission chokepoint. Model resolution establishes eligibility from the /// dispatch snapshot and records a complete authority tuple, so this path must @@ -447,15 +475,7 @@ pub(crate) fn metric_model_label_pair<'a>( /// has no upstream call to price. Detached gateway work has no caller /// attribution and therefore cannot price itself as the parent request. fn apply_wildcard_pricing_model(event: &mut UsageEvent, surface: Surface, dispatched: bool) { - // Files, batches, and fine-tuning rows are observable zero-token - // management requests, not model inference. They stay unpriced even when - // their route resolves through a wildcard alias; batch completion - // accounting is a separate surface with its own attribution rules. - if !dispatched - || surface == crate::operation::FILES - || surface == crate::operation::BATCHES - || surface == crate::operation::FINE_TUNING - { + if !dispatched || !is_billable_inference_surface(surface) { return; } let Some(resolved) = crate::attribution::current() else { @@ -1202,7 +1222,8 @@ mod tests { ); assert_eq!(event.resolved_pricing_model, "gpt-4o-2024-08-06"); - for (surface, management_kind) in [ + for (surface, non_inference_kind) in [ + (crate::operation::COUNT_TOKENS, "count-tokens"), (crate::operation::FILES, "file"), (crate::operation::BATCHES, "batch"), (crate::operation::FINE_TUNING, "fine-tuning"), @@ -1218,11 +1239,11 @@ mod tests { apply_wildcard_pricing_model(&mut management_event, surface, true); assert!( management_event.pricing_authority_id.is_empty(), - "{management_kind} management event must not select wildcard pricing" + "{non_inference_kind} non-inference event must not select wildcard pricing" ); assert!( management_event.resolved_pricing_model.is_empty(), - "{management_kind} management event must not select wildcard pricing" + "{non_inference_kind} non-inference event must not select wildcard pricing" ); } diff --git a/tests/e2e/src/cases/wildcard-pricing-telemetry-e2e.test.ts b/tests/e2e/src/cases/wildcard-pricing-telemetry-e2e.test.ts index 247cde20..be8e93c7 100644 --- a/tests/e2e/src/cases/wildcard-pricing-telemetry-e2e.test.ts +++ b/tests/e2e/src/cases/wildcard-pricing-telemetry-e2e.test.ts @@ -56,6 +56,10 @@ function embeddingUpstreamResponse() { }; } +function routedId(raw: string, model: string): string { + return `aisix-${Buffer.from(`${raw};model,${model}`).toString("base64url")}`; +} + describe("wildcard pricing telemetry e2e", () => { let app: SpawnedApp | undefined; let sls: MockSls | undefined; @@ -238,4 +242,59 @@ describe("wildcard pricing telemetry e2e", () => { expect(event.get("resolved_pricing_model")).toBe(EMBEDDING_UPSTREAM_MODEL); expect(event.get("prompt_tokens")).toBe("7"); }); + + test("wildcard-routed job management events remain unpriced", async (ctx) => { + if (!etcdReachable || !app || !sls || !upstream || !wildcardID) { + ctx.skip(); + return; + } + + const model = `openrouter/${KNOWN_MODEL}`; + const calls = [ + [ + "file", + "files", + `/v1/files/${routedId("file-wildcard", model)}`, + "/v1/files/file-wildcard", + ], + [ + "batch", + "batches", + `/v1/batches/${routedId("batch-wildcard", model)}`, + "/v1/batches/batch-wildcard", + ], + [ + "fine-tuning", + "fine_tuning", + `/v1/fine_tuning/jobs/${routedId("ftjob-wildcard", model)}`, + "/v1/fine_tuning/jobs/ftjob-wildcard", + ], + ] as const; + + for (const [kind, operation, gatewayPath, upstreamPath] of calls) { + const before = upstream.receivedRequests.length; + const res = await fetch(`${app.proxyUrl}${gatewayPath}`, { + headers: { authorization: `Bearer ${CALLER_PLAINTEXT}` }, + }); + const body = await res.text(); + expect(res.status, `${kind}: ${body}`).toBe(200); + + const forwarded = upstream.receivedRequests.slice(before); + expect(forwarded).toHaveLength(1); + expect(forwarded[0]!.method).toBe("GET"); + expect(forwarded[0]!.path).toBe(upstreamPath); + + const event = await waitForSlsLog( + sls, + LOGSTORE, + (log) => log.get("operation") === operation && log.get("model_id") === wildcardID, + `${kind} wildcard management usage event`, + ); + expect(event.get("requested_model")).toBe(WILDCARD_ALIAS); + expect(event.get("prompt_tokens")).toBe("0"); + expect(event.get("completion_tokens")).toBe("0"); + expect(event.get("pricing_authority_id")).toBeUndefined(); + expect(event.get("resolved_pricing_model")).toBeUndefined(); + } + }); }); From 579ab0f142f0be6543866625a169d8234d6170c2 Mon Sep 17 00:00:00 2001 From: Ming Wen Date: Thu, 1 Oct 2026 11:21:28 +0800 Subject: [PATCH 14/16] test: cover wildcard pricing boundary surfaces --- crates/aisix-proxy/src/usage_attr.rs | 42 +++++++++++ .../wildcard-pricing-telemetry-e2e.test.ts | 74 +++++++++++++++++++ 2 files changed, 116 insertions(+) diff --git a/crates/aisix-proxy/src/usage_attr.rs b/crates/aisix-proxy/src/usage_attr.rs index 8140e87f..2e715f94 100644 --- a/crates/aisix-proxy/src/usage_attr.rs +++ b/crates/aisix-proxy/src/usage_attr.rs @@ -1172,6 +1172,48 @@ mod tests { assert_eq!((m.as_ref(), u.as_ref()), ("no-such/model", "raw-upstream")); } + #[test] + fn billable_inference_surface_allowlist_is_exhaustive() { + for surface in [ + crate::operation::CHAT, + crate::operation::COMPLETIONS, + crate::operation::MESSAGES, + crate::operation::RESPONSES, + crate::operation::EMBEDDINGS, + crate::operation::RERANK, + crate::operation::REALTIME, + crate::operation::IMAGE_GENERATION, + crate::operation::IMAGE_EDIT, + crate::operation::TRANSCRIPTION, + crate::operation::TRANSLATION, + crate::operation::SPEECH, + crate::operation::VIDEO_GENERATION, + ] { + assert!( + is_billable_inference_surface(surface), + "{} must retain wildcard pricing attribution", + surface.operation, + ); + } + + for surface in [ + crate::operation::COUNT_TOKENS, + crate::operation::FILES, + crate::operation::BATCHES, + crate::operation::FINE_TUNING, + crate::operation::MCP, + crate::operation::A2A, + crate::operation::PASSTHROUGH, + crate::operation::BATCH_COMPLETION, + ] { + assert!( + !is_billable_inference_surface(surface), + "{} must not select a wildcard pricing identity", + surface.operation, + ); + } + } + #[tokio::test] async fn request_attribution_stamps_only_concrete_wildcard_pricing_authorities() { use aisix_core::resource::ResourceEntry; diff --git a/tests/e2e/src/cases/wildcard-pricing-telemetry-e2e.test.ts b/tests/e2e/src/cases/wildcard-pricing-telemetry-e2e.test.ts index be8e93c7..506281c5 100644 --- a/tests/e2e/src/cases/wildcard-pricing-telemetry-e2e.test.ts +++ b/tests/e2e/src/cases/wildcard-pricing-telemetry-e2e.test.ts @@ -31,6 +31,9 @@ const EMBEDDING_REQUEST_MODEL = "embedding/embedding-3-small"; const EMBEDDING_UPSTREAM_MODEL = "text-embedding-3-small"; const EMBEDDING_INPUT = "price this embedding"; const EMBEDDING_VECTOR = [0.1, 0.2, 0.3]; +const COUNT_TOKENS_WILDCARD_ALIAS = "anthropic/*"; +const COUNT_TOKENS_REQUEST_MODEL = "anthropic/claude-haiku-4-5-20251001"; +const COUNT_TOKENS_UPSTREAM_MODEL = "claude-haiku-4-5-20251001"; function upstreamResponse() { return { @@ -65,8 +68,10 @@ describe("wildcard pricing telemetry e2e", () => { let sls: MockSls | undefined; let upstream: OpenAiUpstream | undefined; let embeddingUpstream: OpenAiUpstream | undefined; + let countTokensUpstream: OpenAiUpstream | undefined; let wildcardID = ""; let embeddingWildcardID = ""; + let countTokensWildcardID = ""; let etcdReachable = false; beforeAll(async () => { @@ -77,6 +82,7 @@ describe("wildcard pricing telemetry e2e", () => { sls = await startMockSls(); upstream = await startOpenAiUpstream({ nonStreamBody: upstreamResponse() }); embeddingUpstream = await startOpenAiUpstream({ nonStreamBody: embeddingUpstreamResponse() }); + countTokensUpstream = await startOpenAiUpstream({ nonStreamBody: { input_tokens: 42 } }); app = await spawnApp({ extraEnv: { [`SLS_CRED_${CREDENTIAL_REF.toUpperCase()}_AK_ID`]: "mock-akid", @@ -125,6 +131,23 @@ describe("wildcard pricing telemetry e2e", () => { embedding: { dimensions: EMBEDDING_VECTOR.length }, }); embeddingWildcardID = embeddingWildcard.id; + const countTokensProviderKey = await seed.createProviderKey({ + display_name: "wildcard-pricing-count-tokens-pk", + provider: "anthropic", + adapter: "anthropic", + secret: "sk-mock", + // The Anthropic bridge appends `/v1/messages/count_tokens` to its + // bare provider base URL. + api_base: countTokensUpstream.baseUrl, + }); + const countTokensWildcard = await seed.createModel({ + display_name: COUNT_TOKENS_WILDCARD_ALIAS, + provider: "anthropic", + model_name: "*", + provider_key_id: countTokensProviderKey.id, + pricing_authority_id: PRICING_AUTHORITY_ID, + }); + countTokensWildcardID = countTokensWildcard.id; // Seeded last: a successful models-list gate proves that all preceding // resources, including the exporter, are in the same gateway snapshot. @@ -142,6 +165,7 @@ describe("wildcard pricing telemetry e2e", () => { await app?.exit(); await upstream?.close(); await embeddingUpstream?.close(); + await countTokensUpstream?.close(); await sls?.close(); }); @@ -243,6 +267,56 @@ describe("wildcard pricing telemetry e2e", () => { expect(event.get("prompt_tokens")).toBe("7"); }); + test("wildcard count_tokens remains unpriced after a real Anthropic dispatch", async (ctx) => { + if (!etcdReachable || !app || !sls || !countTokensUpstream || !countTokensWildcardID) { + ctx.skip(); + return; + } + + const baseline = countTokensUpstream.receivedRequests.length; + const res = await fetch(`${app.proxyUrl}/v1/messages/count_tokens`, { + method: "POST", + headers: { + "content-type": "application/json", + "x-api-key": CALLER_PLAINTEXT, + "anthropic-version": "2023-06-01", + }, + body: JSON.stringify({ + model: COUNT_TOKENS_REQUEST_MODEL, + messages: [{ role: "user", content: "count this prompt" }], + }), + }); + const body = await res.text(); + expect(res.status, body).toBe(200); + expect(JSON.parse(body)).toMatchObject({ input_tokens: 42 }); + + const calls = countTokensUpstream.receivedRequests.slice(baseline); + expect(calls).toHaveLength(1); + expect(calls[0]!.method).toBe("POST"); + expect(calls[0]!.path).toBe("/v1/messages/count_tokens"); + expect(JSON.parse(calls[0]!.body)).toMatchObject({ + model: COUNT_TOKENS_UPSTREAM_MODEL, + messages: [{ role: "user", content: "count this prompt" }], + }); + + const event = await waitForSlsLog( + sls, + LOGSTORE, + (log) => + log.get("operation") === "count_tokens" && + log.get("requested_model") === COUNT_TOKENS_REQUEST_MODEL, + `usage event for ${COUNT_TOKENS_REQUEST_MODEL}`, + ); + expect(event.get("model_id")).toBe(countTokensWildcardID); + expect(event.get("prompt_tokens")).toBe("0"); + expect(event.get("completion_tokens")).toBe("0"); + expect(event.get("cached_prompt_tokens")).toBeUndefined(); + expect(event.get("reasoning_tokens")).toBeUndefined(); + expect(event.get("total_tokens")).toBeUndefined(); + expect(event.get("pricing_authority_id")).toBeUndefined(); + expect(event.get("resolved_pricing_model")).toBeUndefined(); + }); + test("wildcard-routed job management events remain unpriced", async (ctx) => { if (!etcdReachable || !app || !sls || !upstream || !wildcardID) { ctx.skip(); From cec7e6af37b6b360f009965a32e077886223db40 Mon Sep 17 00:00:00 2001 From: Ming Wen Date: Thu, 1 Oct 2026 15:08:46 +0800 Subject: [PATCH 15/16] fix: retain wildcard pricing in detached telemetry --- crates/aisix-proxy/src/attribution.rs | 119 ---------- crates/aisix-proxy/src/jobs.rs | 108 +++++---- crates/aisix-proxy/src/realtime.rs | 34 ++- crates/aisix-proxy/src/usage_attr.rs | 147 +++++++++++-- .../wildcard-pricing-telemetry-e2e.test.ts | 207 +++++++++++++++++- 5 files changed, 425 insertions(+), 190 deletions(-) diff --git a/crates/aisix-proxy/src/attribution.rs b/crates/aisix-proxy/src/attribution.rs index 33a502fa..abef5633 100644 --- a/crates/aisix-proxy/src/attribution.rs +++ b/crates/aisix-proxy/src/attribution.rs @@ -615,21 +615,6 @@ pub(crate) struct RequestAttribution { } impl RequestAttribution { - /// A continuation for a detached realtime session. It deliberately copies - /// only the wildcard pricing tuple: the HTTP request's body counters and - /// pending access log are finalized when its upgrade response completes, - /// while the session owns a separate terminal event and access-log line. - fn wildcard_pricing_continuation(&self) -> Self { - let resolved = self.get(); - let continuation = Self::default(); - let mut cell = continuation.lock(); - cell.resolved.wildcard_pricing_model_id = resolved.wildcard_pricing_model_id; - cell.resolved.wildcard_pricing_authority_id = resolved.wildcard_pricing_authority_id; - cell.resolved.wildcard_pricing_model = resolved.wildcard_pricing_model; - drop(cell); - continuation - } - /// A sub-call scope that cannot change the parent request's target /// attribution but can still account for gateway-initiated embeddings on /// the parent's eventual terminal usage event. @@ -946,18 +931,6 @@ pub(crate) fn current() -> Option { CURRENT.try_with(|a| a.get()).ok() } -/// A fresh attribution cell for a detached realtime session. -/// -/// A WebSocket upgrade moves its session onto axum's upgrade task, which does -/// not inherit Tokio task-locals. The session needs a wildcard pricing tuple -/// captured before the upgrade, but must not reuse the HTTP cell: the latter -/// owns the completed upgrade response's body counters and access-log line. -pub(crate) fn current_wildcard_pricing_continuation() -> Option> { - CURRENT - .try_with(|parent| Arc::new(parent.wildcard_pricing_continuation())) - .ok() -} - /// Record one actual gateway-initiated embedding bridge call. This is safe /// inside [`detached`]: that scope keeps a throwaway target cell but shares /// the parent's [`GatewayEmbeddingLedger`]. Calls outside an HTTP request are @@ -1384,98 +1357,6 @@ mod tests { assert!(current().is_none()); } - #[test] - fn wildcard_pricing_continuation_keeps_only_the_pricing_tuple() { - let parent = RequestAttribution::default(); - parent.track(tracing::info_span!("parent_request")); - parent.note_head_written(); - parent.add_request_bytes(41); - parent.note_request_complete(); - parent.add_response_bytes(17); - parent.park_access_log( - &aisix_obs::AccessLog { - method: "POST", - path: "/v1/realtime", - status: 101, - latency: Duration::from_secs(0), - duration: Duration::from_secs(0), - provider: Some("openai"), - model: Some("realtime/*"), - upstream_model: Some("gpt-realtime"), - provider_key_id: Some("pk-1"), - api_key_id: Some("api-key"), - prompt_tokens: Some(1), - completion_tokens: Some(2), - total_tokens: Some(3), - request_id: "req-parent", - provider_request_id: Some("resp-parent"), - served_by_model: None, - routing_attempt_count: None, - routing_fallback_count: None, - error_kind: None, - error: None, - mcp: None, - cache: None, - request_body_bytes: None, - response_body_bytes: None, - }, - None, - ); - { - let mut cell = parent.lock(); - cell.resolved = Resolved { - requested_model: "realtime/customer-model".into(), - provider: "openai".into(), - upstream_model: "gpt-realtime".into(), - provider_key_id: "pk-1".into(), - wildcard_pricing_model_id: "wildcard-row".into(), - wildcard_pricing_authority_id: "a3ebdc63-e921-4323-a75c-3b911f950046".into(), - wildcard_pricing_model: "gpt-realtime".into(), - cache_hit_layer: Some("exact"), - }; - cell.cancel.api_key_id = "api-key".into(); - cell.cancel.emitted_terminal = true; - cell.pending_log = Some(PendingAccessLog::new( - "POST", - "/v1/realtime", - "req-parent", - "api-key", - Instant::now(), - )); - cell.stream_owns_log = true; - cell.finished = true; - assert!( - cell.ready_line.is_some(), - "premise: parent owns a pending access log" - ); - } - - let continuation = parent.wildcard_pricing_continuation(); - let resolved = continuation.get(); - assert_eq!(resolved.wildcard_pricing_model_id, "wildcard-row"); - assert_eq!( - resolved.wildcard_pricing_authority_id, - "a3ebdc63-e921-4323-a75c-3b911f950046" - ); - assert_eq!(resolved.wildcard_pricing_model, "gpt-realtime"); - assert!(resolved.requested_model.is_empty()); - assert!(resolved.provider.is_empty()); - assert!(resolved.upstream_model.is_empty()); - assert!(resolved.provider_key_id.is_empty()); - assert!(resolved.cache_hit_layer.is_none()); - assert_eq!(continuation.body_sizes(true), (None, Some(0))); - - let cell = continuation.lock(); - assert!(cell.cancel.api_key_id.is_empty()); - assert!(!cell.cancel.emitted_terminal); - assert!(cell.pending_log.is_none()); - assert!(!cell.stream_owns_log); - assert!(cell.request_span.is_none()); - assert!(!cell.head_written); - assert!(!cell.finished); - assert!(cell.ready_line.is_none()); - } - /// A gateway-owned embedding must retain the caller's actual target while /// still reaching that caller's one terminal usage event. This is the /// reason the detached cell shares only the child-call ledger, not its diff --git a/crates/aisix-proxy/src/jobs.rs b/crates/aisix-proxy/src/jobs.rs index fde7df9d..f1b0a608 100644 --- a/crates/aisix-proxy/src/jobs.rs +++ b/crates/aisix-proxy/src/jobs.rs @@ -144,6 +144,10 @@ pub(crate) struct JobTarget { pub pk_entry: Arc>, pub secret: String, pub adapter: Adapter, + /// The exact, authorized routing selector that dispatched this request. + /// It is embedded in returned resource ids so later batch/file retrievals + /// resolve the same concrete wildcard model rather than the `*` row. + routing_model: String, /// The ProviderKey's rendered `default_headers`, resolved once when the /// target is resolved so every round-trip on this surface (upload, /// poll, download) sends the same set (AISIX-Cloud#1112). @@ -161,6 +165,9 @@ impl JobTarget { fn display_name(&self) -> &str { &self.model_entry.value.display_name } + fn routing_model(&self) -> &str { + &self.routing_model + } fn provider_label(&self) -> &str { self.model_entry.value.provider.as_deref().unwrap_or("") } @@ -199,6 +206,7 @@ pub(crate) fn resolve_target( wanted: Option<&str>, client_ctx: &ClientContext, ) -> Result { + let routing_model = wanted.map(str::to_string); let model_entry = match wanted { Some(name) => { let entry = crate::model_resolve::resolve_model(snapshot, name) @@ -265,7 +273,9 @@ pub(crate) fn resolve_target( ); let forwarded_client = aisix_gateway::ForwardedClientHeaders::resolve(&header_ctx); let extra_headers = aisix_gateway::resolve_default_headers(&header_ctx); + let routing_model = routing_model.unwrap_or_else(|| model.display_name.clone()); Ok(JobTarget { + routing_model, model_entry, pk_entry, secret, @@ -1035,7 +1045,7 @@ pub(crate) async fn create_file( &request_id, ) .await?; - let model = target.display_name().to_string(); + let model = target.routing_model().to_string(); Ok(( json_response(status, &resp_headers, bytes, Some(&model)), target, @@ -1308,7 +1318,7 @@ pub(crate) async fn create_batch( &mut applied, ) .await?; - let model = target.display_name().to_string(); + let model = target.routing_model().to_string(); Ok(( json_response(status, &resp_headers, bytes, Some(&model)), target, @@ -1402,7 +1412,7 @@ pub(crate) async fn get_batch( } } - let model = target.display_name().to_string(); + let model = target.routing_model().to_string(); Ok(( json_response(status, &resp_headers, bytes, Some(&model)), target, @@ -1610,7 +1620,7 @@ pub(crate) async fn create_ft_job( &mut applied, ) .await?; - let model = target.display_name().to_string(); + let model = target.routing_model().to_string(); Ok(( json_response(status, &resp_headers, bytes, Some(&model)), target, @@ -1841,7 +1851,7 @@ async fn forward_simple( } resp } else { - let model = spec.rewrite_ids.then(|| target.display_name().to_string()); + let model = spec.rewrite_ids.then(|| target.routing_model().to_string()); json_response(status, &resp_headers, bytes, model.as_deref()) }; Ok((resp, target)) @@ -1923,6 +1933,11 @@ fn maybe_attribute_batch( let jwt = auth.jwt.clone(); let model_id = target.model_entry.id.clone(); let display_name = target.display_name().to_string(); + // This runs while the retrieval request still owns its attribution + // scope. The download/aggregation task below is detached, so it must + // carry the verified dispatch-time tuple rather than trying to derive a + // price from the provider's output line or a later snapshot. + let wildcard_pricing = crate::usage_attr::capture_wildcard_pricing_identity(); // Resolved before the spawn, off the live snapshot, through the same // index `least_cost` ranks with: `pricing_key` first, inline `cost` // second. The completed batch is priced at what the model costs when @@ -1950,6 +1965,7 @@ fn maybe_attribute_batch( user_name.as_deref(), &model_id, &display_name, + wildcard_pricing, cost.as_ref(), &pk_id, &secret, @@ -1984,6 +2000,7 @@ async fn attribute_batch_usage( user_name: Option<&str>, model_id: &str, display_name: &str, + wildcard_pricing: Option, cost: Option<&aisix_core::models::model::ModelCost>, pk_id: &str, secret: &str, @@ -2028,9 +2045,9 @@ async fn attribute_batch_usage( } let body = resp.bytes().await.map_err(|e| e.to_string())?; - // Keep the provider-reported model only as diagnostic version data. - // A batch completion has no persisted per-line dispatch identity, so an - // upstream response field must not select a wildcard catalog price. + // Keep the provider-reported model only as diagnostic version data. The + // captured dispatch tuple, not an upstream output field, selects the + // wildcard catalog price for every completed batch slice. #[derive(Default)] struct Agg { prompt: u64, @@ -2071,8 +2088,6 @@ async fn attribute_batch_usage( let snap = state.snapshot.load(); let pk = crate::usage_attr::ResolvedPk::resolve(&snap, pk_id); - // Same exporter set for every model slice of one batch. - let exporters = crate::usage_attr::live_exporters(state, &snap); let multi = per_model.len() > 1; for (idx, (provider_model, agg)) in per_model.iter().enumerate() { let request_id = batch_attribution_request_id(raw_batch_id, idx, multi); @@ -2094,29 +2109,35 @@ async fn attribute_batch_usage( .map(|c| c.calculate(agg.prompt, agg.completion)) .unwrap_or(0.0), inbound_protocol: "batch".to_string(), - // Set here rather than by the emit chokepoint: this path - // deliberately bypasses it (no live request, so no trace - // bundle), and both labels still come from one constant. - operation: crate::operation::BATCH_COMPLETION.operation.to_string(), ..Default::default() }; + crate::usage_attr::apply_captured_wildcard_pricing_identity( + &mut event, + crate::operation::BATCH_COMPLETION, + /* dispatched */ true, + wildcard_pricing.as_ref(), + ); crate::usage_attr::apply_pk_telemetry(&mut event, &pk); // Attribution names the identity that observed completion — the // same caller the event's api_key_id already reflects. crate::usage_attr::apply_caller_identity(&mut event, jwt, user_id, user_name); let usage_model = crate::usage_attr::usage_event_model_label(&snap, &event.requested_model).into_owned(); - state.usage_sink.try_emit( - crate::operation::BATCH_COMPLETION.handler, - event.clone(), + // Completion attribution is a real terminal inference event even + // though it has no live request trace. Send it through the one + // chokepoint so CP telemetry and exporter fan-out receive the same + // captured pricing tuple. + crate::usage_attr::emit_usage( + state, + &snap, + crate::operation::BATCH_COMPLETION, + event, crate::usage_attr::usage_event_labels(&usage_model, &pk), + None, + None, + /* terminal */ true, + /* dispatched */ true, ); - // A background poll attributes usage after the fact — there is no - // live request and therefore no trace bundle; the exporter falls - // back to the legacy flat span (AISIX-Cloud#1279). - state - .otlp_fan_out - .fan_out(&event, None, None, exporters.iter().map(|e| &e.value)); tracing::info!( batch_id = %raw_batch_id, provider_model = %provider_model, @@ -2531,11 +2552,18 @@ mod tests { snap.provider_keys .insert(openai_pk(PK_B, &upstream_b.uri())); snap.models.insert(model("m-a", "jobs-a", PK_A)); - snap.models.insert(model("m-b", "jobs-b", PK_B)); + let wildcard: Model = serde_json::from_value(serde_json::json!({ + "display_name": "jobs/*", + "provider": "openai", + "model_name": "*", + "provider_key_id": PK_B, + })) + .unwrap(); + snap.models.insert(ResourceEntry::new("m-b", wildcard, 1)); snap.apikeys.insert(apikey_entry(&["*"])); let app = build_app(snap); - let encoded = encode_routed_id("file-realB", "jobs-b"); + let encoded = encode_routed_id("file-realB", "jobs/gpt-4o-2024-08-06"); let body = serde_json::json!({ "input_file_id": encoded, "endpoint": "/v1/chat/completions", @@ -2561,11 +2589,17 @@ mod tests { let v: Value = serde_json::from_slice(&bytes).unwrap(); assert_eq!( decode_routed_id(v["id"].as_str().unwrap()), - Some(("batch_777".to_string(), "jobs-b".to_string())) + Some(( + "batch_777".to_string(), + "jobs/gpt-4o-2024-08-06".to_string() + )) ); assert_eq!( decode_routed_id(v["input_file_id"].as_str().unwrap()), - Some(("file-realB".to_string(), "jobs-b".to_string())) + Some(( + "file-realB".to_string(), + "jobs/gpt-4o-2024-08-06".to_string() + )) ); assert!( upstream_a.received_requests().await.unwrap().is_empty(), @@ -2591,7 +2625,7 @@ mod tests { "id": "batch_req_1", "custom_id": "r1", "response": {"status_code": 200, "body": { - "model": "gpt-4o-2024-08-06", + "model": "provider-reported-batch-model", "usage": {"prompt_tokens": 10, "completion_tokens": 5, "prompt_tokens_details": {"cached_tokens": 2}} }} @@ -2600,7 +2634,7 @@ mod tests { "id": "batch_req_2", "custom_id": "r2", "response": {"status_code": 200, "body": { - "model": "gpt-4o-2024-08-06", + "model": "provider-reported-batch-model", "usage": {"prompt_tokens": 7, "completion_tokens": 3} }} }); @@ -2647,7 +2681,7 @@ mod tests { let v: Value = serde_json::from_slice(&bytes).unwrap(); assert_eq!( decode_routed_id(v["output_file_id"].as_str().unwrap()), - Some(("file-out".to_string(), "jobs/*".to_string())) + Some(("file-out".to_string(), "jobs/gpt-4o-2024-08-06".to_string())) ); // Two events expected: the zero-token management event plus ONE @@ -2692,15 +2726,15 @@ mod tests { assert_eq!(agg.prompt_tokens, 17); assert_eq!(agg.completion_tokens, 8); assert_eq!(agg.cached_prompt_tokens, 2); - assert_eq!(agg.provider_model_version, "gpt-4o-2024-08-06"); + assert_eq!(agg.provider_model_version, "provider-reported-batch-model"); assert_eq!(agg.requested_model, "jobs/*"); - assert!( - agg.pricing_authority_id.is_empty(), - "a detached batch aggregate must not select wildcard pricing" + assert_eq!( + agg.pricing_authority_id, "a3ebdc63-e921-4323-a75c-3b911f950046", + "the aggregate must retain the captured wildcard authority" ); - assert!( - agg.resolved_pricing_model.is_empty(), - "a detached batch aggregate must not select wildcard pricing" + assert_eq!( + agg.resolved_pricing_model, "gpt-4o-2024-08-06", + "the aggregate must use the dispatch model, not provider output" ); // Second retrieve: management event only — the attribution is diff --git a/crates/aisix-proxy/src/realtime.rs b/crates/aisix-proxy/src/realtime.rs index 1fc9eb83..fb5a53fe 100644 --- a/crates/aisix-proxy/src/realtime.rs +++ b/crates/aisix-proxy/src/realtime.rs @@ -271,10 +271,10 @@ pub(crate) async fn realtime( let state2 = state.clone(); let client2 = client.clone(); // `on_upgrade` runs on a new Tokio task, which does not inherit - // the request task-local. Copy only the wildcard pricing tuple: - // the HTTP cell is already owned by the completed upgrade - // response and cannot also own the session's terminal log. - let attribution = crate::attribution::current_wildcard_pricing_continuation(); + // the request task-local. Copy the verified wildcard price + // selection now; the HTTP cell is already owned by the completed + // upgrade response and cannot own the session terminal event. + let wildcard_pricing = crate::usage_attr::capture_wildcard_pricing_identity(); // `on_upgrade` runs the session on a detached task, so the // request span has to be attached to the future rather than // inherited — without it the session's guardrail checks log @@ -290,7 +290,7 @@ pub(crate) async fn realtime( client2, request_id, started, - attribution, + wildcard_pricing, ) .await; } @@ -739,13 +739,18 @@ async fn run_session( client: ClientContext, request_id: String, started: Instant, - attribution: Option>, + wildcard_pricing: Option, ) { - let session = run_session_inner(state, prep, client_ws, client, request_id, started); - match attribution { - Some(cell) => crate::attribution::scope(cell, session).await, - None => session.await, - } + run_session_inner( + state, + prep, + client_ws, + client, + request_id, + started, + wildcard_pricing, + ) + .await; } async fn run_session_inner( @@ -755,6 +760,7 @@ async fn run_session_inner( client: ClientContext, request_id: String, started: Instant, + wildcard_pricing: Option, ) { let Prepared { auth, @@ -1123,6 +1129,12 @@ async fn run_session_inner( guardrail_bypassed_reason: crate::usage_attr::bypass_reason(&audit), ..Default::default() }; + crate::usage_attr::apply_captured_wildcard_pricing_identity( + &mut event, + crate::operation::REALTIME, + /* dispatched */ true, + wildcard_pricing.as_ref(), + ); crate::usage_attr::apply_pk_telemetry(&mut event, &pk); crate::usage_attr::apply_caller_identity( &mut event, diff --git a/crates/aisix-proxy/src/usage_attr.rs b/crates/aisix-proxy/src/usage_attr.rs index 2e715f94..4806379a 100644 --- a/crates/aisix-proxy/src/usage_attr.rs +++ b/crates/aisix-proxy/src/usage_attr.rs @@ -445,9 +445,10 @@ pub(crate) fn metric_model_label_pair<'a>( /// This is intentionally an allowlist rather than a list of the management /// surfaces we currently know about. A new zero-token or control-plane route /// must remain unpriced until it explicitly establishes the same billing -/// contract as a model-inference surface. Batch completion is also absent: -/// it represents detached, after-the-fact output aggregation and has no live -/// request attribution from which to take a wildcard identity. +/// contract as a model-inference surface. `BATCH_COMPLETION` is the one +/// detached terminal surface: it may use an identity captured from the +/// completed batch's dispatch request, while the `BATCHES` management route +/// remains unpriced. fn is_billable_inference_surface(surface: Surface) -> bool { [ crate::operation::CHAT, @@ -463,10 +464,67 @@ fn is_billable_inference_surface(surface: Surface) -> bool { crate::operation::TRANSLATION, crate::operation::SPEECH, crate::operation::VIDEO_GENERATION, + crate::operation::BATCH_COMPLETION, ] .contains(&surface) } +/// A complete wildcard pricing selection captured at dispatch time. +/// +/// Detached terminal work cannot depend on a Tokio task-local surviving past +/// its originating request. Keep only the price-selection tuple: unlike the +/// rest of request attribution, it is safe and necessary to carry to the +/// terminal event, and its `model_id` prevents it being applied to another +/// attempt or model row. +#[derive(Clone, Debug, PartialEq, Eq)] +pub(crate) struct WildcardPricingIdentity { + model_id: String, + pricing_authority_id: String, + resolved_pricing_model: String, +} + +/// Snapshot the current request's dispatch-time wildcard price selection for +/// a detached terminal emitter. Cache hits, fixed rows, incomplete legacy +/// projections, and code outside a request scope intentionally yield `None`. +pub(crate) fn capture_wildcard_pricing_identity() -> Option { + let resolved = crate::attribution::current()?; + if resolved.cache_hit_layer.is_some() + || resolved.wildcard_pricing_model_id.is_empty() + || resolved.wildcard_pricing_authority_id.is_empty() + || resolved.wildcard_pricing_model.is_empty() + { + return None; + } + Some(WildcardPricingIdentity { + model_id: resolved.wildcard_pricing_model_id, + pricing_authority_id: resolved.wildcard_pricing_authority_id, + resolved_pricing_model: resolved.wildcard_pricing_model, + }) +} + +/// Apply a wildcard pricing identity captured at dispatch time. The caller +/// still has to prove that this terminal event describes an upstream dispatch +/// on a priced inference surface; a captured tuple can never price a +/// management, cached, failed-before-dispatch, or different-model event. +pub(crate) fn apply_captured_wildcard_pricing_identity( + event: &mut UsageEvent, + surface: Surface, + dispatched: bool, + identity: Option<&WildcardPricingIdentity>, +) { + if !dispatched || !is_billable_inference_surface(surface) { + return; + } + let Some(identity) = identity else { + return; + }; + if identity.model_id != event.model_id { + return; + } + event.pricing_authority_id = identity.pricing_authority_id.clone(); + event.resolved_pricing_model = identity.resolved_pricing_model.clone(); +} + /// Fill the optional DP-to-CP wildcard-pricing authority at the one usage /// emission chokepoint. Model resolution establishes eligibility from the /// dispatch snapshot and records a complete authority tuple, so this path must @@ -475,21 +533,8 @@ fn is_billable_inference_surface(surface: Surface) -> bool { /// has no upstream call to price. Detached gateway work has no caller /// attribution and therefore cannot price itself as the parent request. fn apply_wildcard_pricing_model(event: &mut UsageEvent, surface: Surface, dispatched: bool) { - if !dispatched || !is_billable_inference_surface(surface) { - return; - } - let Some(resolved) = crate::attribution::current() else { - return; - }; - if resolved.cache_hit_layer.is_some() || resolved.wildcard_pricing_model_id != event.model_id { - return; - } - if !resolved.wildcard_pricing_authority_id.is_empty() - && !resolved.wildcard_pricing_model.is_empty() - { - event.pricing_authority_id = resolved.wildcard_pricing_authority_id; - event.resolved_pricing_model = resolved.wildcard_pricing_model; - } + let identity = capture_wildcard_pricing_identity(); + apply_captured_wildcard_pricing_identity(event, surface, dispatched, identity.as_ref()); } /// Stamp the five per-PK attribution fields onto an in-progress UsageEvent, @@ -1188,6 +1233,7 @@ mod tests { crate::operation::TRANSLATION, crate::operation::SPEECH, crate::operation::VIDEO_GENERATION, + crate::operation::BATCH_COMPLETION, ] { assert!( is_billable_inference_surface(surface), @@ -1204,7 +1250,6 @@ mod tests { crate::operation::MCP, crate::operation::A2A, crate::operation::PASSTHROUGH, - crate::operation::BATCH_COMPLETION, ] { assert!( !is_billable_inference_surface(surface), @@ -1214,6 +1259,66 @@ mod tests { } } + #[test] + fn captured_wildcard_pricing_identity_only_prices_the_matching_dispatched_terminal() { + let identity = WildcardPricingIdentity { + model_id: "wildcard".to_string(), + pricing_authority_id: "a3ebdc63-e921-4323-a75c-3b911f950046".to_string(), + resolved_pricing_model: "gpt-4o-2024-08-06".to_string(), + }; + + let mut completed_batch = UsageEvent { + model_id: "wildcard".to_string(), + ..Default::default() + }; + apply_captured_wildcard_pricing_identity( + &mut completed_batch, + crate::operation::BATCH_COMPLETION, + true, + Some(&identity), + ); + assert_eq!( + completed_batch.pricing_authority_id, + "a3ebdc63-e921-4323-a75c-3b911f950046" + ); + assert_eq!(completed_batch.resolved_pricing_model, "gpt-4o-2024-08-06"); + + for (surface, dispatched, model_id, case) in [ + (crate::operation::BATCHES, true, "wildcard", "management"), + ( + crate::operation::BATCH_COMPLETION, + false, + "wildcard", + "undispatched", + ), + ( + crate::operation::BATCH_COMPLETION, + true, + "other", + "different model", + ), + ] { + let mut event = UsageEvent { + model_id: model_id.to_string(), + ..Default::default() + }; + apply_captured_wildcard_pricing_identity( + &mut event, + surface, + dispatched, + Some(&identity), + ); + assert!( + event.pricing_authority_id.is_empty(), + "{case} event must not select wildcard pricing" + ); + assert!( + event.resolved_pricing_model.is_empty(), + "{case} event must not select wildcard pricing" + ); + } + } + #[tokio::test] async fn request_attribution_stamps_only_concrete_wildcard_pricing_authorities() { use aisix_core::resource::ResourceEntry; @@ -1294,6 +1399,10 @@ mod tests { crate::attribution::current().expect("in request attribution scope"); assert_eq!(cached_attribution.cache_hit_layer, Some("exact")); assert_eq!(cached_attribution.upstream_model, "*"); + assert!( + capture_wildcard_pricing_identity().is_none(), + "a cache hit must not hand a detached emitter a wildcard price" + ); // `note_cache_hit_entry` clears the identity above. Restore // one here to pin the separate emission gate too: a cache // hit is never billable as an upstream wildcard dispatch. diff --git a/tests/e2e/src/cases/wildcard-pricing-telemetry-e2e.test.ts b/tests/e2e/src/cases/wildcard-pricing-telemetry-e2e.test.ts index 506281c5..61e3bbc6 100644 --- a/tests/e2e/src/cases/wildcard-pricing-telemetry-e2e.test.ts +++ b/tests/e2e/src/cases/wildcard-pricing-telemetry-e2e.test.ts @@ -13,10 +13,12 @@ import { type SpawnedApp, } from "../harness/index.js"; -// E2E for AISIX-Cloud#1746: the configured wildcard row remains the model -// identity, while the concrete upstream model travels on the usage-export -// wire as `resolved_pricing_model`. AISIX-Cloud's real PostgreSQL receiver -// uses that field only when this row's configured model_name is a template. +// Gateway/exporter boundary regression for AISIX-Cloud#1746: the configured +// wildcard row remains the model identity, while the concrete upstream model +// travels on the usage-export wire as `resolved_pricing_model`. This runs a +// real gateway process but `startMockSls` is only the exporter protocol +// receiver; it is not evidence for the CP/DPM/PostgreSQL ingestion contract, +// which belongs to AISIX-Cloud's mTLS telemetry E2E. const CALLER_PLAINTEXT = "sk-wildcard-pricing-telemetry-caller"; const CALLER_KEY_HASH = createHash("sha256").update(CALLER_PLAINTEXT).digest("hex"); @@ -34,6 +36,11 @@ const EMBEDDING_VECTOR = [0.1, 0.2, 0.3]; const COUNT_TOKENS_WILDCARD_ALIAS = "anthropic/*"; const COUNT_TOKENS_REQUEST_MODEL = "anthropic/claude-haiku-4-5-20251001"; const COUNT_TOKENS_UPSTREAM_MODEL = "claude-haiku-4-5-20251001"; +const STREAMING_WILDCARD_ALIAS = "stream/*"; +const STREAMING_REQUEST_MODEL = "stream/gpt-4o-2024-08-06"; +const STREAMING_UPSTREAM_MODEL = "gpt-4o-2024-08-06"; +const BATCH_WILDCARD_ALIAS = "batch/*"; +const BATCH_REQUEST_MODEL = "batch/gpt-4o-2024-08-06"; function upstreamResponse() { return { @@ -59,6 +66,57 @@ function embeddingUpstreamResponse() { }; } +function streamEvents(): string[] { + return [ + JSON.stringify({ + id: "chatcmpl-wildcard-stream", + object: "chat.completion.chunk", + created: 0, + model: "provider-response-stream-model", + choices: [{ index: 0, delta: { role: "assistant", content: "ok" }, finish_reason: null }], + }), + JSON.stringify({ + id: "chatcmpl-wildcard-stream", + object: "chat.completion.chunk", + created: 0, + model: "provider-response-stream-model", + choices: [], + usage: { prompt_tokens: 11, completion_tokens: 6, total_tokens: 17 }, + }), + "[DONE]", + ]; +} + +function completedBatchResponse() { + return { + id: "batch-wildcard-completed", + object: "batch", + status: "completed", + input_file_id: "file-wildcard-input", + output_file_id: "file-wildcard-output", + }; +} + +function completedBatchOutput(): string { + return `${JSON.stringify({ + id: "batch-request-1", + custom_id: "request-1", + response: { + status_code: 200, + body: { + // The provider's response field is deliberately different: it is + // diagnostic only and must not override the dispatch price model. + model: "provider-reported-batch-model", + usage: { + prompt_tokens: 13, + completion_tokens: 7, + prompt_tokens_details: { cached_tokens: 3 }, + }, + }, + }, + })}\n`; +} + function routedId(raw: string, model: string): string { return `aisix-${Buffer.from(`${raw};model,${model}`).toString("base64url")}`; } @@ -69,9 +127,13 @@ describe("wildcard pricing telemetry e2e", () => { let upstream: OpenAiUpstream | undefined; let embeddingUpstream: OpenAiUpstream | undefined; let countTokensUpstream: OpenAiUpstream | undefined; + let streamingUpstream: OpenAiUpstream | undefined; + let batchUpstream: OpenAiUpstream | undefined; let wildcardID = ""; let embeddingWildcardID = ""; let countTokensWildcardID = ""; + let streamingWildcardID = ""; + let batchWildcardID = ""; let etcdReachable = false; beforeAll(async () => { @@ -83,6 +145,13 @@ describe("wildcard pricing telemetry e2e", () => { upstream = await startOpenAiUpstream({ nonStreamBody: upstreamResponse() }); embeddingUpstream = await startOpenAiUpstream({ nonStreamBody: embeddingUpstreamResponse() }); countTokensUpstream = await startOpenAiUpstream({ nonStreamBody: { input_tokens: 42 } }); + streamingUpstream = await startOpenAiUpstream({ streamEvents: streamEvents() }); + batchUpstream = await startOpenAiUpstream({ + scriptedResponses: [ + { nonStreamBody: completedBatchResponse() }, + { rawBody: completedBatchOutput(), rawContentType: "application/jsonl" }, + ], + }); app = await spawnApp({ extraEnv: { [`SLS_CRED_${CREDENTIAL_REF.toUpperCase()}_AK_ID`]: "mock-akid", @@ -148,6 +217,36 @@ describe("wildcard pricing telemetry e2e", () => { pricing_authority_id: PRICING_AUTHORITY_ID, }); countTokensWildcardID = countTokensWildcard.id; + const streamingProviderKey = await seed.createProviderKey({ + display_name: "wildcard-pricing-stream-pk", + provider: "openai", + adapter: "openai", + secret: "sk-mock", + api_base: `${streamingUpstream.baseUrl}/v1`, + }); + const streamingWildcard = await seed.createModel({ + display_name: STREAMING_WILDCARD_ALIAS, + provider: "openai", + model_name: "*", + provider_key_id: streamingProviderKey.id, + pricing_authority_id: PRICING_AUTHORITY_ID, + }); + streamingWildcardID = streamingWildcard.id; + const batchProviderKey = await seed.createProviderKey({ + display_name: "wildcard-pricing-batch-pk", + provider: "openai", + adapter: "openai", + secret: "sk-mock", + api_base: `${batchUpstream.baseUrl}/v1`, + }); + const batchWildcard = await seed.createModel({ + display_name: BATCH_WILDCARD_ALIAS, + provider: "openai", + model_name: "*", + provider_key_id: batchProviderKey.id, + pricing_authority_id: PRICING_AUTHORITY_ID, + }); + batchWildcardID = batchWildcard.id; // Seeded last: a successful models-list gate proves that all preceding // resources, including the exporter, are in the same gateway snapshot. @@ -166,6 +265,8 @@ describe("wildcard pricing telemetry e2e", () => { await upstream?.close(); await embeddingUpstream?.close(); await countTokensUpstream?.close(); + await streamingUpstream?.close(); + await batchUpstream?.close(); await sls?.close(); }); @@ -222,6 +323,51 @@ describe("wildcard pricing telemetry e2e", () => { expect(unknown.get("resolved_pricing_model")).toBe(UNKNOWN_MODEL); }); + test("stream terminal telemetry keeps the wildcard dispatch price after the response scope ends", async (ctx) => { + if (!etcdReachable || !app || !sls || !streamingUpstream || !streamingWildcardID) { + ctx.skip(); + return; + } + + const before = streamingUpstream.receivedRequests.length; + const res = await fetch(`${app.proxyUrl}/v1/chat/completions`, { + method: "POST", + headers: { + authorization: `Bearer ${CALLER_PLAINTEXT}`, + "content-type": "application/json", + }, + body: JSON.stringify({ + model: STREAMING_REQUEST_MODEL, + stream: true, + messages: [{ role: "user", content: "stream this prompt" }], + }), + }); + const body = await res.text(); + expect(res.status, body).toBe(200); + expect(body).toContain("[DONE]"); + + const calls = streamingUpstream.receivedRequests.slice(before); + expect(calls).toHaveLength(1); + expect(calls[0]!.method).toBe("POST"); + expect(calls[0]!.path).toBe("/v1/chat/completions"); + expect(JSON.parse(calls[0]!.body)).toMatchObject({ + model: STREAMING_UPSTREAM_MODEL, + stream: true, + }); + + const event = await waitForSlsLog( + sls, + LOGSTORE, + (log) => log.get("requested_model") === STREAMING_REQUEST_MODEL, + `stream terminal usage event for ${STREAMING_REQUEST_MODEL}`, + ); + expect(event.get("model_id")).toBe(streamingWildcardID); + expect(event.get("pricing_authority_id")).toBe(PRICING_AUTHORITY_ID); + expect(event.get("resolved_pricing_model")).toBe(STREAMING_UPSTREAM_MODEL); + expect(event.get("prompt_tokens")).toBe("11"); + expect(event.get("completion_tokens")).toBe("6"); + }); + test("direct embedding wildcard dispatch exports its concrete pricing identity", async (ctx) => { if (!etcdReachable || !app || !sls || !embeddingUpstream || !embeddingWildcardID) { ctx.skip(); @@ -371,4 +517,57 @@ describe("wildcard pricing telemetry e2e", () => { expect(event.get("resolved_pricing_model")).toBeUndefined(); } }); + + test("completed batch telemetry retains dispatch pricing while its management event stays unpriced", async (ctx) => { + if (!etcdReachable || !app || !sls || !batchUpstream || !batchWildcardID) { + ctx.skip(); + return; + } + + const before = batchUpstream.receivedRequests.length; + const route = routedId("batch-wildcard-completed", BATCH_REQUEST_MODEL); + const res = await fetch(`${app.proxyUrl}/v1/batches/${route}`, { + headers: { authorization: `Bearer ${CALLER_PLAINTEXT}` }, + }); + const body = await res.text(); + expect(res.status, body).toBe(200); + expect(JSON.parse(body)).toMatchObject({ + output_file_id: routedId("file-wildcard-output", BATCH_REQUEST_MODEL), + }); + + const management = await waitForSlsLog( + sls, + LOGSTORE, + (log) => + log.get("operation") === "batches" && + log.get("model_id") === batchWildcardID && + log.get("requested_model") === BATCH_WILDCARD_ALIAS, + "wildcard batch management usage event", + ); + expect(management.get("prompt_tokens")).toBe("0"); + expect(management.get("completion_tokens")).toBe("0"); + expect(management.get("pricing_authority_id")).toBeUndefined(); + expect(management.get("resolved_pricing_model")).toBeUndefined(); + + const aggregate = await waitForSlsLog( + sls, + LOGSTORE, + (log) => + log.get("inbound_protocol") === "batch" && + log.get("model_id") === batchWildcardID && + log.get("requested_model") === BATCH_WILDCARD_ALIAS, + "completed wildcard batch usage event", + ); + expect(aggregate.get("provider_model_version")).toBe("provider-reported-batch-model"); + expect(aggregate.get("pricing_authority_id")).toBe(PRICING_AUTHORITY_ID); + expect(aggregate.get("resolved_pricing_model")).toBe("gpt-4o-2024-08-06"); + expect(aggregate.get("prompt_tokens")).toBe("13"); + expect(aggregate.get("completion_tokens")).toBe("7"); + expect(aggregate.get("cached_prompt_tokens")).toBe("3"); + + const calls = batchUpstream.receivedRequests.slice(before); + expect(calls).toHaveLength(2); + expect(calls[0]).toMatchObject({ method: "GET", path: "/v1/batches/batch-wildcard-completed" }); + expect(calls[1]).toMatchObject({ method: "GET", path: "/v1/files/file-wildcard-output/content" }); + }); }); From aca3fb63ab18cd33e6180b4da652cee656006438 Mon Sep 17 00:00:00 2001 From: Ming Wen Date: Thu, 1 Oct 2026 15:21:42 +0800 Subject: [PATCH 16/16] test: mark wildcard pricing fixtures as non-emitters --- crates/aisix-proxy/src/usage_attr.rs | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/crates/aisix-proxy/src/usage_attr.rs b/crates/aisix-proxy/src/usage_attr.rs index 4806379a..e6eeada8 100644 --- a/crates/aisix-proxy/src/usage_attr.rs +++ b/crates/aisix-proxy/src/usage_attr.rs @@ -1268,6 +1268,8 @@ mod tests { }; let mut completed_batch = UsageEvent { + // NO-GUARDRAIL-CHAIN: unit-only value used to exercise the + // captured-identity gate, not an emitted gateway event. model_id: "wildcard".to_string(), ..Default::default() }; @@ -1299,6 +1301,8 @@ mod tests { ), ] { let mut event = UsageEvent { + // NO-GUARDRAIL-CHAIN: unit-only value used to exercise the + // captured-identity gate, not an emitted gateway event. model_id: model_id.to_string(), ..Default::default() };