From 03c914799da9ca6710e6235a1373e8aa70de2a1e Mon Sep 17 00:00:00 2001 From: Abhishek Choudhary Date: Wed, 23 Sep 2026 22:33:24 +0545 Subject: [PATCH 1/4] feat(ai-rate-limiting): expose nested usage fields to cost_expr by leaf name --- apisix/plugins/ai-rate-limiting.lua | 38 ++++- docs/en/latest/plugins/ai-rate-limiting.md | 2 +- docs/zh/latest/plugins/ai-rate-limiting.md | 2 +- t/fixtures/openai/responses-usage-clash.json | 23 +++ t/plugin/ai-rate-limiting-expression.t | 158 +++++++++++++++++++ 5 files changed, 216 insertions(+), 7 deletions(-) create mode 100644 t/fixtures/openai/responses-usage-clash.json diff --git a/apisix/plugins/ai-rate-limiting.lua b/apisix/plugins/ai-rate-limiting.lua index 489176a63358..a9144e5534c5 100644 --- a/apisix/plugins/ai-rate-limiting.lua +++ b/apisix/plugins/ai-rate-limiting.lua @@ -19,10 +19,12 @@ local setmetatable = setmetatable local ipairs = ipairs local type = type local pairs = pairs +local rawget = rawget local pcall = pcall local load = load local math_floor = math.floor local math_huge = math.huge +local table_sort = table.sort local core = require("apisix.core") local limit_count = require("apisix.plugins.limit-count.init") local policy_to_additional_properties = limit_count.policy_to_additional_properties @@ -350,6 +352,36 @@ function _M.check_instance_status(conf, ctx, instance_name) end +-- expose usage leaves by bare name, breadth first so shallower keys win +local function inject_usage_vars(env, raw) + local level = {raw} + while #level > 0 do + local next_level = {} + for _, tab in ipairs(level) do + local keys = {} + for k in pairs(tab) do + if type(k) == "string" then + keys[#keys + 1] = k + end + end + -- sorted for a stable winner on same-depth clashes + table_sort(keys) + for _, k in ipairs(keys) do + local v = tab[k] + if type(v) == "number" then + if rawget(env, k) == nil and not expr_safe_env[k] then + env[k] = v + end + elseif type(v) == "table" then + next_level[#next_level + 1] = v + end + end + end + level = next_level + end +end + + local function eval_cost_expr(conf_cost_expr, raw) local fn_code = "return " .. conf_cost_expr -- build environment: safe math + usage variables (missing vars default to 0) @@ -362,11 +394,7 @@ local function eval_cost_expr(conf_cost_expr, raw) return 0 end }) - for k, v in pairs(raw) do - if type(v) == "number" and not expr_safe_env[k] then - env[k] = v - end - end + inject_usage_vars(env, raw) local fn, err = load(fn_code, "cost_expr", "t", env) if not fn then return nil, "failed to compile cost_expr: " .. err diff --git a/docs/en/latest/plugins/ai-rate-limiting.md b/docs/en/latest/plugins/ai-rate-limiting.md index 4d8757fae9c4..56d61c91307b 100644 --- a/docs/en/latest/plugins/ai-rate-limiting.md +++ b/docs/en/latest/plugins/ai-rate-limiting.md @@ -48,7 +48,7 @@ The `ai-rate-limiting` Plugin enforces token-based rate limiting for requests se | time_window | integer | False | | >0 | The time interval corresponding to the rate limiting `limit` in seconds. At least one of `time_window` and `instances.time_window` should be configured. Required if `rules` is not configured. | | show_limit_quota_header | boolean | False | true | | If true, includes rate limiting response headers. When `rules` is not set, the headers are `X-AI-RateLimit-Limit-*`, `X-AI-RateLimit-Remaining-*`, and `X-AI-RateLimit-Reset-*`, where `*` is the instance name. When `rules` is set, see `rules.header_prefix` for details. | | limit_strategy | string | False | total_tokens | [`total_tokens`, `prompt_tokens`, `completion_tokens`, `expression`] | Type of token to apply rate limiting. `total_tokens` is the sum of `prompt_tokens` and `completion_tokens`. When set to `expression`, the `cost_expr` field is used to dynamically calculate token cost. | -| cost_expr | string | False | | | Lua arithmetic expression for dynamic token cost calculation. Variables are injected from the LLM API raw usage response fields. Missing variables default to 0. Only valid when `limit_strategy` is `expression`. Example: `input_tokens + cache_creation_input_tokens + output_tokens`. | +| cost_expr | string | False | | | Lua arithmetic expression for dynamic token cost calculation. Variables are injected from the LLM API raw usage response fields. Nested fields are exposed by their leaf name, so `input_tokens_details.cached_tokens` is available as `cached_tokens`. On a name clash, the shallower field wins. Missing variables default to 0. Only valid when `limit_strategy` is `expression`. Example: `input_tokens + cache_creation_input_tokens + output_tokens`. | | instances | array[object] | False | | | LLM instance rate limiting configurations. | | instances.name | string | True | | | Name of the LLM service instance. | | instances.limit | integer | True | | >0 | The maximum number of tokens allowed within a given time interval for an instance. | diff --git a/docs/zh/latest/plugins/ai-rate-limiting.md b/docs/zh/latest/plugins/ai-rate-limiting.md index 7cf41a44486e..7035321dbd1e 100644 --- a/docs/zh/latest/plugins/ai-rate-limiting.md +++ b/docs/zh/latest/plugins/ai-rate-limiting.md @@ -48,7 +48,7 @@ import TabItem from '@theme/TabItem'; | time_window | integer | False | | >0 | 与速率限制 `limit` 对应的时间间隔(秒)。`time_window` 和 `instances.time_window` 中至少应配置一个。如果未配置 `rules`,则为必填项。 | | show_limit_quota_header | boolean | False | true | | 如果为 true,则在响应中包含速率限制头部。当未设置 `rules` 时,头部为 `X-AI-RateLimit-Limit-*`、`X-AI-RateLimit-Remaining-*` 和 `X-AI-RateLimit-Reset-*`,其中 `*` 是实例名称。当设置了 `rules` 时,详见 `rules.header_prefix`。 | | limit_strategy | string | False | total_tokens | [`total_tokens`, `prompt_tokens`, `completion_tokens`, `expression`] | 应用速率限制的令牌类型。`total_tokens` 是 `prompt_tokens` 和 `completion_tokens` 的总和。当设置为 `expression` 时,使用 `cost_expr` 字段动态计算令牌消耗。 | -| cost_expr | string | False | | | 用于动态计算令牌消耗的 Lua 算术表达式。变量从 LLM API 原始使用量响应字段注入。缺失的变量默认为 0。仅在 `limit_strategy` 为 `expression` 时有效。示例:`input_tokens + cache_creation_input_tokens + output_tokens`。 | +| cost_expr | string | False | | | 用于动态计算令牌消耗的 Lua 算术表达式。变量从 LLM API 原始使用量响应字段注入。嵌套字段以叶子字段名暴露,例如 `input_tokens_details.cached_tokens` 可直接用 `cached_tokens` 引用。字段名冲突时,层级较浅的字段优先。缺失的变量默认为 0。仅在 `limit_strategy` 为 `expression` 时有效。示例:`input_tokens + cache_creation_input_tokens + output_tokens`。 | | instances | array[object] | False | | | LLM 实例速率限制配置。 | | instances.name | string | True | | | LLM 服务实例的名称。 | | instances.limit | integer | True | | >0 | 实例在给定时间间隔内允许的最大令牌数。 | diff --git a/t/fixtures/openai/responses-usage-clash.json b/t/fixtures/openai/responses-usage-clash.json new file mode 100644 index 000000000000..6883abcad7e4 --- /dev/null +++ b/t/fixtures/openai/responses-usage-clash.json @@ -0,0 +1,23 @@ +{ + "id": "resp_clash1", + "object": "response", + "created_at": 1723780938, + "model": "{{model}}", + "output": [ + { + "type": "message", + "role": "assistant", + "content": [ + { "type": "output_text", "text": "Hello" } + ] + } + ], + "usage": { + "input_tokens": 40, + "output_tokens": 20, + "total_tokens": 60, + "cached_tokens": 5, + "input_tokens_details": { "cached_tokens": 12 }, + "output_tokens_details": { "reasoning_tokens": 8 } + } +} diff --git a/t/plugin/ai-rate-limiting-expression.t b/t/plugin/ai-rate-limiting-expression.t index dd69aa548f66..bdd931d25ce0 100644 --- a/t/plugin/ai-rate-limiting-expression.t +++ b/t/plugin/ai-rate-limiting-expression.t @@ -519,3 +519,161 @@ X-AI-Fixture: anthropic/messages-with-cache.json ] --- no_error_log [error] + + + +=== TEST 14: set route with expression reading nested usage leaves (OpenAI Responses) +--- config + location /t { + content_by_lua_block { + local t = require("lib.test_admin").test + local code, body = t('/apisix/admin/routes/1', + ngx.HTTP_PUT, + [[{ + "uri": "/v1/responses", + "plugins": { + "ai-proxy": { + "provider": "openai", + "auth": { + "header": { + "Authorization": "Bearer test-key" + } + }, + "options": { + "model": "gpt-4o-mini" + }, + "override": { + "endpoint": "http://127.0.0.1:1980" + }, + "ssl_verify": false + }, + "ai-rate-limiting": { + "limit": 500, + "time_window": 60, + "limit_strategy": "expression", + "cost_expr": "input_tokens - cached_tokens + output_tokens + reasoning_tokens" + } + }, + "upstream": { + "type": "roundrobin", + "nodes": { + "canbeanything.com": 1 + } + } + }]] + ) + + if code >= 300 then + ngx.status = code + end + ngx.say(body) + } + } +--- response_body +passed + + + +=== TEST 15: nested leaves - cost = 40 - 12 + 20 + 8 = 56 per request +--- pipelined_requests eval +[ + "POST /v1/responses\n" . '{"model":"gpt-4o-mini","stream":false,"input":"Hello"}', + "POST /v1/responses\n" . '{"model":"gpt-4o-mini","stream":false,"input":"Hello"}', +] +--- more_headers +X-AI-Fixture: openai/responses-with-cache.json +--- response_headers_like eval +[ + "X-AI-RateLimit-Remaining-ai-proxy-openai: 500", + "X-AI-RateLimit-Remaining-ai-proxy-openai: 444", +] +--- no_error_log +[error] + + + +=== TEST 16: nested leaves, streaming - cost = 20 - 10 + 5 + 3 = 18 per request +--- pipelined_requests eval +[ + "POST /v1/responses\n" . '{"model":"gpt-4o-mini","stream":true,"input":"Hello"}', + "POST /v1/responses\n" . '{"model":"gpt-4o-mini","stream":true,"input":"Hello"}', +] +--- more_headers +X-AI-Fixture: openai/responses-streaming-with-cache.sse +--- response_headers_like eval +[ + "X-AI-RateLimit-Remaining-ai-proxy-openai: 500", + "X-AI-RateLimit-Remaining-ai-proxy-openai: 482", +] +--- no_error_log +[error] + + + +=== TEST 17: set route with expression reading a leaf name present at two depths +--- config + location /t { + content_by_lua_block { + local t = require("lib.test_admin").test + local code, body = t('/apisix/admin/routes/1', + ngx.HTTP_PUT, + [[{ + "uri": "/v1/responses", + "plugins": { + "ai-proxy": { + "provider": "openai", + "auth": { + "header": { + "Authorization": "Bearer test-key" + } + }, + "options": { + "model": "gpt-4o-mini" + }, + "override": { + "endpoint": "http://127.0.0.1:1980" + }, + "ssl_verify": false + }, + "ai-rate-limiting": { + "limit": 500, + "time_window": 60, + "limit_strategy": "expression", + "cost_expr": "cached_tokens" + } + }, + "upstream": { + "type": "roundrobin", + "nodes": { + "canbeanything.com": 1 + } + } + }]] + ) + + if code >= 300 then + ngx.status = code + end + ngx.say(body) + } + } +--- response_body +passed + + + +=== TEST 18: top-level field wins over nested one - cost = 5 per request +--- pipelined_requests eval +[ + "POST /v1/responses\n" . '{"model":"gpt-4o-mini","stream":false,"input":"Hello"}', + "POST /v1/responses\n" . '{"model":"gpt-4o-mini","stream":false,"input":"Hello"}', +] +--- more_headers +X-AI-Fixture: openai/responses-usage-clash.json +--- response_headers_like eval +[ + "X-AI-RateLimit-Remaining-ai-proxy-openai: 500", + "X-AI-RateLimit-Remaining-ai-proxy-openai: 495", +] +--- no_error_log +[error] From fb1e1e8fa1c12e6e89f583ad9491eecea4cfef78 Mon Sep 17 00:00:00 2001 From: Abhishek Choudhary Date: Thu, 24 Sep 2026 13:06:29 +0545 Subject: [PATCH 2/4] feat(ai-rate-limiting): support explicit usage paths in cost_expr and precompile it --- apisix/plugins/ai-rate-limiting.lua | 204 ++++++--- docs/en/latest/plugins/ai-rate-limiting.md | 2 +- docs/zh/latest/plugins/ai-rate-limiting.md | 2 +- t/fixtures/openai/chat-usage-audio.json | 19 + t/fixtures/openai/chat-usage-deep.json | 22 + t/plugin/ai-rate-limiting-expression.t | 459 +++++++++++++++++++++ 6 files changed, 655 insertions(+), 53 deletions(-) create mode 100644 t/fixtures/openai/chat-usage-audio.json create mode 100644 t/fixtures/openai/chat-usage-deep.json diff --git a/apisix/plugins/ai-rate-limiting.lua b/apisix/plugins/ai-rate-limiting.lua index a9144e5534c5..d9f00ed7bdb8 100644 --- a/apisix/plugins/ai-rate-limiting.lua +++ b/apisix/plugins/ai-rate-limiting.lua @@ -20,11 +20,17 @@ local ipairs = ipairs local type = type local pairs = pairs local rawget = rawget +local rawset = rawset local pcall = pcall local load = load local math_floor = math.floor local math_huge = math.huge local table_sort = table.sort +local table_concat = table.concat +local str_find = string.find +local str_sub = string.sub +local str_gsub = string.gsub +local str_gmatch = string.gmatch local core = require("apisix.core") local limit_count = require("apisix.plugins.limit-count.init") local policy_to_additional_properties = limit_count.policy_to_additional_properties @@ -183,6 +189,10 @@ local limit_conf_cache = core.lrucache.new({ ttl = 300, count = 512 }) +local cost_expr_cache = core.lrucache.new({ + ttl = 300, count = 512 +}) + -- safe math functions allowed in cost expressions local expr_safe_env = { @@ -194,14 +204,147 @@ local expr_safe_env = { min = math.min, } +local lua_keywords = { + ["and"] = true, ["break"] = true, ["do"] = true, ["else"] = true, + ["elseif"] = true, ["end"] = true, ["false"] = true, ["for"] = true, + ["function"] = true, ["goto"] = true, ["if"] = true, ["in"] = true, + ["local"] = true, ["nil"] = true, ["not"] = true, ["or"] = true, + ["repeat"] = true, ["return"] = true, ["then"] = true, ["true"] = true, + ["until"] = true, ["while"] = true, +} + +-- private keys, unreachable from the expression +local RAW_KEY = {} +local PATHS_KEY = {} + + +local function walk_path(raw, segs) + local v = raw + for i = 1, #segs do + if type(v) ~= "table" then + return 0 + end + v = v[segs[i]] + end + if type(v) == "number" then + return v + end + return 0 +end + + +-- search level by level; a clash on one level charges the larger value +local function find_leaf(raw, name) + local v = raw[name] + if type(v) == "number" then + return v + end + + local level, prefixes = {raw}, {""} + while true do + local best, hits = nil, 0 + for i = 1, #level do + for k, tab in pairs(level[i]) do + if type(k) == "string" and type(tab) == "table" then + local x = tab[name] + if type(x) == "number" then + hits = hits + 1 + if not best or x > best then + best = x + end + end + end + end + end + + if best then + if hits > 1 then + local paths = {} + for i = 1, #level do + for k, tab in pairs(level[i]) do + if type(k) == "string" and type(tab) == "table" + and type(tab[name]) == "number" then + paths[#paths + 1] = prefixes[i] .. k .. "." .. name + end + end + end + table_sort(paths) + core.log.error("ambiguous usage field '", name, "' in cost_expr matches ", + table_concat(paths, ", "), ", charging the larger value, ", + "use an explicit path instead") + end + return best + end + + local next_level, next_prefixes = {}, {} + for i = 1, #level do + for k, tab in pairs(level[i]) do + if type(k) == "string" and type(tab) == "table" then + next_level[#next_level + 1] = tab + next_prefixes[#next_prefixes + 1] = prefixes[i] .. k .. "." + end + end + end + if #next_level == 0 then + return 0 + end + level, prefixes = next_level, next_prefixes + end +end + + +local usage_mt = { + __index = function(t, k) + local segs = rawget(t, PATHS_KEY)[k] + local v + if segs then + v = walk_path(rawget(t, RAW_KEY), segs) + else + v = find_leaf(rawget(t, RAW_KEY), k) + end + rawset(t, k, v) + return v + end +} + + +-- rewrite usage names to reads on the `usage` argument local function compile_cost_expr(expr_str) - local fn_code = "return " .. expr_str - -- validate syntax by loading first - local fn, err = load(fn_code, "cost_expr", "t", expr_safe_env) + local paths = {} + local bad_name + local code = str_gsub(expr_str, "()([%a_][%w_%.]*)", function(pos, name) + -- skip number literals like 1e5 or 0x1F + if pos > 1 and str_find(str_sub(expr_str, pos - 1, pos - 1), "[%w_%.]") then + return nil + end + local head = str_gsub(name, "%..*", "") + if lua_keywords[name] or expr_safe_env[head] then + return nil + end + if str_find(name, ".", 1, true) then + if str_find(name, "..", 1, true) or str_sub(name, -1) == "." then + bad_name = bad_name or name + return nil + end + local segs = {} + for seg in str_gmatch(name, "[^%.]+") do + segs[#segs + 1] = seg + end + paths[name] = segs + end + return 'usage["' .. name .. '"]' + end) + if bad_name then + return nil, "invalid field reference: " .. bad_name + end + + -- own env per expression so writes stay local to it + local env = setmetatable({}, {__index = expr_safe_env}) + local fn, err = load("local usage = ...\nreturn " .. code, "cost_expr", "t", env) if not fn then return nil, err end - return fn_code + return {fn = fn, paths = paths} end @@ -352,54 +495,13 @@ function _M.check_instance_status(conf, ctx, instance_name) end --- expose usage leaves by bare name, breadth first so shallower keys win -local function inject_usage_vars(env, raw) - local level = {raw} - while #level > 0 do - local next_level = {} - for _, tab in ipairs(level) do - local keys = {} - for k in pairs(tab) do - if type(k) == "string" then - keys[#keys + 1] = k - end - end - -- sorted for a stable winner on same-depth clashes - table_sort(keys) - for _, k in ipairs(keys) do - local v = tab[k] - if type(v) == "number" then - if rawget(env, k) == nil and not expr_safe_env[k] then - env[k] = v - end - elseif type(v) == "table" then - next_level[#next_level + 1] = v - end - end - end - level = next_level - end -end - - -local function eval_cost_expr(conf_cost_expr, raw) - local fn_code = "return " .. conf_cost_expr - -- build environment: safe math + usage variables (missing vars default to 0) - local env = setmetatable({}, { - __index = function(_, k) - local v = expr_safe_env[k] - if v ~= nil then - return v - end - return 0 - end - }) - inject_usage_vars(env, raw) - local fn, err = load(fn_code, "cost_expr", "t", env) - if not fn then +local function eval_cost_expr(conf, raw) + local compiled, err = cost_expr_cache(conf, nil, compile_cost_expr, conf.cost_expr) + if not compiled then return nil, "failed to compile cost_expr: " .. err end - local ok, result = pcall(fn) + local usage = setmetatable({[RAW_KEY] = raw, [PATHS_KEY] = compiled.paths}, usage_mt) + local ok, result = pcall(compiled.fn, usage) if not ok then return nil, "failed to evaluate cost_expr: " .. result end @@ -421,7 +523,7 @@ local function get_token_usage(conf, ctx) if not raw then return end - local result, err = eval_cost_expr(conf.cost_expr, raw) + local result, err = eval_cost_expr(conf, raw) if not result then core.log.error(err) return diff --git a/docs/en/latest/plugins/ai-rate-limiting.md b/docs/en/latest/plugins/ai-rate-limiting.md index 56d61c91307b..b2300d8b449b 100644 --- a/docs/en/latest/plugins/ai-rate-limiting.md +++ b/docs/en/latest/plugins/ai-rate-limiting.md @@ -48,7 +48,7 @@ The `ai-rate-limiting` Plugin enforces token-based rate limiting for requests se | time_window | integer | False | | >0 | The time interval corresponding to the rate limiting `limit` in seconds. At least one of `time_window` and `instances.time_window` should be configured. Required if `rules` is not configured. | | show_limit_quota_header | boolean | False | true | | If true, includes rate limiting response headers. When `rules` is not set, the headers are `X-AI-RateLimit-Limit-*`, `X-AI-RateLimit-Remaining-*`, and `X-AI-RateLimit-Reset-*`, where `*` is the instance name. When `rules` is set, see `rules.header_prefix` for details. | | limit_strategy | string | False | total_tokens | [`total_tokens`, `prompt_tokens`, `completion_tokens`, `expression`] | Type of token to apply rate limiting. `total_tokens` is the sum of `prompt_tokens` and `completion_tokens`. When set to `expression`, the `cost_expr` field is used to dynamically calculate token cost. | -| cost_expr | string | False | | | Lua arithmetic expression for dynamic token cost calculation. Variables are injected from the LLM API raw usage response fields. Nested fields are exposed by their leaf name, so `input_tokens_details.cached_tokens` is available as `cached_tokens`. On a name clash, the shallower field wins. Missing variables default to 0. Only valid when `limit_strategy` is `expression`. Example: `input_tokens + cache_creation_input_tokens + output_tokens`. | +| cost_expr | string | False | | | Lua arithmetic expression for dynamic token cost calculation. Variables are injected from the LLM API raw usage response fields. Nested fields can be referenced by their leaf name, e.g. `cached_tokens` for `input_tokens_details.cached_tokens`, or by an explicit path such as `input_tokens_details.cached_tokens`. A leaf name resolves to the shallowest match. If it matches several fields at the same depth, such as `audio_tokens` in both `prompt_tokens_details` and `completion_tokens_details`, the larger value is used and an error is logged; use an explicit path to pick one. Missing variables default to 0. Only valid when `limit_strategy` is `expression`. Example: `input_tokens + cache_creation_input_tokens + output_tokens`. | | instances | array[object] | False | | | LLM instance rate limiting configurations. | | instances.name | string | True | | | Name of the LLM service instance. | | instances.limit | integer | True | | >0 | The maximum number of tokens allowed within a given time interval for an instance. | diff --git a/docs/zh/latest/plugins/ai-rate-limiting.md b/docs/zh/latest/plugins/ai-rate-limiting.md index 7035321dbd1e..0644507c3170 100644 --- a/docs/zh/latest/plugins/ai-rate-limiting.md +++ b/docs/zh/latest/plugins/ai-rate-limiting.md @@ -48,7 +48,7 @@ import TabItem from '@theme/TabItem'; | time_window | integer | False | | >0 | 与速率限制 `limit` 对应的时间间隔(秒)。`time_window` 和 `instances.time_window` 中至少应配置一个。如果未配置 `rules`,则为必填项。 | | show_limit_quota_header | boolean | False | true | | 如果为 true,则在响应中包含速率限制头部。当未设置 `rules` 时,头部为 `X-AI-RateLimit-Limit-*`、`X-AI-RateLimit-Remaining-*` 和 `X-AI-RateLimit-Reset-*`,其中 `*` 是实例名称。当设置了 `rules` 时,详见 `rules.header_prefix`。 | | limit_strategy | string | False | total_tokens | [`total_tokens`, `prompt_tokens`, `completion_tokens`, `expression`] | 应用速率限制的令牌类型。`total_tokens` 是 `prompt_tokens` 和 `completion_tokens` 的总和。当设置为 `expression` 时,使用 `cost_expr` 字段动态计算令牌消耗。 | -| cost_expr | string | False | | | 用于动态计算令牌消耗的 Lua 算术表达式。变量从 LLM API 原始使用量响应字段注入。嵌套字段以叶子字段名暴露,例如 `input_tokens_details.cached_tokens` 可直接用 `cached_tokens` 引用。字段名冲突时,层级较浅的字段优先。缺失的变量默认为 0。仅在 `limit_strategy` 为 `expression` 时有效。示例:`input_tokens + cache_creation_input_tokens + output_tokens`。 | +| cost_expr | string | False | | | 用于动态计算令牌消耗的 Lua 算术表达式。变量从 LLM API 原始使用量响应字段注入。嵌套字段可以用叶子字段名引用,例如用 `cached_tokens` 引用 `input_tokens_details.cached_tokens`,也可以用显式路径引用,例如 `input_tokens_details.cached_tokens`。叶子字段名解析为层级最浅的匹配字段。若同一层级有多个同名字段,例如 `prompt_tokens_details` 和 `completion_tokens_details` 中都有 `audio_tokens`,则取较大值并记录错误日志;请使用显式路径指定字段。缺失的变量默认为 0。仅在 `limit_strategy` 为 `expression` 时有效。示例:`input_tokens + cache_creation_input_tokens + output_tokens`。 | | instances | array[object] | False | | | LLM 实例速率限制配置。 | | instances.name | string | True | | | LLM 服务实例的名称。 | | instances.limit | integer | True | | >0 | 实例在给定时间间隔内允许的最大令牌数。 | diff --git a/t/fixtures/openai/chat-usage-audio.json b/t/fixtures/openai/chat-usage-audio.json new file mode 100644 index 000000000000..5cdc7c6d4c15 --- /dev/null +++ b/t/fixtures/openai/chat-usage-audio.json @@ -0,0 +1,19 @@ +{ + "id": "chatcmpl-audio1", + "object": "chat.completion", + "model": "{{model}}", + "choices": [ + { + "index": 0, + "message": { "role": "assistant", "content": "Hello" }, + "finish_reason": "stop" + } + ], + "usage": { + "prompt_tokens": 100, + "completion_tokens": 50, + "total_tokens": 150, + "prompt_tokens_details": { "cached_tokens": 20, "audio_tokens": 30 }, + "completion_tokens_details": { "reasoning_tokens": 10, "audio_tokens": 70 } + } +} diff --git a/t/fixtures/openai/chat-usage-deep.json b/t/fixtures/openai/chat-usage-deep.json new file mode 100644 index 000000000000..e0da1fafc048 --- /dev/null +++ b/t/fixtures/openai/chat-usage-deep.json @@ -0,0 +1,22 @@ +{ + "id": "chatcmpl-deep1", + "object": "chat.completion", + "model": "{{model}}", + "choices": [ + { + "index": 0, + "message": { "role": "assistant", "content": "Hello" }, + "finish_reason": "stop" + } + ], + "usage": { + "prompt_tokens": 100, + "completion_tokens": 50, + "total_tokens": 150, + "prompt_tokens_details": { + "cached_tokens": 20, + "cached_tokens_details": { "text_tokens": 9 } + }, + "modality_details": [ { "text_tokens": 100 } ] + } +} diff --git a/t/plugin/ai-rate-limiting-expression.t b/t/plugin/ai-rate-limiting-expression.t index bdd931d25ce0..aa8de06d8e2a 100644 --- a/t/plugin/ai-rate-limiting-expression.t +++ b/t/plugin/ai-rate-limiting-expression.t @@ -677,3 +677,462 @@ X-AI-Fixture: openai/responses-usage-clash.json ] --- no_error_log [error] + + + +=== TEST 19: set route with an explicit path to a field that also exists at the top level +--- config + location /t { + content_by_lua_block { + local t = require("lib.test_admin").test + local code, body = t('/apisix/admin/routes/1', + ngx.HTTP_PUT, + [[{ + "uri": "/v1/responses", + "plugins": { + "ai-proxy": { + "provider": "openai", + "auth": { + "header": { + "Authorization": "Bearer test-key" + } + }, + "options": { + "model": "gpt-4o-mini" + }, + "override": { + "endpoint": "http://127.0.0.1:1980" + }, + "ssl_verify": false + }, + "ai-rate-limiting": { + "limit": 500, + "time_window": 60, + "limit_strategy": "expression", + "cost_expr": "input_tokens_details.cached_tokens" + } + }, + "upstream": { + "type": "roundrobin", + "nodes": { + "canbeanything.com": 1 + } + } + }]] + ) + + if code >= 300 then + ngx.status = code + end + ngx.say(body) + } + } +--- response_body +passed + + + +=== TEST 20: explicit path reads the nested field - cost = 12 per request +--- pipelined_requests eval +[ + "POST /v1/responses\n" . '{"model":"gpt-4o-mini","stream":false,"input":"Hello"}', + "POST /v1/responses\n" . '{"model":"gpt-4o-mini","stream":false,"input":"Hello"}', +] +--- more_headers +X-AI-Fixture: openai/responses-usage-clash.json +--- response_headers_like eval +[ + "X-AI-RateLimit-Remaining-ai-proxy-openai: 500", + "X-AI-RateLimit-Remaining-ai-proxy-openai: 488", +] +--- no_error_log +[error] + + + +=== TEST 21: set route with a bare name present in two sibling objects +--- config + location /t { + content_by_lua_block { + local t = require("lib.test_admin").test + local code, body = t('/apisix/admin/routes/1', + ngx.HTTP_PUT, + [[{ + "uri": "/v1/chat/completions", + "plugins": { + "ai-proxy": { + "provider": "openai", + "auth": { + "header": { + "Authorization": "Bearer test-key" + } + }, + "options": { + "model": "gpt-4o-mini" + }, + "override": { + "endpoint": "http://127.0.0.1:1980" + }, + "ssl_verify": false + }, + "ai-rate-limiting": { + "limit": 500, + "time_window": 60, + "limit_strategy": "expression", + "cost_expr": "audio_tokens" + } + }, + "upstream": { + "type": "roundrobin", + "nodes": { + "canbeanything.com": 1 + } + } + }]] + ) + + if code >= 300 then + ngx.status = code + end + ngx.say(body) + } + } +--- response_body +passed + + + +=== TEST 22: ambiguous bare name charges the larger value - cost = 70 per request +--- pipelined_requests eval +[ + "POST /v1/chat/completions\n" . '{"model":"gpt-4o-mini","messages":[{"role":"user","content":"Hello"}]}', + "POST /v1/chat/completions\n" . '{"model":"gpt-4o-mini","messages":[{"role":"user","content":"Hello"}]}', +] +--- more_headers +X-AI-Fixture: openai/chat-usage-audio.json +--- response_headers_like eval +[ + "X-AI-RateLimit-Remaining-ai-proxy-openai: 500", + "X-AI-RateLimit-Remaining-ai-proxy-openai: 430", +] +--- no_error_log +[alert] + + + +=== TEST 23: ambiguous bare name logs the explicit paths to use +--- request +POST /v1/chat/completions +{"model":"gpt-4o-mini","messages":[{"role":"user","content":"Hello"}]} +--- more_headers +X-AI-Fixture: openai/chat-usage-audio.json +--- error_log +ambiguous usage field 'audio_tokens' in cost_expr matches completion_tokens_details.audio_tokens, prompt_tokens_details.audio_tokens, charging the larger value, use an explicit path instead + + + +=== TEST 24: set route mixing explicit paths and bare names +--- config + location /t { + content_by_lua_block { + local t = require("lib.test_admin").test + local code, body = t('/apisix/admin/routes/1', + ngx.HTTP_PUT, + [[{ + "uri": "/v1/chat/completions", + "plugins": { + "ai-proxy": { + "provider": "openai", + "auth": { + "header": { + "Authorization": "Bearer test-key" + } + }, + "options": { + "model": "gpt-4o-mini" + }, + "override": { + "endpoint": "http://127.0.0.1:1980" + }, + "ssl_verify": false + }, + "ai-rate-limiting": { + "limit": 500, + "time_window": 60, + "limit_strategy": "expression", + "cost_expr": "prompt_tokens_details.audio_tokens * 2 + completion_tokens_details.audio_tokens + cached_tokens" + } + }, + "upstream": { + "type": "roundrobin", + "nodes": { + "canbeanything.com": 1 + } + } + }]] + ) + + if code >= 300 then + ngx.status = code + end + ngx.say(body) + } + } +--- response_body +passed + + + +=== TEST 25: explicit paths resolve the clash - cost = 30 * 2 + 70 + 20 = 150 per request +--- pipelined_requests eval +[ + "POST /v1/chat/completions\n" . '{"model":"gpt-4o-mini","messages":[{"role":"user","content":"Hello"}]}', + "POST /v1/chat/completions\n" . '{"model":"gpt-4o-mini","messages":[{"role":"user","content":"Hello"}]}', +] +--- more_headers +X-AI-Fixture: openai/chat-usage-audio.json +--- response_headers_like eval +[ + "X-AI-RateLimit-Remaining-ai-proxy-openai: 500", + "X-AI-RateLimit-Remaining-ai-proxy-openai: 350", +] +--- no_error_log +[error] + + + +=== TEST 26: set route with explicit paths that do not exist +--- config + location /t { + content_by_lua_block { + local t = require("lib.test_admin").test + local code, body = t('/apisix/admin/routes/1', + ngx.HTTP_PUT, + [[{ + "uri": "/v1/chat/completions", + "plugins": { + "ai-proxy": { + "provider": "openai", + "auth": { + "header": { + "Authorization": "Bearer test-key" + } + }, + "options": { + "model": "gpt-4o-mini" + }, + "override": { + "endpoint": "http://127.0.0.1:1980" + }, + "ssl_verify": false + }, + "ai-rate-limiting": { + "limit": 500, + "time_window": 60, + "limit_strategy": "expression", + "cost_expr": "prompt_tokens + no_such_details.audio_tokens + prompt_tokens_details.no_such_field" + } + }, + "upstream": { + "type": "roundrobin", + "nodes": { + "canbeanything.com": 1 + } + } + }]] + ) + + if code >= 300 then + ngx.status = code + end + ngx.say(body) + } + } +--- response_body +passed + + + +=== TEST 27: missing explicit paths default to 0 - cost = 100 per request +--- pipelined_requests eval +[ + "POST /v1/chat/completions\n" . '{"model":"gpt-4o-mini","messages":[{"role":"user","content":"Hello"}]}', + "POST /v1/chat/completions\n" . '{"model":"gpt-4o-mini","messages":[{"role":"user","content":"Hello"}]}', +] +--- more_headers +X-AI-Fixture: openai/chat-usage-audio.json +--- response_headers_like eval +[ + "X-AI-RateLimit-Remaining-ai-proxy-openai: 500", + "X-AI-RateLimit-Remaining-ai-proxy-openai: 400", +] +--- no_error_log +[error] + + + +=== TEST 28: set route reading a field nested two levels deep +--- config + location /t { + content_by_lua_block { + local t = require("lib.test_admin").test + local code, body = t('/apisix/admin/routes/1', + ngx.HTTP_PUT, + [[{ + "uri": "/v1/chat/completions", + "plugins": { + "ai-proxy": { + "provider": "openai", + "auth": { + "header": { + "Authorization": "Bearer test-key" + } + }, + "options": { + "model": "gpt-4o-mini" + }, + "override": { + "endpoint": "http://127.0.0.1:1980" + }, + "ssl_verify": false + }, + "ai-rate-limiting": { + "limit": 500, + "time_window": 60, + "limit_strategy": "expression", + "cost_expr": "text_tokens + prompt_tokens_details.cached_tokens_details.text_tokens" + } + }, + "upstream": { + "type": "roundrobin", + "nodes": { + "canbeanything.com": 1 + } + } + }]] + ) + + if code >= 300 then + ngx.status = code + end + ngx.say(body) + } + } +--- response_body +passed + + + +=== TEST 29: deep bare name and deep path both resolve, arrays are skipped - cost = 9 + 9 = 18 per request +--- pipelined_requests eval +[ + "POST /v1/chat/completions\n" . '{"model":"gpt-4o-mini","messages":[{"role":"user","content":"Hello"}]}', + "POST /v1/chat/completions\n" . '{"model":"gpt-4o-mini","messages":[{"role":"user","content":"Hello"}]}', +] +--- more_headers +X-AI-Fixture: openai/chat-usage-deep.json +--- response_headers_like eval +[ + "X-AI-RateLimit-Remaining-ai-proxy-openai: 500", + "X-AI-RateLimit-Remaining-ai-proxy-openai: 482", +] +--- no_error_log +[error] + + + +=== TEST 30: set route using math helpers and number literals around explicit paths +--- config + location /t { + content_by_lua_block { + local t = require("lib.test_admin").test + local code, body = t('/apisix/admin/routes/1', + ngx.HTTP_PUT, + [[{ + "uri": "/v1/chat/completions", + "plugins": { + "ai-proxy": { + "provider": "openai", + "auth": { + "header": { + "Authorization": "Bearer test-key" + } + }, + "options": { + "model": "gpt-4o-mini" + }, + "override": { + "endpoint": "http://127.0.0.1:1980" + }, + "ssl_verify": false + }, + "ai-rate-limiting": { + "limit": 500, + "time_window": 60, + "limit_strategy": "expression", + "cost_expr": "floor(completion_tokens_details.audio_tokens / 1e1) + math.max(reasoning_tokens, 1)" + } + }, + "upstream": { + "type": "roundrobin", + "nodes": { + "canbeanything.com": 1 + } + } + }]] + ) + + if code >= 300 then + ngx.status = code + end + ngx.say(body) + } + } +--- response_body +passed + + + +=== TEST 31: helpers and literals are left as is - cost = 7 + 10 = 17 per request +--- pipelined_requests eval +[ + "POST /v1/chat/completions\n" . '{"model":"gpt-4o-mini","messages":[{"role":"user","content":"Hello"}]}', + "POST /v1/chat/completions\n" . '{"model":"gpt-4o-mini","messages":[{"role":"user","content":"Hello"}]}', +] +--- more_headers +X-AI-Fixture: openai/chat-usage-audio.json +--- response_headers_like eval +[ + "X-AI-RateLimit-Remaining-ai-proxy-openai: 500", + "X-AI-RateLimit-Remaining-ai-proxy-openai: 483", +] +--- no_error_log +[error] + + + +=== TEST 32: schema validation - malformed explicit paths are rejected +--- config + location /t { + content_by_lua_block { + local plugin = require("apisix.plugins.ai-rate-limiting") + local exprs = { + "input_tokens_details..cached_tokens", + "input_tokens_details. + 1", + "math.floor(input_tokens_details.cached_tokens) + 1e3", + } + for i, expr in ipairs(exprs) do + local ok, err = plugin.check_schema({ + limit = 100, + time_window = 60, + limit_strategy = "expression", + cost_expr = expr, + }) + ngx.say("expr ", i, ": ", ok and "valid" or err) + end + } + } +--- response_body +expr 1: invalid cost_expr: invalid field reference: input_tokens_details..cached_tokens +expr 2: invalid cost_expr: invalid field reference: input_tokens_details. +expr 3: valid From c79a2709cd018e4d4788ba0305195b133f3a073a Mon Sep 17 00:00:00 2001 From: Abhishek Choudhary Date: Thu, 24 Sep 2026 14:41:14 +0545 Subject: [PATCH 3/4] refactor(ai-rate-limiting): reference nested usage fields as parent__child in cost_expr --- apisix/plugins/ai-rate-limiting.lua | 206 ++++++--------------- docs/en/latest/plugins/ai-rate-limiting.md | 2 +- docs/zh/latest/plugins/ai-rate-limiting.md | 2 +- t/plugin/ai-rate-limiting-expression.t | 175 ++++------------- 4 files changed, 88 insertions(+), 297 deletions(-) diff --git a/apisix/plugins/ai-rate-limiting.lua b/apisix/plugins/ai-rate-limiting.lua index d9f00ed7bdb8..846263899ec0 100644 --- a/apisix/plugins/ai-rate-limiting.lua +++ b/apisix/plugins/ai-rate-limiting.lua @@ -20,17 +20,11 @@ local ipairs = ipairs local type = type local pairs = pairs local rawget = rawget -local rawset = rawset local pcall = pcall local load = load local math_floor = math.floor local math_huge = math.huge local table_sort = table.sort -local table_concat = table.concat -local str_find = string.find -local str_sub = string.sub -local str_gsub = string.gsub -local str_gmatch = string.gmatch local core = require("apisix.core") local limit_count = require("apisix.plugins.limit-count.init") local policy_to_additional_properties = limit_count.policy_to_additional_properties @@ -189,10 +183,6 @@ local limit_conf_cache = core.lrucache.new({ ttl = 300, count = 512 }) -local cost_expr_cache = core.lrucache.new({ - ttl = 300, count = 512 -}) - -- safe math functions allowed in cost expressions local expr_safe_env = { @@ -204,147 +194,14 @@ local expr_safe_env = { min = math.min, } -local lua_keywords = { - ["and"] = true, ["break"] = true, ["do"] = true, ["else"] = true, - ["elseif"] = true, ["end"] = true, ["false"] = true, ["for"] = true, - ["function"] = true, ["goto"] = true, ["if"] = true, ["in"] = true, - ["local"] = true, ["nil"] = true, ["not"] = true, ["or"] = true, - ["repeat"] = true, ["return"] = true, ["then"] = true, ["true"] = true, - ["until"] = true, ["while"] = true, -} - --- private keys, unreachable from the expression -local RAW_KEY = {} -local PATHS_KEY = {} - - -local function walk_path(raw, segs) - local v = raw - for i = 1, #segs do - if type(v) ~= "table" then - return 0 - end - v = v[segs[i]] - end - if type(v) == "number" then - return v - end - return 0 -end - - --- search level by level; a clash on one level charges the larger value -local function find_leaf(raw, name) - local v = raw[name] - if type(v) == "number" then - return v - end - - local level, prefixes = {raw}, {""} - while true do - local best, hits = nil, 0 - for i = 1, #level do - for k, tab in pairs(level[i]) do - if type(k) == "string" and type(tab) == "table" then - local x = tab[name] - if type(x) == "number" then - hits = hits + 1 - if not best or x > best then - best = x - end - end - end - end - end - - if best then - if hits > 1 then - local paths = {} - for i = 1, #level do - for k, tab in pairs(level[i]) do - if type(k) == "string" and type(tab) == "table" - and type(tab[name]) == "number" then - paths[#paths + 1] = prefixes[i] .. k .. "." .. name - end - end - end - table_sort(paths) - core.log.error("ambiguous usage field '", name, "' in cost_expr matches ", - table_concat(paths, ", "), ", charging the larger value, ", - "use an explicit path instead") - end - return best - end - - local next_level, next_prefixes = {}, {} - for i = 1, #level do - for k, tab in pairs(level[i]) do - if type(k) == "string" and type(tab) == "table" then - next_level[#next_level + 1] = tab - next_prefixes[#next_prefixes + 1] = prefixes[i] .. k .. "." - end - end - end - if #next_level == 0 then - return 0 - end - level, prefixes = next_level, next_prefixes - end -end - - -local usage_mt = { - __index = function(t, k) - local segs = rawget(t, PATHS_KEY)[k] - local v - if segs then - v = walk_path(rawget(t, RAW_KEY), segs) - else - v = find_leaf(rawget(t, RAW_KEY), k) - end - rawset(t, k, v) - return v - end -} - - --- rewrite usage names to reads on the `usage` argument local function compile_cost_expr(expr_str) - local paths = {} - local bad_name - local code = str_gsub(expr_str, "()([%a_][%w_%.]*)", function(pos, name) - -- skip number literals like 1e5 or 0x1F - if pos > 1 and str_find(str_sub(expr_str, pos - 1, pos - 1), "[%w_%.]") then - return nil - end - local head = str_gsub(name, "%..*", "") - if lua_keywords[name] or expr_safe_env[head] then - return nil - end - if str_find(name, ".", 1, true) then - if str_find(name, "..", 1, true) or str_sub(name, -1) == "." then - bad_name = bad_name or name - return nil - end - local segs = {} - for seg in str_gmatch(name, "[^%.]+") do - segs[#segs + 1] = seg - end - paths[name] = segs - end - return 'usage["' .. name .. '"]' - end) - if bad_name then - return nil, "invalid field reference: " .. bad_name - end - - -- own env per expression so writes stay local to it - local env = setmetatable({}, {__index = expr_safe_env}) - local fn, err = load("local usage = ...\nreturn " .. code, "cost_expr", "t", env) + local fn_code = "return " .. expr_str + -- validate syntax by loading first + local fn, err = load(fn_code, "cost_expr", "t", expr_safe_env) if not fn then return nil, err end - return {fn = fn, paths = paths} + return fn_code end @@ -495,13 +352,56 @@ function _M.check_instance_status(conf, ctx, instance_name) end -local function eval_cost_expr(conf, raw) - local compiled, err = cost_expr_cache(conf, nil, compile_cost_expr, conf.cost_expr) - if not compiled then +-- expose nested usage fields as parent__child, shallower keys win +local function inject_usage_vars(env, raw) + local level = {{raw, nil}} + while #level > 0 do + local next_level = {} + for _, item in ipairs(level) do + local tab, prefix = item[1], item[2] + local keys = {} + for k in pairs(tab) do + if type(k) == "string" then + keys[#keys + 1] = k + end + end + -- sorted for a stable winner on same-depth collisions + table_sort(keys) + for _, k in ipairs(keys) do + local v = tab[k] + local path = prefix and (prefix .. "__" .. k) or k + if type(v) == "number" then + if rawget(env, path) == nil and not expr_safe_env[path] then + env[path] = v + end + elseif type(v) == "table" then + next_level[#next_level + 1] = {v, path} + end + end + end + level = next_level + end +end + + +local function eval_cost_expr(conf_cost_expr, raw) + local fn_code = "return " .. conf_cost_expr + -- build environment: safe math + usage variables (missing vars default to 0) + local env = setmetatable({}, { + __index = function(_, k) + local v = expr_safe_env[k] + if v ~= nil then + return v + end + return 0 + end + }) + inject_usage_vars(env, raw) + local fn, err = load(fn_code, "cost_expr", "t", env) + if not fn then return nil, "failed to compile cost_expr: " .. err end - local usage = setmetatable({[RAW_KEY] = raw, [PATHS_KEY] = compiled.paths}, usage_mt) - local ok, result = pcall(compiled.fn, usage) + local ok, result = pcall(fn) if not ok then return nil, "failed to evaluate cost_expr: " .. result end @@ -523,7 +423,7 @@ local function get_token_usage(conf, ctx) if not raw then return end - local result, err = eval_cost_expr(conf, raw) + local result, err = eval_cost_expr(conf.cost_expr, raw) if not result then core.log.error(err) return diff --git a/docs/en/latest/plugins/ai-rate-limiting.md b/docs/en/latest/plugins/ai-rate-limiting.md index b2300d8b449b..76f758a8c9b1 100644 --- a/docs/en/latest/plugins/ai-rate-limiting.md +++ b/docs/en/latest/plugins/ai-rate-limiting.md @@ -48,7 +48,7 @@ The `ai-rate-limiting` Plugin enforces token-based rate limiting for requests se | time_window | integer | False | | >0 | The time interval corresponding to the rate limiting `limit` in seconds. At least one of `time_window` and `instances.time_window` should be configured. Required if `rules` is not configured. | | show_limit_quota_header | boolean | False | true | | If true, includes rate limiting response headers. When `rules` is not set, the headers are `X-AI-RateLimit-Limit-*`, `X-AI-RateLimit-Remaining-*`, and `X-AI-RateLimit-Reset-*`, where `*` is the instance name. When `rules` is set, see `rules.header_prefix` for details. | | limit_strategy | string | False | total_tokens | [`total_tokens`, `prompt_tokens`, `completion_tokens`, `expression`] | Type of token to apply rate limiting. `total_tokens` is the sum of `prompt_tokens` and `completion_tokens`. When set to `expression`, the `cost_expr` field is used to dynamically calculate token cost. | -| cost_expr | string | False | | | Lua arithmetic expression for dynamic token cost calculation. Variables are injected from the LLM API raw usage response fields. Nested fields can be referenced by their leaf name, e.g. `cached_tokens` for `input_tokens_details.cached_tokens`, or by an explicit path such as `input_tokens_details.cached_tokens`. A leaf name resolves to the shallowest match. If it matches several fields at the same depth, such as `audio_tokens` in both `prompt_tokens_details` and `completion_tokens_details`, the larger value is used and an error is logged; use an explicit path to pick one. Missing variables default to 0. Only valid when `limit_strategy` is `expression`. Example: `input_tokens + cache_creation_input_tokens + output_tokens`. | +| cost_expr | string | False | | | Lua arithmetic expression for dynamic token cost calculation. Variables are injected from the LLM API raw usage response fields. Nested fields are referenced by joining the parent and child keys with `__`, e.g. `input_tokens_details__cached_tokens` for `input_tokens_details.cached_tokens`. Arrays are skipped. Missing variables default to 0. Only valid when `limit_strategy` is `expression`. Example: `input_tokens + cache_creation_input_tokens + output_tokens`. | | instances | array[object] | False | | | LLM instance rate limiting configurations. | | instances.name | string | True | | | Name of the LLM service instance. | | instances.limit | integer | True | | >0 | The maximum number of tokens allowed within a given time interval for an instance. | diff --git a/docs/zh/latest/plugins/ai-rate-limiting.md b/docs/zh/latest/plugins/ai-rate-limiting.md index 0644507c3170..7a2a728a9bf3 100644 --- a/docs/zh/latest/plugins/ai-rate-limiting.md +++ b/docs/zh/latest/plugins/ai-rate-limiting.md @@ -48,7 +48,7 @@ import TabItem from '@theme/TabItem'; | time_window | integer | False | | >0 | 与速率限制 `limit` 对应的时间间隔(秒)。`time_window` 和 `instances.time_window` 中至少应配置一个。如果未配置 `rules`,则为必填项。 | | show_limit_quota_header | boolean | False | true | | 如果为 true,则在响应中包含速率限制头部。当未设置 `rules` 时,头部为 `X-AI-RateLimit-Limit-*`、`X-AI-RateLimit-Remaining-*` 和 `X-AI-RateLimit-Reset-*`,其中 `*` 是实例名称。当设置了 `rules` 时,详见 `rules.header_prefix`。 | | limit_strategy | string | False | total_tokens | [`total_tokens`, `prompt_tokens`, `completion_tokens`, `expression`] | 应用速率限制的令牌类型。`total_tokens` 是 `prompt_tokens` 和 `completion_tokens` 的总和。当设置为 `expression` 时,使用 `cost_expr` 字段动态计算令牌消耗。 | -| cost_expr | string | False | | | 用于动态计算令牌消耗的 Lua 算术表达式。变量从 LLM API 原始使用量响应字段注入。嵌套字段可以用叶子字段名引用,例如用 `cached_tokens` 引用 `input_tokens_details.cached_tokens`,也可以用显式路径引用,例如 `input_tokens_details.cached_tokens`。叶子字段名解析为层级最浅的匹配字段。若同一层级有多个同名字段,例如 `prompt_tokens_details` 和 `completion_tokens_details` 中都有 `audio_tokens`,则取较大值并记录错误日志;请使用显式路径指定字段。缺失的变量默认为 0。仅在 `limit_strategy` 为 `expression` 时有效。示例:`input_tokens + cache_creation_input_tokens + output_tokens`。 | +| cost_expr | string | False | | | 用于动态计算令牌消耗的 Lua 算术表达式。变量从 LLM API 原始使用量响应字段注入。嵌套字段通过用 `__` 连接父字段名和子字段名来引用,例如用 `input_tokens_details__cached_tokens` 引用 `input_tokens_details.cached_tokens`。数组会被跳过。缺失的变量默认为 0。仅在 `limit_strategy` 为 `expression` 时有效。示例:`input_tokens + cache_creation_input_tokens + output_tokens`。 | | instances | array[object] | False | | | LLM 实例速率限制配置。 | | instances.name | string | True | | | LLM 服务实例的名称。 | | instances.limit | integer | True | | >0 | 实例在给定时间间隔内允许的最大令牌数。 | diff --git a/t/plugin/ai-rate-limiting-expression.t b/t/plugin/ai-rate-limiting-expression.t index aa8de06d8e2a..3598ce0acafc 100644 --- a/t/plugin/ai-rate-limiting-expression.t +++ b/t/plugin/ai-rate-limiting-expression.t @@ -522,7 +522,7 @@ X-AI-Fixture: anthropic/messages-with-cache.json -=== TEST 14: set route with expression reading nested usage leaves (OpenAI Responses) +=== TEST 14: set route with expression reading nested usage fields as parent__child (OpenAI Responses) --- config location /t { content_by_lua_block { @@ -551,7 +551,7 @@ X-AI-Fixture: anthropic/messages-with-cache.json "limit": 500, "time_window": 60, "limit_strategy": "expression", - "cost_expr": "input_tokens - cached_tokens + output_tokens + reasoning_tokens" + "cost_expr": "input_tokens - input_tokens_details__cached_tokens + output_tokens + output_tokens_details__reasoning_tokens" } }, "upstream": { @@ -574,7 +574,7 @@ passed -=== TEST 15: nested leaves - cost = 40 - 12 + 20 + 8 = 56 per request +=== TEST 15: nested fields - cost = 40 - 12 + 20 + 8 = 56 per request --- pipelined_requests eval [ "POST /v1/responses\n" . '{"model":"gpt-4o-mini","stream":false,"input":"Hello"}', @@ -592,7 +592,7 @@ X-AI-Fixture: openai/responses-with-cache.json -=== TEST 16: nested leaves, streaming - cost = 20 - 10 + 5 + 3 = 18 per request +=== TEST 16: nested fields, streaming - cost = 20 - 10 + 5 + 3 = 18 per request --- pipelined_requests eval [ "POST /v1/responses\n" . '{"model":"gpt-4o-mini","stream":true,"input":"Hello"}', @@ -610,7 +610,7 @@ X-AI-Fixture: openai/responses-streaming-with-cache.sse -=== TEST 17: set route with expression reading a leaf name present at two depths +=== TEST 17: set route with a bare name that is nested only --- config location /t { content_by_lua_block { @@ -639,7 +639,7 @@ X-AI-Fixture: openai/responses-streaming-with-cache.sse "limit": 500, "time_window": 60, "limit_strategy": "expression", - "cost_expr": "cached_tokens" + "cost_expr": "input_tokens + cached_tokens" } }, "upstream": { @@ -662,25 +662,25 @@ passed -=== TEST 18: top-level field wins over nested one - cost = 5 per request +=== TEST 18: nested fields are not bound by their bare name - cost = 40 + 0 = 40 per request --- pipelined_requests eval [ "POST /v1/responses\n" . '{"model":"gpt-4o-mini","stream":false,"input":"Hello"}', "POST /v1/responses\n" . '{"model":"gpt-4o-mini","stream":false,"input":"Hello"}', ] --- more_headers -X-AI-Fixture: openai/responses-usage-clash.json +X-AI-Fixture: openai/responses-with-cache.json --- response_headers_like eval [ "X-AI-RateLimit-Remaining-ai-proxy-openai: 500", - "X-AI-RateLimit-Remaining-ai-proxy-openai: 495", + "X-AI-RateLimit-Remaining-ai-proxy-openai: 460", ] --- no_error_log [error] -=== TEST 19: set route with an explicit path to a field that also exists at the top level +=== TEST 19: set route with a bare name present at the top level and nested --- config location /t { content_by_lua_block { @@ -709,7 +709,7 @@ X-AI-Fixture: openai/responses-usage-clash.json "limit": 500, "time_window": 60, "limit_strategy": "expression", - "cost_expr": "input_tokens_details.cached_tokens" + "cost_expr": "cached_tokens" } }, "upstream": { @@ -732,7 +732,7 @@ passed -=== TEST 20: explicit path reads the nested field - cost = 12 per request +=== TEST 20: bare name reads the top-level field - cost = 5 per request --- pipelined_requests eval [ "POST /v1/responses\n" . '{"model":"gpt-4o-mini","stream":false,"input":"Hello"}', @@ -743,14 +743,14 @@ X-AI-Fixture: openai/responses-usage-clash.json --- response_headers_like eval [ "X-AI-RateLimit-Remaining-ai-proxy-openai: 500", - "X-AI-RateLimit-Remaining-ai-proxy-openai: 488", + "X-AI-RateLimit-Remaining-ai-proxy-openai: 495", ] --- no_error_log [error] -=== TEST 21: set route with a bare name present in two sibling objects +=== TEST 21: set route with the parent__child name of the same field --- config location /t { content_by_lua_block { @@ -758,7 +758,7 @@ X-AI-Fixture: openai/responses-usage-clash.json local code, body = t('/apisix/admin/routes/1', ngx.HTTP_PUT, [[{ - "uri": "/v1/chat/completions", + "uri": "/v1/responses", "plugins": { "ai-proxy": { "provider": "openai", @@ -779,7 +779,7 @@ X-AI-Fixture: openai/responses-usage-clash.json "limit": 500, "time_window": 60, "limit_strategy": "expression", - "cost_expr": "audio_tokens" + "cost_expr": "input_tokens_details__cached_tokens" } }, "upstream": { @@ -802,36 +802,25 @@ passed -=== TEST 22: ambiguous bare name charges the larger value - cost = 70 per request +=== TEST 22: parent__child reads the nested field - cost = 12 per request --- pipelined_requests eval [ - "POST /v1/chat/completions\n" . '{"model":"gpt-4o-mini","messages":[{"role":"user","content":"Hello"}]}', - "POST /v1/chat/completions\n" . '{"model":"gpt-4o-mini","messages":[{"role":"user","content":"Hello"}]}', + "POST /v1/responses\n" . '{"model":"gpt-4o-mini","stream":false,"input":"Hello"}', + "POST /v1/responses\n" . '{"model":"gpt-4o-mini","stream":false,"input":"Hello"}', ] --- more_headers -X-AI-Fixture: openai/chat-usage-audio.json +X-AI-Fixture: openai/responses-usage-clash.json --- response_headers_like eval [ "X-AI-RateLimit-Remaining-ai-proxy-openai: 500", - "X-AI-RateLimit-Remaining-ai-proxy-openai: 430", + "X-AI-RateLimit-Remaining-ai-proxy-openai: 488", ] --- no_error_log -[alert] - - - -=== TEST 23: ambiguous bare name logs the explicit paths to use ---- request -POST /v1/chat/completions -{"model":"gpt-4o-mini","messages":[{"role":"user","content":"Hello"}]} ---- more_headers -X-AI-Fixture: openai/chat-usage-audio.json ---- error_log -ambiguous usage field 'audio_tokens' in cost_expr matches completion_tokens_details.audio_tokens, prompt_tokens_details.audio_tokens, charging the larger value, use an explicit path instead +[error] -=== TEST 24: set route mixing explicit paths and bare names +=== TEST 23: set route with a leaf name present in two sibling objects --- config location /t { content_by_lua_block { @@ -860,7 +849,7 @@ ambiguous usage field 'audio_tokens' in cost_expr matches completion_tokens_deta "limit": 500, "time_window": 60, "limit_strategy": "expression", - "cost_expr": "prompt_tokens_details.audio_tokens * 2 + completion_tokens_details.audio_tokens + cached_tokens" + "cost_expr": "prompt_tokens_details__audio_tokens * 2 + completion_tokens_details__audio_tokens + audio_tokens" } }, "upstream": { @@ -883,7 +872,7 @@ passed -=== TEST 25: explicit paths resolve the clash - cost = 30 * 2 + 70 + 20 = 150 per request +=== TEST 24: parent__child names pick each sibling field - cost = 30 * 2 + 70 + 0 = 130 per request --- pipelined_requests eval [ "POST /v1/chat/completions\n" . '{"model":"gpt-4o-mini","messages":[{"role":"user","content":"Hello"}]}', @@ -894,14 +883,14 @@ X-AI-Fixture: openai/chat-usage-audio.json --- response_headers_like eval [ "X-AI-RateLimit-Remaining-ai-proxy-openai: 500", - "X-AI-RateLimit-Remaining-ai-proxy-openai: 350", + "X-AI-RateLimit-Remaining-ai-proxy-openai: 370", ] --- no_error_log [error] -=== TEST 26: set route with explicit paths that do not exist +=== TEST 25: set route with parent__child names that do not exist --- config location /t { content_by_lua_block { @@ -930,7 +919,7 @@ X-AI-Fixture: openai/chat-usage-audio.json "limit": 500, "time_window": 60, "limit_strategy": "expression", - "cost_expr": "prompt_tokens + no_such_details.audio_tokens + prompt_tokens_details.no_such_field" + "cost_expr": "prompt_tokens + no_such_details__audio_tokens + prompt_tokens_details__no_such_field" } }, "upstream": { @@ -953,7 +942,7 @@ passed -=== TEST 27: missing explicit paths default to 0 - cost = 100 per request +=== TEST 26: missing parent__child names default to 0 - cost = 100 per request --- pipelined_requests eval [ "POST /v1/chat/completions\n" . '{"model":"gpt-4o-mini","messages":[{"role":"user","content":"Hello"}]}', @@ -971,7 +960,7 @@ X-AI-Fixture: openai/chat-usage-audio.json -=== TEST 28: set route reading a field nested two levels deep +=== TEST 27: set route reading a field nested two levels deep --- config location /t { content_by_lua_block { @@ -1000,7 +989,7 @@ X-AI-Fixture: openai/chat-usage-audio.json "limit": 500, "time_window": 60, "limit_strategy": "expression", - "cost_expr": "text_tokens + prompt_tokens_details.cached_tokens_details.text_tokens" + "cost_expr": "prompt_tokens_details__cached_tokens_details__text_tokens + modality_details__text_tokens" } }, "upstream": { @@ -1023,7 +1012,7 @@ passed -=== TEST 29: deep bare name and deep path both resolve, arrays are skipped - cost = 9 + 9 = 18 per request +=== TEST 28: deep fields join every level, arrays are skipped - cost = 9 + 0 = 9 per request --- pipelined_requests eval [ "POST /v1/chat/completions\n" . '{"model":"gpt-4o-mini","messages":[{"role":"user","content":"Hello"}]}', @@ -1034,105 +1023,7 @@ X-AI-Fixture: openai/chat-usage-deep.json --- response_headers_like eval [ "X-AI-RateLimit-Remaining-ai-proxy-openai: 500", - "X-AI-RateLimit-Remaining-ai-proxy-openai: 482", -] ---- no_error_log -[error] - - - -=== TEST 30: set route using math helpers and number literals around explicit paths ---- config - location /t { - content_by_lua_block { - local t = require("lib.test_admin").test - local code, body = t('/apisix/admin/routes/1', - ngx.HTTP_PUT, - [[{ - "uri": "/v1/chat/completions", - "plugins": { - "ai-proxy": { - "provider": "openai", - "auth": { - "header": { - "Authorization": "Bearer test-key" - } - }, - "options": { - "model": "gpt-4o-mini" - }, - "override": { - "endpoint": "http://127.0.0.1:1980" - }, - "ssl_verify": false - }, - "ai-rate-limiting": { - "limit": 500, - "time_window": 60, - "limit_strategy": "expression", - "cost_expr": "floor(completion_tokens_details.audio_tokens / 1e1) + math.max(reasoning_tokens, 1)" - } - }, - "upstream": { - "type": "roundrobin", - "nodes": { - "canbeanything.com": 1 - } - } - }]] - ) - - if code >= 300 then - ngx.status = code - end - ngx.say(body) - } - } ---- response_body -passed - - - -=== TEST 31: helpers and literals are left as is - cost = 7 + 10 = 17 per request ---- pipelined_requests eval -[ - "POST /v1/chat/completions\n" . '{"model":"gpt-4o-mini","messages":[{"role":"user","content":"Hello"}]}', - "POST /v1/chat/completions\n" . '{"model":"gpt-4o-mini","messages":[{"role":"user","content":"Hello"}]}', -] ---- more_headers -X-AI-Fixture: openai/chat-usage-audio.json ---- response_headers_like eval -[ - "X-AI-RateLimit-Remaining-ai-proxy-openai: 500", - "X-AI-RateLimit-Remaining-ai-proxy-openai: 483", + "X-AI-RateLimit-Remaining-ai-proxy-openai: 491", ] --- no_error_log [error] - - - -=== TEST 32: schema validation - malformed explicit paths are rejected ---- config - location /t { - content_by_lua_block { - local plugin = require("apisix.plugins.ai-rate-limiting") - local exprs = { - "input_tokens_details..cached_tokens", - "input_tokens_details. + 1", - "math.floor(input_tokens_details.cached_tokens) + 1e3", - } - for i, expr in ipairs(exprs) do - local ok, err = plugin.check_schema({ - limit = 100, - time_window = 60, - limit_strategy = "expression", - cost_expr = expr, - }) - ngx.say("expr ", i, ": ", ok and "valid" or err) - end - } - } ---- response_body -expr 1: invalid cost_expr: invalid field reference: input_tokens_details..cached_tokens -expr 2: invalid cost_expr: invalid field reference: input_tokens_details. -expr 3: valid From b64b7c7b55a07ed56e0d587c433ec56731bd6a34 Mon Sep 17 00:00:00 2001 From: Abhishek Choudhary Date: Thu, 24 Sep 2026 14:49:43 +0545 Subject: [PATCH 4/4] refactor(ai-rate-limiting): drop key sort from usage injection --- apisix/plugins/ai-rate-limiting.lua | 24 ++++++++---------------- 1 file changed, 8 insertions(+), 16 deletions(-) diff --git a/apisix/plugins/ai-rate-limiting.lua b/apisix/plugins/ai-rate-limiting.lua index 846263899ec0..31a4c1376962 100644 --- a/apisix/plugins/ai-rate-limiting.lua +++ b/apisix/plugins/ai-rate-limiting.lua @@ -24,7 +24,6 @@ local pcall = pcall local load = load local math_floor = math.floor local math_huge = math.huge -local table_sort = table.sort local core = require("apisix.core") local limit_count = require("apisix.plugins.limit-count.init") local policy_to_additional_properties = limit_count.policy_to_additional_properties @@ -359,23 +358,16 @@ local function inject_usage_vars(env, raw) local next_level = {} for _, item in ipairs(level) do local tab, prefix = item[1], item[2] - local keys = {} - for k in pairs(tab) do + for k, v in pairs(tab) do if type(k) == "string" then - keys[#keys + 1] = k - end - end - -- sorted for a stable winner on same-depth collisions - table_sort(keys) - for _, k in ipairs(keys) do - local v = tab[k] - local path = prefix and (prefix .. "__" .. k) or k - if type(v) == "number" then - if rawget(env, path) == nil and not expr_safe_env[path] then - env[path] = v + local path = prefix and (prefix .. "__" .. k) or k + if type(v) == "number" then + if rawget(env, path) == nil and not expr_safe_env[path] then + env[path] = v + end + elseif type(v) == "table" then + next_level[#next_level + 1] = {v, path} end - elseif type(v) == "table" then - next_level[#next_level + 1] = {v, path} end end end