Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
32 changes: 27 additions & 5 deletions apisix/plugins/ai-rate-limiting.lua
Original file line number Diff line number Diff line change
Expand Up @@ -19,6 +19,7 @@ local setmetatable = setmetatable
local ipairs = ipairs
local type = type
local pairs = pairs
local rawget = rawget
local pcall = pcall
local load = load
local math_floor = math.floor
Expand Down Expand Up @@ -350,6 +351,31 @@ function _M.check_instance_status(conf, ctx, instance_name)
end


-- expose nested usage fields as parent__child, shallower keys win
local function inject_usage_vars(env, raw)
local level = {{raw, nil}}
while #level > 0 do
local next_level = {}
for _, item in ipairs(level) do
local tab, prefix = item[1], item[2]
for k, v in pairs(tab) do
if type(k) == "string" then
local path = prefix and (prefix .. "__" .. k) or k
if type(v) == "number" then
if rawget(env, path) == nil and not expr_safe_env[path] then
env[path] = v
end
elseif type(v) == "table" then
next_level[#next_level + 1] = {v, path}
end
end
end
end
level = next_level
end
end


local function eval_cost_expr(conf_cost_expr, raw)
local fn_code = "return " .. conf_cost_expr
-- build environment: safe math + usage variables (missing vars default to 0)
Expand All @@ -362,11 +388,7 @@ local function eval_cost_expr(conf_cost_expr, raw)
return 0
end
})
for k, v in pairs(raw) do
if type(v) == "number" and not expr_safe_env[k] then
env[k] = v
end
end
inject_usage_vars(env, raw)
local fn, err = load(fn_code, "cost_expr", "t", env)
if not fn then
return nil, "failed to compile cost_expr: " .. err
Expand Down
2 changes: 1 addition & 1 deletion docs/en/latest/plugins/ai-rate-limiting.md
Original file line number Diff line number Diff line change
Expand Up @@ -48,7 +48,7 @@ The `ai-rate-limiting` Plugin enforces token-based rate limiting for requests se
| time_window | integer | False | | >0 | The time interval corresponding to the rate limiting `limit` in seconds. At least one of `time_window` and `instances.time_window` should be configured. Required if `rules` is not configured. |
| show_limit_quota_header | boolean | False | true | | If true, includes rate limiting response headers. When `rules` is not set, the headers are `X-AI-RateLimit-Limit-*`, `X-AI-RateLimit-Remaining-*`, and `X-AI-RateLimit-Reset-*`, where `*` is the instance name. When `rules` is set, see `rules.header_prefix` for details. |
| limit_strategy | string | False | total_tokens | [`total_tokens`, `prompt_tokens`, `completion_tokens`, `expression`] | Type of token to apply rate limiting. `total_tokens` is the sum of `prompt_tokens` and `completion_tokens`. When set to `expression`, the `cost_expr` field is used to dynamically calculate token cost. |
| cost_expr | string | False | | | Lua arithmetic expression for dynamic token cost calculation. Variables are injected from the LLM API raw usage response fields. Missing variables default to 0. Only valid when `limit_strategy` is `expression`. Example: `input_tokens + cache_creation_input_tokens + output_tokens`. |
| cost_expr | string | False | | | Lua arithmetic expression for dynamic token cost calculation. Variables are injected from the LLM API raw usage response fields. Nested fields are referenced by joining the parent and child keys with `__`, e.g. `input_tokens_details__cached_tokens` for `input_tokens_details.cached_tokens`. Arrays are skipped. Missing variables default to 0. Only valid when `limit_strategy` is `expression`. Example: `input_tokens + cache_creation_input_tokens + output_tokens`. |
| instances | array[object] | False | | | LLM instance rate limiting configurations. |
| instances.name | string | True | | | Name of the LLM service instance. |
| instances.limit | integer | True | | >0 | The maximum number of tokens allowed within a given time interval for an instance. |
Expand Down
2 changes: 1 addition & 1 deletion docs/zh/latest/plugins/ai-rate-limiting.md
Original file line number Diff line number Diff line change
Expand Up @@ -48,7 +48,7 @@ import TabItem from '@theme/TabItem';
| time_window | integer | False | | >0 | 与速率限制 `limit` 对应的时间间隔(秒)。`time_window` 和 `instances.time_window` 中至少应配置一个。如果未配置 `rules`,则为必填项。 |
| show_limit_quota_header | boolean | False | true | | 如果为 true,则在响应中包含速率限制头部。当未设置 `rules` 时,头部为 `X-AI-RateLimit-Limit-*`、`X-AI-RateLimit-Remaining-*` 和 `X-AI-RateLimit-Reset-*`,其中 `*` 是实例名称。当设置了 `rules` 时,详见 `rules.header_prefix`。 |
| limit_strategy | string | False | total_tokens | [`total_tokens`, `prompt_tokens`, `completion_tokens`, `expression`] | 应用速率限制的令牌类型。`total_tokens` 是 `prompt_tokens` 和 `completion_tokens` 的总和。当设置为 `expression` 时,使用 `cost_expr` 字段动态计算令牌消耗。 |
| cost_expr | string | False | | | 用于动态计算令牌消耗的 Lua 算术表达式。变量从 LLM API 原始使用量响应字段注入。缺失的变量默认为 0。仅在 `limit_strategy` 为 `expression` 时有效。示例:`input_tokens + cache_creation_input_tokens + output_tokens`。 |
| cost_expr | string | False | | | 用于动态计算令牌消耗的 Lua 算术表达式。变量从 LLM API 原始使用量响应字段注入。嵌套字段通过用 `__` 连接父字段名和子字段名来引用,例如用 `input_tokens_details__cached_tokens` 引用 `input_tokens_details.cached_tokens`。数组会被跳过。缺失的变量默认为 0。仅在 `limit_strategy` 为 `expression` 时有效。示例:`input_tokens + cache_creation_input_tokens + output_tokens`。 |
| instances | array[object] | False | | | LLM 实例速率限制配置。 |
| instances.name | string | True | | | LLM 服务实例的名称。 |
| instances.limit | integer | True | | >0 | 实例在给定时间间隔内允许的最大令牌数。 |
Expand Down
19 changes: 19 additions & 0 deletions t/fixtures/openai/chat-usage-audio.json
Original file line number Diff line number Diff line change
@@ -0,0 +1,19 @@
{
"id": "chatcmpl-audio1",
"object": "chat.completion",
"model": "{{model}}",
"choices": [
{
"index": 0,
"message": { "role": "assistant", "content": "Hello" },
"finish_reason": "stop"
}
],
"usage": {
"prompt_tokens": 100,
"completion_tokens": 50,
"total_tokens": 150,
"prompt_tokens_details": { "cached_tokens": 20, "audio_tokens": 30 },
"completion_tokens_details": { "reasoning_tokens": 10, "audio_tokens": 70 }
}
}
22 changes: 22 additions & 0 deletions t/fixtures/openai/chat-usage-deep.json
Original file line number Diff line number Diff line change
@@ -0,0 +1,22 @@
{
"id": "chatcmpl-deep1",
"object": "chat.completion",
"model": "{{model}}",
"choices": [
{
"index": 0,
"message": { "role": "assistant", "content": "Hello" },
"finish_reason": "stop"
}
],
"usage": {
"prompt_tokens": 100,
"completion_tokens": 50,
"total_tokens": 150,
"prompt_tokens_details": {
"cached_tokens": 20,
"cached_tokens_details": { "text_tokens": 9 }
},
"modality_details": [ { "text_tokens": 100 } ]
}
}
23 changes: 23 additions & 0 deletions t/fixtures/openai/responses-usage-clash.json
Original file line number Diff line number Diff line change
@@ -0,0 +1,23 @@
{
"id": "resp_clash1",
"object": "response",
"created_at": 1723780938,
"model": "{{model}}",
"output": [
{
"type": "message",
"role": "assistant",
"content": [
{ "type": "output_text", "text": "Hello" }
]
}
],
"usage": {
"input_tokens": 40,
"output_tokens": 20,
"total_tokens": 60,
"cached_tokens": 5,
"input_tokens_details": { "cached_tokens": 12 },
"output_tokens_details": { "reasoning_tokens": 8 }
}
}
Loading
Loading