feat(ai-rate-limiting): expose nested usage fields to cost_expr (#13984)
diff --git a/apisix/plugins/ai-rate-limiting.lua b/apisix/plugins/ai-rate-limiting.lua
index 489176a..31a4c13 100644
--- a/apisix/plugins/ai-rate-limiting.lua
+++ b/apisix/plugins/ai-rate-limiting.lua
@@ -19,6 +19,7 @@
 local ipairs = ipairs
 local type = type
 local pairs = pairs
+local rawget = rawget
 local pcall = pcall
 local load = load
 local math_floor = math.floor
@@ -350,6 +351,31 @@
 end
 
 
+-- expose nested usage fields as parent__child, shallower keys win
+local function inject_usage_vars(env, raw)
+    local level = {{raw, nil}}
+    while #level > 0 do
+        local next_level = {}
+        for _, item in ipairs(level) do
+            local tab, prefix = item[1], item[2]
+            for k, v in pairs(tab) do
+                if type(k) == "string" then
+                    local path = prefix and (prefix .. "__" .. k) or k
+                    if type(v) == "number" then
+                        if rawget(env, path) == nil and not expr_safe_env[path] then
+                            env[path] = v
+                        end
+                    elseif type(v) == "table" then
+                        next_level[#next_level + 1] = {v, path}
+                    end
+                end
+            end
+        end
+        level = next_level
+    end
+end
+
+
 local function eval_cost_expr(conf_cost_expr, raw)
     local fn_code = "return " .. conf_cost_expr
     -- build environment: safe math + usage variables (missing vars default to 0)
@@ -362,11 +388,7 @@
             return 0
         end
     })
-    for k, v in pairs(raw) do
-        if type(v) == "number" and not expr_safe_env[k] then
-            env[k] = v
-        end
-    end
+    inject_usage_vars(env, raw)
     local fn, err = load(fn_code, "cost_expr", "t", env)
     if not fn then
         return nil, "failed to compile cost_expr: " .. err
diff --git a/docs/en/latest/plugins/ai-rate-limiting.md b/docs/en/latest/plugins/ai-rate-limiting.md
index 4d8757f..76f758a 100644
--- a/docs/en/latest/plugins/ai-rate-limiting.md
+++ b/docs/en/latest/plugins/ai-rate-limiting.md
@@ -48,7 +48,7 @@
 | time_window | integer | False | | >0 | The time interval corresponding to the rate limiting `limit` in seconds. At least one of `time_window` and `instances.time_window` should be configured. Required if `rules` is not configured. |
 | show_limit_quota_header | boolean | False | true | | If true, includes rate limiting response headers. When `rules` is not set, the headers are `X-AI-RateLimit-Limit-*`, `X-AI-RateLimit-Remaining-*`, and `X-AI-RateLimit-Reset-*`, where `*` is the instance name. When `rules` is set, see `rules.header_prefix` for details. |
 | limit_strategy | string | False | total_tokens | [`total_tokens`, `prompt_tokens`, `completion_tokens`, `expression`] | Type of token to apply rate limiting. `total_tokens` is the sum of `prompt_tokens` and `completion_tokens`. When set to `expression`, the `cost_expr` field is used to dynamically calculate token cost. |
-| cost_expr | string | False | | | Lua arithmetic expression for dynamic token cost calculation. Variables are injected from the LLM API raw usage response fields. Missing variables default to 0. Only valid when `limit_strategy` is `expression`. Example: `input_tokens + cache_creation_input_tokens + output_tokens`. |
+| cost_expr | string | False | | | Lua arithmetic expression for dynamic token cost calculation. Variables are injected from the LLM API raw usage response fields. Nested fields are referenced by joining the parent and child keys with `__`, e.g. `input_tokens_details__cached_tokens` for `input_tokens_details.cached_tokens`. Arrays are skipped. Missing variables default to 0. Only valid when `limit_strategy` is `expression`. Example: `input_tokens + cache_creation_input_tokens + output_tokens`. |
 | instances | array[object] | False | | | LLM instance rate limiting configurations. |
 | instances.name | string | True | | | Name of the LLM service instance. |
 | instances.limit | integer | True | | >0 | The maximum number of tokens allowed within a given time interval for an instance. |
diff --git a/docs/zh/latest/plugins/ai-rate-limiting.md b/docs/zh/latest/plugins/ai-rate-limiting.md
index 7cf41a4..7a2a728 100644
--- a/docs/zh/latest/plugins/ai-rate-limiting.md
+++ b/docs/zh/latest/plugins/ai-rate-limiting.md
@@ -48,7 +48,7 @@
 | time_window | integer | False | | >0 | 与速率限制 `limit` 对应的时间间隔(秒)。`time_window` 和 `instances.time_window` 中至少应配置一个。如果未配置 `rules`,则为必填项。 |
 | show_limit_quota_header | boolean | False | true | | 如果为 true,则在响应中包含速率限制头部。当未设置 `rules` 时,头部为 `X-AI-RateLimit-Limit-*`、`X-AI-RateLimit-Remaining-*` 和 `X-AI-RateLimit-Reset-*`,其中 `*` 是实例名称。当设置了 `rules` 时,详见 `rules.header_prefix`。 |
 | limit_strategy | string | False | total_tokens | [`total_tokens`, `prompt_tokens`, `completion_tokens`, `expression`] | 应用速率限制的令牌类型。`total_tokens` 是 `prompt_tokens` 和 `completion_tokens` 的总和。当设置为 `expression` 时,使用 `cost_expr` 字段动态计算令牌消耗。 |
-| cost_expr | string | False | | | 用于动态计算令牌消耗的 Lua 算术表达式。变量从 LLM API 原始使用量响应字段注入。缺失的变量默认为 0。仅在 `limit_strategy` 为 `expression` 时有效。示例:`input_tokens + cache_creation_input_tokens + output_tokens`。 |
+| cost_expr | string | False | | | 用于动态计算令牌消耗的 Lua 算术表达式。变量从 LLM API 原始使用量响应字段注入。嵌套字段通过用 `__` 连接父字段名和子字段名来引用,例如用 `input_tokens_details__cached_tokens` 引用 `input_tokens_details.cached_tokens`。数组会被跳过。缺失的变量默认为 0。仅在 `limit_strategy` 为 `expression` 时有效。示例:`input_tokens + cache_creation_input_tokens + output_tokens`。 |
 | instances | array[object] | False | | | LLM 实例速率限制配置。 |
 | instances.name | string | True | | | LLM 服务实例的名称。 |
 | instances.limit | integer | True | | >0 | 实例在给定时间间隔内允许的最大令牌数。 |
diff --git a/t/fixtures/openai/chat-usage-audio.json b/t/fixtures/openai/chat-usage-audio.json
new file mode 100644
index 0000000..5cdc7c6
--- /dev/null
+++ b/t/fixtures/openai/chat-usage-audio.json
@@ -0,0 +1,19 @@
+{
+  "id": "chatcmpl-audio1",
+  "object": "chat.completion",
+  "model": "{{model}}",
+  "choices": [
+    {
+      "index": 0,
+      "message": { "role": "assistant", "content": "Hello" },
+      "finish_reason": "stop"
+    }
+  ],
+  "usage": {
+    "prompt_tokens": 100,
+    "completion_tokens": 50,
+    "total_tokens": 150,
+    "prompt_tokens_details": { "cached_tokens": 20, "audio_tokens": 30 },
+    "completion_tokens_details": { "reasoning_tokens": 10, "audio_tokens": 70 }
+  }
+}
diff --git a/t/fixtures/openai/chat-usage-deep.json b/t/fixtures/openai/chat-usage-deep.json
new file mode 100644
index 0000000..e0da1fa
--- /dev/null
+++ b/t/fixtures/openai/chat-usage-deep.json
@@ -0,0 +1,22 @@
+{
+  "id": "chatcmpl-deep1",
+  "object": "chat.completion",
+  "model": "{{model}}",
+  "choices": [
+    {
+      "index": 0,
+      "message": { "role": "assistant", "content": "Hello" },
+      "finish_reason": "stop"
+    }
+  ],
+  "usage": {
+    "prompt_tokens": 100,
+    "completion_tokens": 50,
+    "total_tokens": 150,
+    "prompt_tokens_details": {
+      "cached_tokens": 20,
+      "cached_tokens_details": { "text_tokens": 9 }
+    },
+    "modality_details": [ { "text_tokens": 100 } ]
+  }
+}
diff --git a/t/fixtures/openai/responses-usage-clash.json b/t/fixtures/openai/responses-usage-clash.json
new file mode 100644
index 0000000..6883abc
--- /dev/null
+++ b/t/fixtures/openai/responses-usage-clash.json
@@ -0,0 +1,23 @@
+{
+  "id": "resp_clash1",
+  "object": "response",
+  "created_at": 1723780938,
+  "model": "{{model}}",
+  "output": [
+    {
+      "type": "message",
+      "role": "assistant",
+      "content": [
+        { "type": "output_text", "text": "Hello" }
+      ]
+    }
+  ],
+  "usage": {
+    "input_tokens": 40,
+    "output_tokens": 20,
+    "total_tokens": 60,
+    "cached_tokens": 5,
+    "input_tokens_details": { "cached_tokens": 12 },
+    "output_tokens_details": { "reasoning_tokens": 8 }
+  }
+}
diff --git a/t/plugin/ai-rate-limiting-expression.t b/t/plugin/ai-rate-limiting-expression.t
index dd69aa5..3598ce0 100644
--- a/t/plugin/ai-rate-limiting-expression.t
+++ b/t/plugin/ai-rate-limiting-expression.t
@@ -519,3 +519,511 @@
 ]
 --- no_error_log
 [error]
+
+
+
+=== TEST 14: set route with expression reading nested usage fields as parent__child (OpenAI Responses)
+--- config
+    location /t {
+        content_by_lua_block {
+            local t = require("lib.test_admin").test
+            local code, body = t('/apisix/admin/routes/1',
+                 ngx.HTTP_PUT,
+                 [[{
+                    "uri": "/v1/responses",
+                    "plugins": {
+                        "ai-proxy": {
+                            "provider": "openai",
+                            "auth": {
+                                "header": {
+                                    "Authorization": "Bearer test-key"
+                                }
+                            },
+                            "options": {
+                                "model": "gpt-4o-mini"
+                            },
+                            "override": {
+                                "endpoint": "http://127.0.0.1:1980"
+                            },
+                            "ssl_verify": false
+                        },
+                        "ai-rate-limiting": {
+                            "limit": 500,
+                            "time_window": 60,
+                            "limit_strategy": "expression",
+                            "cost_expr": "input_tokens - input_tokens_details__cached_tokens + output_tokens + output_tokens_details__reasoning_tokens"
+                        }
+                    },
+                    "upstream": {
+                        "type": "roundrobin",
+                        "nodes": {
+                            "canbeanything.com": 1
+                        }
+                    }
+                }]]
+            )
+
+            if code >= 300 then
+                ngx.status = code
+            end
+            ngx.say(body)
+        }
+    }
+--- response_body
+passed
+
+
+
+=== TEST 15: nested fields - cost = 40 - 12 + 20 + 8 = 56 per request
+--- pipelined_requests eval
+[
+    "POST /v1/responses\n" . '{"model":"gpt-4o-mini","stream":false,"input":"Hello"}',
+    "POST /v1/responses\n" . '{"model":"gpt-4o-mini","stream":false,"input":"Hello"}',
+]
+--- more_headers
+X-AI-Fixture: openai/responses-with-cache.json
+--- response_headers_like eval
+[
+    "X-AI-RateLimit-Remaining-ai-proxy-openai: 500",
+    "X-AI-RateLimit-Remaining-ai-proxy-openai: 444",
+]
+--- no_error_log
+[error]
+
+
+
+=== TEST 16: nested fields, streaming - cost = 20 - 10 + 5 + 3 = 18 per request
+--- pipelined_requests eval
+[
+    "POST /v1/responses\n" . '{"model":"gpt-4o-mini","stream":true,"input":"Hello"}',
+    "POST /v1/responses\n" . '{"model":"gpt-4o-mini","stream":true,"input":"Hello"}',
+]
+--- more_headers
+X-AI-Fixture: openai/responses-streaming-with-cache.sse
+--- response_headers_like eval
+[
+    "X-AI-RateLimit-Remaining-ai-proxy-openai: 500",
+    "X-AI-RateLimit-Remaining-ai-proxy-openai: 482",
+]
+--- no_error_log
+[error]
+
+
+
+=== TEST 17: set route with a bare name that is nested only
+--- config
+    location /t {
+        content_by_lua_block {
+            local t = require("lib.test_admin").test
+            local code, body = t('/apisix/admin/routes/1',
+                 ngx.HTTP_PUT,
+                 [[{
+                    "uri": "/v1/responses",
+                    "plugins": {
+                        "ai-proxy": {
+                            "provider": "openai",
+                            "auth": {
+                                "header": {
+                                    "Authorization": "Bearer test-key"
+                                }
+                            },
+                            "options": {
+                                "model": "gpt-4o-mini"
+                            },
+                            "override": {
+                                "endpoint": "http://127.0.0.1:1980"
+                            },
+                            "ssl_verify": false
+                        },
+                        "ai-rate-limiting": {
+                            "limit": 500,
+                            "time_window": 60,
+                            "limit_strategy": "expression",
+                            "cost_expr": "input_tokens + cached_tokens"
+                        }
+                    },
+                    "upstream": {
+                        "type": "roundrobin",
+                        "nodes": {
+                            "canbeanything.com": 1
+                        }
+                    }
+                }]]
+            )
+
+            if code >= 300 then
+                ngx.status = code
+            end
+            ngx.say(body)
+        }
+    }
+--- response_body
+passed
+
+
+
+=== TEST 18: nested fields are not bound by their bare name - cost = 40 + 0 = 40 per request
+--- pipelined_requests eval
+[
+    "POST /v1/responses\n" . '{"model":"gpt-4o-mini","stream":false,"input":"Hello"}',
+    "POST /v1/responses\n" . '{"model":"gpt-4o-mini","stream":false,"input":"Hello"}',
+]
+--- more_headers
+X-AI-Fixture: openai/responses-with-cache.json
+--- response_headers_like eval
+[
+    "X-AI-RateLimit-Remaining-ai-proxy-openai: 500",
+    "X-AI-RateLimit-Remaining-ai-proxy-openai: 460",
+]
+--- no_error_log
+[error]
+
+
+
+=== TEST 19: set route with a bare name present at the top level and nested
+--- config
+    location /t {
+        content_by_lua_block {
+            local t = require("lib.test_admin").test
+            local code, body = t('/apisix/admin/routes/1',
+                 ngx.HTTP_PUT,
+                 [[{
+                    "uri": "/v1/responses",
+                    "plugins": {
+                        "ai-proxy": {
+                            "provider": "openai",
+                            "auth": {
+                                "header": {
+                                    "Authorization": "Bearer test-key"
+                                }
+                            },
+                            "options": {
+                                "model": "gpt-4o-mini"
+                            },
+                            "override": {
+                                "endpoint": "http://127.0.0.1:1980"
+                            },
+                            "ssl_verify": false
+                        },
+                        "ai-rate-limiting": {
+                            "limit": 500,
+                            "time_window": 60,
+                            "limit_strategy": "expression",
+                            "cost_expr": "cached_tokens"
+                        }
+                    },
+                    "upstream": {
+                        "type": "roundrobin",
+                        "nodes": {
+                            "canbeanything.com": 1
+                        }
+                    }
+                }]]
+            )
+
+            if code >= 300 then
+                ngx.status = code
+            end
+            ngx.say(body)
+        }
+    }
+--- response_body
+passed
+
+
+
+=== TEST 20: bare name reads the top-level field - cost = 5 per request
+--- pipelined_requests eval
+[
+    "POST /v1/responses\n" . '{"model":"gpt-4o-mini","stream":false,"input":"Hello"}',
+    "POST /v1/responses\n" . '{"model":"gpt-4o-mini","stream":false,"input":"Hello"}',
+]
+--- more_headers
+X-AI-Fixture: openai/responses-usage-clash.json
+--- response_headers_like eval
+[
+    "X-AI-RateLimit-Remaining-ai-proxy-openai: 500",
+    "X-AI-RateLimit-Remaining-ai-proxy-openai: 495",
+]
+--- no_error_log
+[error]
+
+
+
+=== TEST 21: set route with the parent__child name of the same field
+--- config
+    location /t {
+        content_by_lua_block {
+            local t = require("lib.test_admin").test
+            local code, body = t('/apisix/admin/routes/1',
+                 ngx.HTTP_PUT,
+                 [[{
+                    "uri": "/v1/responses",
+                    "plugins": {
+                        "ai-proxy": {
+                            "provider": "openai",
+                            "auth": {
+                                "header": {
+                                    "Authorization": "Bearer test-key"
+                                }
+                            },
+                            "options": {
+                                "model": "gpt-4o-mini"
+                            },
+                            "override": {
+                                "endpoint": "http://127.0.0.1:1980"
+                            },
+                            "ssl_verify": false
+                        },
+                        "ai-rate-limiting": {
+                            "limit": 500,
+                            "time_window": 60,
+                            "limit_strategy": "expression",
+                            "cost_expr": "input_tokens_details__cached_tokens"
+                        }
+                    },
+                    "upstream": {
+                        "type": "roundrobin",
+                        "nodes": {
+                            "canbeanything.com": 1
+                        }
+                    }
+                }]]
+            )
+
+            if code >= 300 then
+                ngx.status = code
+            end
+            ngx.say(body)
+        }
+    }
+--- response_body
+passed
+
+
+
+=== TEST 22: parent__child reads the nested field - cost = 12 per request
+--- pipelined_requests eval
+[
+    "POST /v1/responses\n" . '{"model":"gpt-4o-mini","stream":false,"input":"Hello"}',
+    "POST /v1/responses\n" . '{"model":"gpt-4o-mini","stream":false,"input":"Hello"}',
+]
+--- more_headers
+X-AI-Fixture: openai/responses-usage-clash.json
+--- response_headers_like eval
+[
+    "X-AI-RateLimit-Remaining-ai-proxy-openai: 500",
+    "X-AI-RateLimit-Remaining-ai-proxy-openai: 488",
+]
+--- no_error_log
+[error]
+
+
+
+=== TEST 23: set route with a leaf name present in two sibling objects
+--- config
+    location /t {
+        content_by_lua_block {
+            local t = require("lib.test_admin").test
+            local code, body = t('/apisix/admin/routes/1',
+                 ngx.HTTP_PUT,
+                 [[{
+                    "uri": "/v1/chat/completions",
+                    "plugins": {
+                        "ai-proxy": {
+                            "provider": "openai",
+                            "auth": {
+                                "header": {
+                                    "Authorization": "Bearer test-key"
+                                }
+                            },
+                            "options": {
+                                "model": "gpt-4o-mini"
+                            },
+                            "override": {
+                                "endpoint": "http://127.0.0.1:1980"
+                            },
+                            "ssl_verify": false
+                        },
+                        "ai-rate-limiting": {
+                            "limit": 500,
+                            "time_window": 60,
+                            "limit_strategy": "expression",
+                            "cost_expr": "prompt_tokens_details__audio_tokens * 2 + completion_tokens_details__audio_tokens + audio_tokens"
+                        }
+                    },
+                    "upstream": {
+                        "type": "roundrobin",
+                        "nodes": {
+                            "canbeanything.com": 1
+                        }
+                    }
+                }]]
+            )
+
+            if code >= 300 then
+                ngx.status = code
+            end
+            ngx.say(body)
+        }
+    }
+--- response_body
+passed
+
+
+
+=== TEST 24: parent__child names pick each sibling field - cost = 30 * 2 + 70 + 0 = 130 per request
+--- pipelined_requests eval
+[
+    "POST /v1/chat/completions\n" . '{"model":"gpt-4o-mini","messages":[{"role":"user","content":"Hello"}]}',
+    "POST /v1/chat/completions\n" . '{"model":"gpt-4o-mini","messages":[{"role":"user","content":"Hello"}]}',
+]
+--- more_headers
+X-AI-Fixture: openai/chat-usage-audio.json
+--- response_headers_like eval
+[
+    "X-AI-RateLimit-Remaining-ai-proxy-openai: 500",
+    "X-AI-RateLimit-Remaining-ai-proxy-openai: 370",
+]
+--- no_error_log
+[error]
+
+
+
+=== TEST 25: set route with parent__child names that do not exist
+--- config
+    location /t {
+        content_by_lua_block {
+            local t = require("lib.test_admin").test
+            local code, body = t('/apisix/admin/routes/1',
+                 ngx.HTTP_PUT,
+                 [[{
+                    "uri": "/v1/chat/completions",
+                    "plugins": {
+                        "ai-proxy": {
+                            "provider": "openai",
+                            "auth": {
+                                "header": {
+                                    "Authorization": "Bearer test-key"
+                                }
+                            },
+                            "options": {
+                                "model": "gpt-4o-mini"
+                            },
+                            "override": {
+                                "endpoint": "http://127.0.0.1:1980"
+                            },
+                            "ssl_verify": false
+                        },
+                        "ai-rate-limiting": {
+                            "limit": 500,
+                            "time_window": 60,
+                            "limit_strategy": "expression",
+                            "cost_expr": "prompt_tokens + no_such_details__audio_tokens + prompt_tokens_details__no_such_field"
+                        }
+                    },
+                    "upstream": {
+                        "type": "roundrobin",
+                        "nodes": {
+                            "canbeanything.com": 1
+                        }
+                    }
+                }]]
+            )
+
+            if code >= 300 then
+                ngx.status = code
+            end
+            ngx.say(body)
+        }
+    }
+--- response_body
+passed
+
+
+
+=== TEST 26: missing parent__child names default to 0 - cost = 100 per request
+--- pipelined_requests eval
+[
+    "POST /v1/chat/completions\n" . '{"model":"gpt-4o-mini","messages":[{"role":"user","content":"Hello"}]}',
+    "POST /v1/chat/completions\n" . '{"model":"gpt-4o-mini","messages":[{"role":"user","content":"Hello"}]}',
+]
+--- more_headers
+X-AI-Fixture: openai/chat-usage-audio.json
+--- response_headers_like eval
+[
+    "X-AI-RateLimit-Remaining-ai-proxy-openai: 500",
+    "X-AI-RateLimit-Remaining-ai-proxy-openai: 400",
+]
+--- no_error_log
+[error]
+
+
+
+=== TEST 27: set route reading a field nested two levels deep
+--- config
+    location /t {
+        content_by_lua_block {
+            local t = require("lib.test_admin").test
+            local code, body = t('/apisix/admin/routes/1',
+                 ngx.HTTP_PUT,
+                 [[{
+                    "uri": "/v1/chat/completions",
+                    "plugins": {
+                        "ai-proxy": {
+                            "provider": "openai",
+                            "auth": {
+                                "header": {
+                                    "Authorization": "Bearer test-key"
+                                }
+                            },
+                            "options": {
+                                "model": "gpt-4o-mini"
+                            },
+                            "override": {
+                                "endpoint": "http://127.0.0.1:1980"
+                            },
+                            "ssl_verify": false
+                        },
+                        "ai-rate-limiting": {
+                            "limit": 500,
+                            "time_window": 60,
+                            "limit_strategy": "expression",
+                            "cost_expr": "prompt_tokens_details__cached_tokens_details__text_tokens + modality_details__text_tokens"
+                        }
+                    },
+                    "upstream": {
+                        "type": "roundrobin",
+                        "nodes": {
+                            "canbeanything.com": 1
+                        }
+                    }
+                }]]
+            )
+
+            if code >= 300 then
+                ngx.status = code
+            end
+            ngx.say(body)
+        }
+    }
+--- response_body
+passed
+
+
+
+=== TEST 28: deep fields join every level, arrays are skipped - cost = 9 + 0 = 9 per request
+--- pipelined_requests eval
+[
+    "POST /v1/chat/completions\n" . '{"model":"gpt-4o-mini","messages":[{"role":"user","content":"Hello"}]}',
+    "POST /v1/chat/completions\n" . '{"model":"gpt-4o-mini","messages":[{"role":"user","content":"Hello"}]}',
+]
+--- more_headers
+X-AI-Fixture: openai/chat-usage-deep.json
+--- response_headers_like eval
+[
+    "X-AI-RateLimit-Remaining-ai-proxy-openai: 500",
+    "X-AI-RateLimit-Remaining-ai-proxy-openai: 491",
+]
+--- no_error_log
+[error]