diff --git a/.requirements b/.requirements index f633665c68fd..a26f9999dfc6 100644 --- a/.requirements +++ b/.requirements @@ -17,5 +17,5 @@ APISIX_PACKAGE_NAME=apisix -APISIX_RUNTIME=1.3.18 +APISIX_RUNTIME=1.3.19 APISIX_DASHBOARD_COMMIT=fa2fd0f60f8afffb096476333b9ba63b4c518fa3 diff --git a/apisix/plugins/prometheus/exporter.lua b/apisix/plugins/prometheus/exporter.lua index fbb600e18687..7fe666c7a7b5 100644 --- a/apisix/plugins/prometheus/exporter.lua +++ b/apisix/plugins/prometheus/exporter.lua @@ -238,25 +238,29 @@ local _M = { local function init_stream_metrics() + -- service and service_id follow the http metrics: both hold the id unless + -- prefer_name puts the name in service, and both are empty for a session + -- that never got as far as a route with a service metrics.stream_connection_total = prometheus:counter("stream_connection_total", "Total number of connections handled per stream route in APISIX", - {"route"}) + {"route", "service", "service_id"}) -- Keyed by listen_addr rather than by route: a session can end before any -- stream route is matched, and the byte counters come from nginx, which - -- only knows the listening address. + -- only knows the listening address. service and service_id split that + -- total further. metrics.stream_active_connections = prometheus:gauge( "stream_active_connections", - "Number of stream sessions currently being proxied per listening address", - {"listen_addr"}) + "Number of stream sessions currently being proxied per listening address and service", + {"listen_addr", "service", "service_id"}) metrics.stream_status = prometheus:counter("stream_status", "Stream sessions per termination status in APISIX", - {"code", "listen_addr", "node"}) + {"code", "listen_addr", "service", "service_id", "node"}) metrics.stream_bandwidth = prometheus:counter("stream_bandwidth", "Total bandwidth in bytes proxied by the stream subsystem in APISIX", - {"listen_addr", "type", "side"}) + {"listen_addr", "service", "service_id", "type", "side"}) xrpc.init_metrics(prometheus) end @@ -317,6 +321,48 @@ local STREAM_PUBLISHED_PREFIX = "stream_bytes_published:" local STREAM_PUBLISH_LOCK = "stream_bytes_publishing" local STREAM_PUBLISH_LOCK_TTL = 10 + +-- The same rule as the http metrics: the id, or the name with prefer_name. +-- Resolved once per session: preread labels the zone with it and the log +-- phase reuses it, so all four metrics name a session the same way even if +-- its service is renamed while it is open. +local function stream_service_labels(conf, ctx) + local labels = ctx.prometheus_stream_service + if labels then + return labels[1], labels[2] + end + + local service, service_id = "", "" + local id = ctx.service_id + local name = ctx.service_name + if not id then + -- a route with an upstream_id keeps its service_id without the service + -- being merged in, and the http metrics still label it with it + id = ctx.matched_route.value.service_id + if id then + local fetched = conf.prefer_name == true and service_fetch(id) + name = fetched and fetched.value.name + end + end + + if id then + service_id = tostring(id) + service = conf.prefer_name == true and name or service_id + end + + ctx.prometheus_stream_service = {service, service_id} + return service, service_id +end + + +-- The zone is read where neither the session nor the plugin conf is at hand, +-- so a session carries its label values into the zone itself; "" on a slot +-- that was never labelled. +local function stream_zone_labels(entry) + local labels = entry.labels + return labels[1] or "", labels[2] or "" +end + local stream_metrics_lib local stream_metrics_lib_checked = false local stream_zone_unavailable = false @@ -343,16 +389,19 @@ local function stream_metrics_zone() end -local function publish_stream_bytes(dict, listen_addr, direction, total) +local function publish_stream_bytes(dict, listen_addr, service, service_id, direction, + total) local field = direction[1] if type(total) ~= "number" then core.log.error("stream metrics zone reported no ", field, " for ", - listen_addr) + listen_addr, " service ", service_id) return end - local key = STREAM_PUBLISHED_PREFIX .. listen_addr .. ":" .. field + -- a service name can hold any character but this one + local key = STREAM_PUBLISHED_PREFIX .. listen_addr .. "\31" .. service .. "\31" + .. service_id .. "\31" .. field local published = dict:get(key) if not published then @@ -384,7 +433,7 @@ local function publish_stream_bytes(dict, listen_addr, direction, total) -- rebaselines rather than emitting a negative delta if total > published then metrics.stream_bandwidth:inc(total - published, - gen_arr(listen_addr, direction[2], direction[3])) + gen_arr(listen_addr, service, service_id, direction[2], direction[3])) end end @@ -430,14 +479,17 @@ local function collect_stream_zone_metrics() for _, entry in ipairs(entries) do local listen_addr = entry.listen_addr + local service, service_id = stream_zone_labels(entry) -- the gauge carries no baseline and every reader writes the same -- value, so it is published whether or not this one took the lock - metrics.stream_active_connections:set(entry.active, gen_arr(listen_addr)) + metrics.stream_active_connections:set(entry.active, + gen_arr(listen_addr, service, service_id)) if publishing then for _, direction in ipairs(STREAM_BANDWIDTH_DIRECTIONS) do - publish_stream_bytes(dict, listen_addr, direction, entry[direction[1]]) + publish_stream_bytes(dict, listen_addr, service, service_id, direction, + entry[direction[1]]) end end end @@ -944,7 +996,10 @@ function _M.stream_log(conf, ctx) end end - metrics.stream_connection_total:inc(1, gen_arr(route_id)) + -- empty when the session ended before a route with a service matched + local service, service_id = stream_service_labels(conf, ctx) + + metrics.stream_connection_total:inc(1, gen_arr(route_id, service, service_id)) -- empty when the session ended before a node was picked local node = "" @@ -953,7 +1008,41 @@ function _M.stream_log(conf, ctx) end metrics.stream_status:inc(1, gen_arr(stream_status_code(ctx), - stream_listen_addr(ctx), node)) + stream_listen_addr(ctx), service, service_id, node)) +end + + +-- logged once per worker, it would otherwise repeat on every session +local stream_label_error_logged = false + + +-- Labels the session in the metrics zone as it starts, so that its active +-- count and every byte it moves from now on are accounted under its service +-- while it is still open. A session that never gets here -- it ended before +-- routing, its route has no service, or a plugin running before this one +-- rejected it -- stays in the unlabelled total of its listen_addr. +function _M.stream_preread(conf, ctx) + local service, service_id = stream_service_labels(conf, ctx) + if service_id == "" then + return + end + + local lib = stream_metrics_zone() + if not lib then + return + end + + -- "not accounted" is a listening address the zone does not count, such as + -- a unix socket. Anything else -- a full zone, or labels too long for it -- + -- leaves the session in the unlabelled total, while apisix_stream_status + -- still has its service. + local ok, err = lib.set_labels({service, service_id}) + if not ok and err ~= "not accounted" and not stream_label_error_logged then + stream_label_error_logged = true + core.log.warn("failed to label stream sessions, first seen on service ", + service_id, ", they are only counted in the listen_addr ", + "total: ", err) + end end diff --git a/apisix/stream/plugins/prometheus.lua b/apisix/stream/plugins/prometheus.lua index 6b30ca2528ab..c7fa9fb0690c 100644 --- a/apisix/stream/plugins/prometheus.lua +++ b/apisix/stream/plugins/prometheus.lua @@ -34,6 +34,7 @@ local _M = { version = 0.1, priority = 500, name = plugin_name, + preread = exporter.stream_preread, log = exporter.stream_log, destroy = exporter.destroy, init = exporter.stream_init, diff --git a/ci/linux-install-openresty.sh b/ci/linux-install-openresty.sh index c02fd180d151..51bdf0fae722 100755 --- a/ci/linux-install-openresty.sh +++ b/ci/linux-install-openresty.sh @@ -61,7 +61,7 @@ else sudo apt-get -y update --fix-missing sudo apt-get install -y build-essential gcc g++ cpanminus libxml2-dev libxslt-dev - if [ "$APISIX_RUNTIME" != "1.3.18" ]; then + if [ "$APISIX_RUNTIME" != "1.3.19" ]; then echo "Please update the apisix-runtime-debug checksum for APISIX_RUNTIME=$APISIX_RUNTIME" >&2 exit 1 fi @@ -69,11 +69,11 @@ else case "$ARCH" in x86_64|amd64) DEB_ARCH="amd64" - EXPECTED_SHA256="0d7cbe27cd0303c6f6b3cad27338a697cf2e2e809ceeac9a57b2d1fb78a3a20e" + EXPECTED_SHA256="0f350a142c915a70d4501d68259a1506ffdd8efd6a1464802c8935b1f36fdf3b" ;; arm64|aarch64) DEB_ARCH="arm64" - EXPECTED_SHA256="a7cf5040837e4d34f456e97ff9b858ce64bf95e3b2f647f5d3021de5a9770741" + EXPECTED_SHA256="67fc887818f2899874e26fc539b9f49c78c7f8e73aed495683df9b694250db54" ;; *) echo "Unsupported architecture: $ARCH" >&2 diff --git a/docs/en/latest/plugins/prometheus.md b/docs/en/latest/plugins/prometheus.md index 95d89f02fedc..2b183020ea6c 100644 --- a/docs/en/latest/plugins/prometheus.md +++ b/docs/en/latest/plugins/prometheus.md @@ -202,6 +202,20 @@ The following labels are used to differentiate `apisix_http_status` metrics. | llm_model | Effective target model for the AI request. Uses the model configured on the AI instance when present; otherwise uses the model requested by the client. Empty for traditional HTTP traffic. | | response_source | Response origin: `apisix` for responses generated by APISIX, `nginx` for NGINX proxy errors, or `upstream` for responses received from the Upstream. | +### Service labels on the Stream metrics + +All four Stream metrics carry `service` and `service_id`, following the same rule as the HTTP metrics: `service_id` is the ID of the Service the session's Stream Route belongs to, and `service` is that ID too, or the Service's name when `prefer_name` is `true` on the Route's `prometheus` Plugin. Both are empty for a session that never reached a Stream Route with a Service, for example one that failed its TLS handshake or matched no Route. Such sessions are only counted in the per `listen_addr` total, so summing over `service` gives the same totals as before. + +`apisix_stream_active_connections` and `apisix_stream_bandwidth` are split per Service while sessions are still open: the Stream `prometheus` Plugin labels each session with its Service when it starts, and from then on its active count and the bytes it moves are accounted under that Service. + +### Labels for `apisix_stream_connection_total` + +| Name | Description | +| ----------- | --------------------------------------------------------------------------------------- | +| route | ID of the matched Stream Route, or its name when `prefer_name` is `true`. | +| service | ID of the Service of the matched Stream Route when `prefer_name` is `false` (default), and name of the Service when `prefer_name` is `true`. Empty when the session did not reach a Stream Route that belongs to a Service. | +| service_id | ID of the Service of the matched Stream Route. Empty when the session did not reach a Stream Route that belongs to a Service. | + ### Labels for `apisix_stream_active_connections` The gauge is incremented when a session is accepted and decremented when it @@ -210,6 +224,8 @@ ends, so it reflects live concurrency without waiting for sessions to finish. | Name | Description | | ----------- | --------------------------------------------------------------------------------------- | | listen_addr | Listening address the client connected to, for example `0.0.0.0:9100`. | +| service | ID of the Service of the matched Stream Route when `prefer_name` is `false` (default), and name of the Service when `prefer_name` is `true`. Empty when the session did not reach a Stream Route that belongs to a Service. | +| service_id | ID of the Service of the matched Stream Route. Empty when the session did not reach a Stream Route that belongs to a Service. | ### Labels for `apisix_stream_status` @@ -225,6 +241,8 @@ clean close. No synthetic code is introduced. | ----------- | --------------------------------------------------------------------------------------- | | code | How the session ended: `200` for a normal close, worker shutdown, or a missing or unrecognized termination reason; `400` for a client-side problem such as a reset or invalid preread data; `403` when rejected by an access rule; `500` for an internal error; `502` for an upstream or transport problem such as a connect failure, reset, or idle timeout; `503` when rejected by a connection limit. | | listen_addr | Listening address the client connected to, for example `0.0.0.0:9100`. | +| service | ID of the Service of the matched Stream Route when `prefer_name` is `false` (default), and name of the Service when `prefer_name` is `true`. Empty when the session did not reach a Stream Route that belongs to a Service. | +| service_id | ID of the Service of the matched Stream Route. Empty when the session did not reach a Stream Route that belongs to a Service. | | node | Address of the upstream node used, empty when no node was selected. | For UDP only a subset of the codes occurs, since UDP has no close, FIN or @@ -239,6 +257,8 @@ counted; the HTTP subsystem cannot contribute to it. | Name | Description | | ----------- | --------------------------------------------------------------------------------------- | | listen_addr | Listening address the client connected to, for example `0.0.0.0:9100`. | +| service | ID of the Service of the matched Stream Route when `prefer_name` is `false` (default), and name of the Service when `prefer_name` is `true`. Empty when the session did not reach a Stream Route that belongs to a Service. | +| service_id | ID of the Service of the matched Stream Route. Empty when the session did not reach a Stream Route that belongs to a Service. | | side | Which connection the bytes crossed: `downstream` between APISIX and the client, `upstream` between APISIX and the upstream. | | type | Direction relative to APISIX, matching `apisix_bandwidth`: `ingress` for bytes APISIX received, `egress` for bytes APISIX sent. | @@ -763,28 +783,31 @@ You should see an output similar to the following: ```text # HELP apisix_stream_connection_total Total number of connections handled per Stream Route in APISIX # TYPE apisix_stream_connection_total counter -apisix_stream_connection_total{route="prometheus-route"} 1 -# HELP apisix_stream_active_connections Number of stream sessions currently being proxied per listening address +apisix_stream_connection_total{route="prometheus-route",service="",service_id=""} 1 +# HELP apisix_stream_active_connections Number of stream sessions currently being proxied per listening address and service # TYPE apisix_stream_active_connections gauge -apisix_stream_active_connections{listen_addr="0.0.0.0:9100"} 0 +apisix_stream_active_connections{listen_addr="0.0.0.0:9100",service="",service_id=""} 0 # HELP apisix_stream_status Stream sessions per termination status in APISIX # TYPE apisix_stream_status counter -apisix_stream_status{code="200",listen_addr="0.0.0.0:9100",node="54.237.103.220:80"} 1 +apisix_stream_status{code="200",listen_addr="0.0.0.0:9100",service="",service_id="",node="54.237.103.220:80"} 1 # HELP apisix_stream_bandwidth Total bandwidth in bytes proxied by the stream subsystem in APISIX # TYPE apisix_stream_bandwidth counter -apisix_stream_bandwidth{listen_addr="0.0.0.0:9100",type="ingress",side="downstream"} 78 -apisix_stream_bandwidth{listen_addr="0.0.0.0:9100",type="egress",side="downstream"} 219 -apisix_stream_bandwidth{listen_addr="0.0.0.0:9100",type="egress",side="upstream"} 78 -apisix_stream_bandwidth{listen_addr="0.0.0.0:9100",type="ingress",side="upstream"} 219 +apisix_stream_bandwidth{listen_addr="0.0.0.0:9100",service="",service_id="",type="ingress",side="downstream"} 78 +apisix_stream_bandwidth{listen_addr="0.0.0.0:9100",service="",service_id="",type="egress",side="downstream"} 219 +apisix_stream_bandwidth{listen_addr="0.0.0.0:9100",service="",service_id="",type="egress",side="upstream"} 78 +apisix_stream_bandwidth{listen_addr="0.0.0.0:9100",service="",service_id="",type="ingress",side="upstream"} 219 ``` -The exact Upstream address and byte counts depend on the request. The active-connections gauge is `0` above because the request completed before the scrape; scrape while a connection remains open to observe a positive value. +The Stream Route in this example does not belong to a Service, so `service` and `service_id` are empty. The exact Upstream address and byte counts depend on the request. The active-connections gauge is `0` above because the request completed before the scrape; scrape while a connection remains open to observe a positive value. :::note `apisix_stream_active_connections` and `apisix_stream_bandwidth` are backed by an NGINX shared memory zone, sized by `nginx_config.stream.metrics_zone_size` (default `1m`). They require APISIX-Runtime; on a runtime without it the two -metrics are simply not published. +metrics are simply not published. The zone holds one slot per listening address +plus one per Service seen on it, about 760 slots for `1m`; at most three +quarters of them go to Services. When they run out, sessions of a new Service +are only counted in their `listen_addr` total and a warning is logged. ::: diff --git a/docs/zh/latest/plugins/prometheus.md b/docs/zh/latest/plugins/prometheus.md index 86081f9ffb91..ce772bfa7fb5 100644 --- a/docs/zh/latest/plugins/prometheus.md +++ b/docs/zh/latest/plugins/prometheus.md @@ -202,6 +202,20 @@ Prometheus 中有不同类型的指标。要了解它们之间的区别,请参 | llm_model | AI 请求实际使用的目标模型。优先使用 AI 实例中配置的模型,否则使用客户端请求的模型;传统 HTTP 流量中为空字符串。 | | response_source | 响应来源:`apisix` 表示由 APISIX 生成,`nginx` 表示 NGINX 代理错误,`upstream` 表示来自上游服务的响应。 | +### Stream 指标的 Service 标签 + +四个 Stream 指标都带有 `service` 和 `service_id` 标签,取值规则与 HTTP 指标相同:`service_id` 为会话所匹配 Stream Route 所属 Service 的 ID;`service` 默认也是该 ID,当 Route 上 `prometheus` 插件的 `prefer_name` 为 `true` 时为 Service 的名称。会话未到达属于某个 Service 的 Stream Route 时(例如 TLS 握手失败或未匹配到任何 Route),两个标签均为空,这类会话只计入对应 `listen_addr` 的总量,因此按 `service` 求和的结果与此前的总量一致。 + +`apisix_stream_active_connections` 和 `apisix_stream_bandwidth` 在会话进行中即按 Service 拆分:Stream `prometheus` 插件在会话开始时为其标记所属 Service,此后该会话的活跃计数和传输的字节都计入该 Service。 + +### `apisix_stream_connection_total` 的标签 + +| 名称 | 描述 | +| --- | --- | +| route | 匹配的 Stream Route 的 ID;`prefer_name` 为 `true` 时为其名称。 | +| service | `prefer_name` 为 `false`(默认)时为匹配的 Stream Route 所属 Service 的 ID,为 `true` 时为该 Service 的名称。会话未到达属于某个 Service 的 Stream Route 时为空。 | +| service_id | 匹配的 Stream Route 所属 Service 的 ID。会话未到达属于某个 Service 的 Stream Route 时为空。 | + ### `apisix_stream_active_connections` 的标签 接受会话时,该 gauge 会递增;会话结束时递减,因此无需等到会话结束即可反映实时并发量。 @@ -209,6 +223,8 @@ Prometheus 中有不同类型的指标。要了解它们之间的区别,请参 | 名称 | 描述 | | --- | --- | | listen_addr | 客户端连接的监听地址,例如 `0.0.0.0:9100`。 | +| service | `prefer_name` 为 `false`(默认)时为匹配的 Stream Route 所属 Service 的 ID,为 `true` 时为该 Service 的名称。会话未到达属于某个 Service 的 Stream Route 时为空。 | +| service_id | 匹配的 Stream Route 所属 Service 的 ID。会话未到达属于某个 Service 的 Stream Route 时为空。 | ### `apisix_stream_status` 的标签 @@ -218,6 +234,8 @@ Prometheus 中有不同类型的指标。要了解它们之间的区别,请参 | --- | --- | | code | 会话结束方式:`200` 表示正常关闭、worker 关闭,或终止原因缺失或无法识别;`400` 表示客户端重置或预读数据无效等客户端问题;`403` 表示被访问规则拒绝;`500` 表示内部错误;`502` 表示连接失败、重置或空闲超时等上游或传输问题;`503` 表示被连接数限制拒绝。 | | listen_addr | 客户端连接的监听地址,例如 `0.0.0.0:9100`。 | +| service | `prefer_name` 为 `false`(默认)时为匹配的 Stream Route 所属 Service 的 ID,为 `true` 时为该 Service 的名称。会话未到达属于某个 Service 的 Stream Route 时为空。 | +| service_id | 匹配的 Stream Route 所属 Service 的 ID。会话未到达属于某个 Service 的 Stream Route 时为空。 | | node | 使用的上游节点地址;未选择节点时为空。 | UDP 没有关闭、FIN 或重置信号,因此只会出现其中一部分状态代码。 @@ -229,6 +247,8 @@ UDP 没有关闭、FIN 或重置信号,因此只会出现其中一部分状态 | 名称 | 描述 | | --- | --- | | listen_addr | 客户端连接的监听地址,例如 `0.0.0.0:9100`。 | +| service | `prefer_name` 为 `false`(默认)时为匹配的 Stream Route 所属 Service 的 ID,为 `true` 时为该 Service 的名称。会话未到达属于某个 Service 的 Stream Route 时为空。 | +| service_id | 匹配的 Stream Route 所属 Service 的 ID。会话未到达属于某个 Service 的 Stream Route 时为空。 | | side | 字节经过的连接侧:`downstream` 表示 APISIX 与客户端之间,`upstream` 表示 APISIX 与上游之间。 | | type | 相对 APISIX 的方向,与 `apisix_bandwidth` 一致:`ingress` 表示 APISIX 接收的字节,`egress` 表示 APISIX 发送的字节。 | @@ -750,25 +770,25 @@ curl "http://127.0.0.1:9091/apisix/prometheus/metrics" ```text # HELP apisix_stream_connection_total APISIX 中每个 Stream Route 处理的总连接数 # TYPE apisix_stream_connection_total counter -apisix_stream_connection_total{route="prometheus-route"} 1 -# HELP apisix_stream_active_connections Number of stream sessions currently being proxied per listening address +apisix_stream_connection_total{route="prometheus-route",service="",service_id=""} 1 +# HELP apisix_stream_active_connections Number of stream sessions currently being proxied per listening address and service # TYPE apisix_stream_active_connections gauge -apisix_stream_active_connections{listen_addr="0.0.0.0:9100"} 0 +apisix_stream_active_connections{listen_addr="0.0.0.0:9100",service="",service_id=""} 0 # HELP apisix_stream_status Stream sessions per termination status in APISIX # TYPE apisix_stream_status counter -apisix_stream_status{code="200",listen_addr="0.0.0.0:9100",node="54.237.103.220:80"} 1 +apisix_stream_status{code="200",listen_addr="0.0.0.0:9100",service="",service_id="",node="54.237.103.220:80"} 1 # HELP apisix_stream_bandwidth Total bandwidth in bytes proxied by the stream subsystem in APISIX # TYPE apisix_stream_bandwidth counter -apisix_stream_bandwidth{listen_addr="0.0.0.0:9100",type="ingress",side="downstream"} 78 -apisix_stream_bandwidth{listen_addr="0.0.0.0:9100",type="egress",side="downstream"} 219 -apisix_stream_bandwidth{listen_addr="0.0.0.0:9100",type="egress",side="upstream"} 78 -apisix_stream_bandwidth{listen_addr="0.0.0.0:9100",type="ingress",side="upstream"} 219 +apisix_stream_bandwidth{listen_addr="0.0.0.0:9100",service="",service_id="",type="ingress",side="downstream"} 78 +apisix_stream_bandwidth{listen_addr="0.0.0.0:9100",service="",service_id="",type="egress",side="downstream"} 219 +apisix_stream_bandwidth{listen_addr="0.0.0.0:9100",service="",service_id="",type="egress",side="upstream"} 78 +apisix_stream_bandwidth{listen_addr="0.0.0.0:9100",service="",service_id="",type="ingress",side="upstream"} 219 ``` -实际的上游地址和字节数取决于请求。上例中的活跃连接 gauge 为 `0`,因为抓取指标时请求已完成;如需观察正值,请在连接保持打开时抓取指标。 +本例中的 Stream Route 不属于任何 Service,因此 `service` 和 `service_id` 为空。实际的上游地址和字节数取决于请求。上例中的活跃连接 gauge 为 `0`,因为抓取指标时请求已完成;如需观察正值,请在连接保持打开时抓取指标。 :::note -`apisix_stream_active_connections` 和 `apisix_stream_bandwidth` 使用由 `nginx_config.stream.metrics_zone_size` 配置的 NGINX 共享内存区,默认大小为 `1m`。这两个指标依赖 APISIX-Runtime;如果运行时不提供对应模块,则不会发布这两个指标。 +`apisix_stream_active_connections` 和 `apisix_stream_bandwidth` 使用由 `nginx_config.stream.metrics_zone_size` 配置的 NGINX 共享内存区,默认大小为 `1m`。这两个指标依赖 APISIX-Runtime;如果运行时不提供对应模块,则不会发布这两个指标。该共享内存区为每个监听地址及其上出现的每个 Service 各分配一个 slot,`1m` 约可容纳 760 个 slot,其中最多四分之三分配给 Service。slot 用尽后,新 Service 的会话只计入其 `listen_addr` 的总量,并记录一条警告日志。 ::: diff --git a/t/cli/test_prometheus_stream.sh b/t/cli/test_prometheus_stream.sh index 72fb96759b72..fcbfb5a2eec3 100755 --- a/t/cli/test_prometheus_stream.sh +++ b/t/cli/test_prometheus_stream.sh @@ -64,7 +64,7 @@ deadline=$(( $(date +%s) + 20 )) while [ "$(date +%s)" -lt "$deadline" ]; do curl -s --connect-timeout 1 --max-time 2 http://127.0.0.1:9100 >/dev/null 2>&1 || true if curl -s --connect-timeout 1 --max-time 2 http://127.0.0.1:9091/apisix/prometheus/metrics \ - | grep -qE 'apisix_stream_connection_total\{route="1"\} [1-9][0-9]*'; then + | grep -qE 'apisix_stream_connection_total\{route="1",service="",service_id=""\} [1-9][0-9]*'; then ok=1 break fi @@ -106,7 +106,7 @@ out="" while [ "$(date +%s)" -lt "$deadline" ]; do curl -s --connect-timeout 1 --max-time 2 http://127.0.0.1:9100 >/dev/null 2>&1 || true out="$(curl -s --connect-timeout 1 --max-time 2 http://127.0.0.1:9091/apisix/prometheus/metrics || true)" - if echo "$out" | grep -qE 'apisix_stream_connection_total\{route="1"\} [1-9][0-9]*'; then + if echo "$out" | grep -qE 'apisix_stream_connection_total\{route="1",service="",service_id=""\} [1-9][0-9]*'; then ok=1 break fi diff --git a/t/stream-plugin/prometheus-metrics-service.t b/t/stream-plugin/prometheus-metrics-service.t new file mode 100644 index 000000000000..5de47102a4af --- /dev/null +++ b/t/stream-plugin/prometheus-metrics-service.t @@ -0,0 +1,580 @@ +# +# Licensed to the Apache Software Foundation (ASF) under one or more +# contributor license agreements. See the NOTICE file distributed with +# this work for additional information regarding copyright ownership. +# The ASF licenses this file to You under the Apache License, Version 2.0 +# (the "License"); you may not use this file except in compliance with +# the License. You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# +BEGIN { + if ($ENV{TEST_NGINX_CHECK_LEAK}) { + $SkipReason = "unavailable for the hup tests"; + + } else { + $ENV{TEST_NGINX_USE_HUP} = 1; + undef $ENV{TEST_NGINX_USE_STAP}; + } +} + +use t::APISIX 'no_plan'; + +repeat_each(1); +no_long_string(); +no_shuffle(); +no_root_location(); + +add_block_preprocessor(sub { + my ($block) = @_; + + # the endpoint serves a cache the privileged agent refills on this + # interval, see t/stream-plugin/prometheus-metrics.t + my $extra_yaml_config = <<_EOC_; +stream_plugins: + - prometheus +plugin_attr: + prometheus: + refresh_interval: 0.5 +_EOC_ + + $block->set_value("extra_yaml_config", $extra_yaml_config); + + if (!defined $block->request) { + $block->set_value("request", "GET /t"); + } + + # see t/stream-plugin/prometheus-metrics.t: a scrape without the stream + # block has no zone to read, and the reload into it drops the zone + if ($block->request =~ m{/apisix/prometheus/metrics}) { + $block->set_value("stream_enable", 1); + } + + # an upstream that answers and then holds the session open, so that a + # scrape can observe it while it is live. Only for the probes: a stream + # block makes Test::Nginx install its own `location = /t`. + if ($block->request =~ m{/probe}) { + my $extra_stream_config = <<_EOC_; +server { + listen 1993; + content_by_lua_block { + local sock = ngx.req.socket() + sock:receive("1") + ngx.say("hello world") + ngx.flush(true) + ngx.sleep(10) + } +} +_EOC_ + + $block->set_value("extra_stream_config", $extra_stream_config); + } + + # The scrapes go over a raw socket: capturing into an APISIX route leaves + # the upstream connect without a usable api_ctx. + my $extra_init_by_lua = <<_EOC_; + function _G.scrape() + -- let the privileged agent refill the cache with what happened so far + ngx.sleep(1.2) + + local sock = ngx.socket.tcp() + local ok, err = sock:connect("127.0.0.1", 1984) + if not ok then + return nil, "scrape connect: " .. err + end + + ok, err = sock:send("GET /apisix/prometheus/metrics HTTP/1.0\\r\\n" + .. "Host: 127.0.0.1\\r\\n\\r\\n") + if not ok then + return nil, "scrape send: " .. err + end + + local body, rerr, partial = sock:receive("*a") + sock:close() + body = body or partial + if not body then + return nil, "scrape: " .. rerr + end + + return body + end + + -- Opens a session on 1985 and scrapes while it is live. A scrape runs + -- first so that the bandwidth baselines are taken before the session + -- exists: the first read of the zone only baselines what it finds. + function _G.scrape_live_session() + local body, err = scrape() + if not body then + return nil, err + end + + local sock = ngx.socket.tcp() + local ok + ok, err = sock:connect("127.0.0.1", 1985) + if not ok then + return nil, "connect: " .. err + end + + local bytes + bytes, err = sock:send("hello") + if not bytes then + return nil, "send: " .. err + end + + local line + line, err = sock:receive("*l") + if not line then + return nil, "receive: " .. err + end + + body, err = scrape() + sock:close() + return body, err + end + + -- the value of the series whose labels are exactly `labels` + function _G.series_value(body, name, labels) + local prefix = name .. "{" .. labels .. "} " + for line in body:gmatch("[^\\n]+") do + if line:sub(1, #prefix) == prefix then + return line:sub(#prefix + 1) + end + end + return "no-series" + end + + function _G.svc_a_ingress(body) + return series_value(body, "apisix_stream_bandwidth", + 'listen_addr="0.0.0.0:1985",service="svc-a",service_id="svc-a",' + .. 'type="ingress",side="downstream"') + end + + function _G.active_on(body, service, service_id) + return series_value(body, "apisix_stream_active_connections", + 'listen_addr="0.0.0.0:1985",service="' .. service + .. '",service_id="' .. service_id .. '"') + end +_EOC_ + + $block->set_value("extra_init_by_lua", $extra_init_by_lua); +}); + +run_tests; + +__DATA__ + +=== TEST 1: pre-create the metrics endpoint, a service and a stream route under it +--- config + location /t { + content_by_lua_block { + local data = { + { + url = "/apisix/admin/routes/metrics", + data = [[{ + "plugins": { + "public-api": {} + }, + "uri": "/apisix/prometheus/metrics" + }]] + }, + { + url = "/apisix/admin/services/svc-a", + data = [[{ + "upstream": { + "type": "roundrobin", + "nodes": [{ + "host": "127.0.0.1", + "port": 1993, + "weight": 1 + }] + } + }]] + }, + { + url = "/apisix/admin/stream_routes/1", + data = [[{ + "plugins": { + "prometheus": {} + }, + "service_id": "svc-a" + }]] + } + } + + local t = require("lib.test_admin").test + + for _, data in ipairs(data) do + local code, body = t(data.url, ngx.HTTP_PUT, data.data) + if code > 300 then + ngx.say(body) + return + end + end + } + } +--- response_body + + + +=== TEST 2: a live session is split out under its service_id +The session is labelled in preread, so the service series carries it and the +unlabelled series of the same listen_addr does not. Like any zone slot, the +service's slot only takes a baseline the first time it is read, so its bytes +are counted from its second session on. +--- config + location /probe { + content_by_lua_block { + -- the route from TEST 1 has to reach the stream workers first + ngx.sleep(1.5) + + local body, err = scrape_live_session() + if not body then + ngx.say(err) + return + end + ngx.say("first session bandwidth=", svc_a_ingress(body)) + + body, err = scrape_live_session() + if not body then + ngx.say(err) + return + end + + ngx.say("service live=", active_on(body, "svc-a", "svc-a")) + ngx.say("unlabelled live=", active_on(body, "", "")) + + local bw = tonumber(svc_a_ingress(body)) + ngx.say("service bandwidth=", bw and tonumber(bw) > 0 and "counted" + or bw or "no-series") + + ngx.sleep(1.5) + } + } +--- request +GET /probe +--- stream_enable +--- timeout: 20 +--- response_body +first session bandwidth=no-series +service live=1 +unlabelled live=0 +service bandwidth=counted + + + +=== TEST 3: once the session is gone its service gauge is back to zero +--- request +GET /apisix/prometheus/metrics +--- response_body_like eval +qr/apisix_stream_active_connections\{listen_addr="0\.0\.0\.0:1985",service="svc-a",service_id="svc-a"\} 0$/m +--- no_error_log +[error] + + + +=== TEST 4: the service carries the gateway to upstream bytes +--- request +GET /apisix/prometheus/metrics +--- response_body_like eval +qr/apisix_stream_bandwidth\{listen_addr="0\.0\.0\.0:1985",service="svc-a",service_id="svc-a",type="egress",side="upstream"\} [1-9]\d*/ +--- no_error_log +[error] + + + +=== TEST 5: the service carries the upstream to gateway bytes +--- request +GET /apisix/prometheus/metrics +--- response_body_like eval +qr/apisix_stream_bandwidth\{listen_addr="0\.0\.0\.0:1985",service="svc-a",service_id="svc-a",type="ingress",side="upstream"\} [1-9]\d*/ +--- no_error_log +[error] + + + +=== TEST 6: a route without a service keeps its sessions in the unlabelled total +--- config + location /probe { + content_by_lua_block { + local t = require("lib.test_admin").test + local code = t("/apisix/admin/stream_routes/1", ngx.HTTP_PUT, [[{ + "plugins": { + "prometheus": {} + }, + "upstream": { + "type": "roundrobin", + "nodes": [{ + "host": "127.0.0.1", + "port": 1993, + "weight": 1 + }] + } + }]]) + if code > 300 then + ngx.say("route: ", code) + return + end + + ngx.sleep(1.5) + + local body, err = scrape_live_session() + if not body then + ngx.say(err) + return + end + + ngx.say("service live=", active_on(body, "svc-a", "svc-a")) + ngx.say("unlabelled live=", active_on(body, "", "")) + + ngx.sleep(1.5) + } + } +--- request +GET /probe +--- stream_enable +--- timeout: 20 +--- response_body +service live=0 +unlabelled live=1 + + + +=== TEST 7: with prefer_name the service label carries the name +The session is labelled when it starts, so the live session is already +reported under the name. The name is its own zone slot: the gauge left under +the id label by the earlier cases stays, at 0. +--- config + location /probe { + content_by_lua_block { + local t = require("lib.test_admin").test + local code = t("/apisix/admin/services/svc-a", ngx.HTTP_PUT, [[{ + "name": "Order TCP", + "upstream": { + "type": "roundrobin", + "nodes": [{ + "host": "127.0.0.1", + "port": 1993, + "weight": 1 + }] + } + }]]) + if code > 300 then + ngx.say("service: ", code) + return + end + + code = t("/apisix/admin/stream_routes/1", ngx.HTTP_PUT, [[{ + "plugins": { + "prometheus": { + "prefer_name": true + } + }, + "service_id": "svc-a" + }]]) + if code > 300 then + ngx.say("route: ", code) + return + end + + ngx.sleep(1.5) + + -- the first session under the name only baselines its new slot + local body, err = scrape_live_session() + if not body then + ngx.say(err) + return + end + + body, err = scrape_live_session() + if not body then + ngx.say(err) + return + end + + ngx.say("name live=", active_on(body, "Order TCP", "svc-a")) + ngx.say("id label live=", active_on(body, "svc-a", "svc-a")) + + local bw = tonumber(series_value(body, "apisix_stream_bandwidth", + 'listen_addr="0.0.0.0:1985",service="Order TCP",service_id="svc-a",' + .. 'type="ingress",side="downstream"')) + ngx.say("name bandwidth=", bw and tonumber(bw) > 0 and "counted" + or bw or "no-series") + + ngx.sleep(1.5) + } + } +--- request +GET /probe +--- stream_enable +--- timeout: 20 +--- response_body +name live=1 +id label live=0 +name bandwidth=counted + + + +=== TEST 8: the termination status follows the same rule +--- request +GET /apisix/prometheus/metrics +--- response_body_like eval +qr/apisix_stream_status\{code="200",listen_addr="0\.0\.0\.0:1985",service="Order TCP",service_id="svc-a",node="127\.0\.0\.1:1993"\} 2$/m +--- no_error_log +[error] + + + +=== TEST 9: a slot that lost its baseline rebaselines instead of replaying +A baseline can go missing while the rest of the dict stays, when it is +evicted or could not be written. Counting that slot from zero would land its +whole lifetime total in one interval; it has to take a new baseline instead, +and later traffic is counted as usual. +--- config + location /probe { + content_by_lua_block { + -- back to the id as the service label + local t = require("lib.test_admin").test + local code = t("/apisix/admin/services/svc-a", ngx.HTTP_PUT, [[{ + "upstream": { + "type": "roundrobin", + "nodes": [{ + "host": "127.0.0.1", + "port": 1993, + "weight": 1 + }] + } + }]]) + if code > 300 then + ngx.say("service: ", code) + return + end + + -- a series with a baseline behind it + local body, err = scrape_live_session() + if not body then + ngx.say(err) + return + end + body, err = scrape() + if not body then + ngx.say(err) + return + end + local before = tonumber(svc_a_ingress(body)) + + ngx.shared["prometheus-metrics"]:delete( + "stream_bytes_published:0.0.0.0:1985\31svc-a\31svc-a\31downstream_ingress") + + body, err = scrape() + if not body then + ngx.say(err) + return + end + ngx.say("after the lost baseline: ", tonumber(svc_a_ingress(body)) == before) + + body, err = scrape_live_session() + if not body then + ngx.say(err) + return + end + ngx.say("after one more session: +", tonumber(svc_a_ingress(body)) - before) + + ngx.sleep(1.5) + } + } +--- request +GET /probe +--- stream_enable +--- timeout: 20 +--- response_body +after the lost baseline: true +after one more session: +5 + + + +=== TEST 10: a session keeps the service label it started with +A route with an upstream_id and a service_id resolves its service name by +itself. The name picked when the session starts labels it in the zone and is +reused when it ends, so a rename while it is open does not split the session +between two names across the metrics. +--- config + location /probe { + content_by_lua_block { + local t = require("lib.test_admin").test + local function put(uri, body) + local code, res = t(uri, ngx.HTTP_PUT, body) + if code > 300 then + ngx.say(uri, ": ", res) + end + return code <= 300 + end + + if not put("/apisix/admin/services/svc-b", [[{ + "name": "Billing", + "upstream": { + "type": "roundrobin", + "nodes": [{"host": "127.0.0.1", "port": 1993, "weight": 1}] + } + }]]) or not put("/apisix/admin/upstreams/up-1", [[{ + "type": "roundrobin", + "nodes": [{"host": "127.0.0.1", "port": 1993, "weight": 1}] + }]]) or not put("/apisix/admin/stream_routes/1", [[{ + "plugins": { + "prometheus": { + "prefer_name": true + } + }, + "upstream_id": "up-1", + "service_id": "svc-b" + }]]) then + return + end + + ngx.sleep(1.5) + + local sock = ngx.socket.tcp() + assert(sock:connect("127.0.0.1", 1985)) + assert(sock:send("hello")) + assert(sock:receive("*l")) + + -- renamed while the session is open + if not put("/apisix/admin/services/svc-b", [[{ + "name": "Billing v2", + "upstream": { + "type": "roundrobin", + "nodes": [{"host": "127.0.0.1", "port": 1993, "weight": 1}] + } + }]]) then + return + end + ngx.sleep(1.5) + + assert(sock:close()) + + local body, err = scrape() + if not body then + ngx.say(err) + return + end + + local status = 'code="200",listen_addr="0.0.0.0:1985",service="%s",' + .. 'service_id="svc-b",node="127.0.0.1:1993"' + ngx.say("status under the first name: ", + series_value(body, "apisix_stream_status", status:format("Billing"))) + ngx.say("status under the new name: ", + series_value(body, "apisix_stream_status", status:format("Billing v2"))) + } + } +--- request +GET /probe +--- stream_enable +--- timeout: 20 +--- response_body +status under the first name: 1 +status under the new name: no-series diff --git a/t/stream-plugin/prometheus-metrics.t b/t/stream-plugin/prometheus-metrics.t index f7ecb83e4f4e..f398cf80ebe2 100644 --- a/t/stream-plugin/prometheus-metrics.t +++ b/t/stream-plugin/prometheus-metrics.t @@ -139,7 +139,7 @@ hello world --- request GET /apisix/prometheus/metrics --- response_body eval -qr/apisix_stream_status\{code="200",listen_addr="[^"]+",node="127.0.0.1:1995"\} 1$/m +qr/apisix_stream_status\{code="200",listen_addr="[^"]+",service="",service_id="",node="127.0.0.1:1995"\} 1$/m @@ -269,7 +269,7 @@ own rather than in one order-dependent pattern. --- request GET /apisix/prometheus/metrics --- response_body_like eval -qr/apisix_stream_bandwidth\{listen_addr="0\.0\.0\.0:1985",type="ingress",side="downstream"\} [1-9]\d*/ +qr/apisix_stream_bandwidth\{listen_addr="0\.0\.0\.0:1985",service="",service_id="",type="ingress",side="downstream"\} [1-9]\d*/ --- no_error_log [error] @@ -281,7 +281,7 @@ own rather than in one order-dependent pattern. --- request GET /apisix/prometheus/metrics --- response_body_like eval -qr/apisix_stream_bandwidth\{listen_addr="0\.0\.0\.0:1985",type="egress",side="downstream"\} [1-9]\d*/ +qr/apisix_stream_bandwidth\{listen_addr="0\.0\.0\.0:1985",service="",service_id="",type="egress",side="downstream"\} [1-9]\d*/ --- no_error_log [error] @@ -293,7 +293,7 @@ own rather than in one order-dependent pattern. --- request GET /apisix/prometheus/metrics --- response_body_like eval -qr/apisix_stream_bandwidth\{listen_addr="0\.0\.0\.0:1985",type="egress",side="upstream"\} [1-9]\d*/ +qr/apisix_stream_bandwidth\{listen_addr="0\.0\.0\.0:1985",service="",service_id="",type="egress",side="upstream"\} [1-9]\d*/ --- no_error_log [error] @@ -305,7 +305,7 @@ own rather than in one order-dependent pattern. --- request GET /apisix/prometheus/metrics --- response_body_like eval -qr/apisix_stream_bandwidth\{listen_addr="0\.0\.0\.0:1985",type="ingress",side="upstream"\} [1-9]\d*/ +qr/apisix_stream_bandwidth\{listen_addr="0\.0\.0\.0:1985",service="",service_id="",type="ingress",side="upstream"\} [1-9]\d*/ --- no_error_log [error] @@ -317,7 +317,7 @@ and outlived another tick, so the published value has to be 0 by now. --- request GET /apisix/prometheus/metrics --- response_body_like eval -qr/apisix_stream_active_connections\{listen_addr="0\.0\.0\.0:1985"\} 0$/m +qr/apisix_stream_active_connections\{listen_addr="0\.0\.0\.0:1985",service="",service_id=""\} 0$/m --- no_error_log [error] @@ -363,7 +363,7 @@ connect() failed --- request GET /apisix/prometheus/metrics --- response_body eval -qr/apisix_stream_status\{code="502",listen_addr="[^"]+",node="127.0.0.1:1979"\} 1$/m +qr/apisix_stream_status\{code="502",listen_addr="[^"]+",service="",service_id="",node="127.0.0.1:1979"\} 1$/m @@ -431,6 +431,202 @@ pins that for the same case -- so an idle timeout has to reach the metric as --- request GET /apisix/prometheus/metrics --- response_body_like eval -qr/apisix_stream_status\{code="502",listen_addr="0\.0\.0\.0:1985",node="127\.0\.0\.1:1993"\}/ +qr/apisix_stream_status\{code="502",listen_addr="0\.0\.0\.0:1985",service="",service_id="",node="127\.0\.0\.1:1993"\}/ --- no_error_log [error] + + + +=== TEST 16: move the route under a service +--- config + location /t { + content_by_lua_block { + local t = require("lib.test_admin").test + local code, body = t("/apisix/admin/services/svc-a", ngx.HTTP_PUT, [[{ + "upstream": { + "type": "roundrobin", + "nodes": [{ + "host": "127.0.0.1", + "port": 1995, + "weight": 1 + }] + } + }]]) + if code > 300 then + ngx.say(body) + return + end + + code, body = t("/apisix/admin/stream_routes/1", ngx.HTTP_PUT, [[{ + "plugins": { + "prometheus": {} + }, + "service_id": "svc-a" + }]]) + if code > 300 then + ngx.say(body) + return + end + } + } +--- response_body + + + +=== TEST 17: proxy a session through the service +--- stream_request +hello +--- stream_response +hello world + + + +=== TEST 18: the termination status carries the service +--- request +GET /apisix/prometheus/metrics +--- response_body_like eval +qr/apisix_stream_status\{code="200",listen_addr="0\.0\.0\.0:1985",service="svc-a",service_id="svc-a",node="127\.0\.0\.1:1995"\} 1$/m + + + +=== TEST 19: and so does the connection count +--- request +GET /apisix/prometheus/metrics +--- response_body_like eval +qr/apisix_stream_connection_total\{route="1",service="svc-a",service_id="svc-a"\} 1$/m + + + +=== TEST 20: name the service and ask for names on the route +--- config + location /t { + content_by_lua_block { + local t = require("lib.test_admin").test + local code, body = t("/apisix/admin/services/svc-a", ngx.HTTP_PUT, [[{ + "name": "Order TCP", + "upstream": { + "type": "roundrobin", + "nodes": [{ + "host": "127.0.0.1", + "port": 1995, + "weight": 1 + }] + } + }]]) + if code > 300 then + ngx.say(body) + return + end + + code, body = t("/apisix/admin/stream_routes/1", ngx.HTTP_PUT, [[{ + "plugins": { + "prometheus": { + "prefer_name": true + } + }, + "service_id": "svc-a" + }]]) + if code > 300 then + ngx.say(body) + return + end + } + } +--- response_body + + + +=== TEST 21: proxy a session through the named service +--- stream_request +hello +--- stream_response +hello world + + + +=== TEST 22: with prefer_name, service carries the name and service_id the id +The same rule as the http metrics. +--- request +GET /apisix/prometheus/metrics +--- response_body_like eval +qr/apisix_stream_status\{code="200",listen_addr="0\.0\.0\.0:1985",service="Order TCP",service_id="svc-a",node="127\.0\.0\.1:1995"\} 1$/m + + + +=== TEST 23: a route with both an upstream_id and a service_id +The service is not merged into such a route, so it is only named on the +route itself; the http metrics still label it, and so must the stream ones. +--- config + location /t { + content_by_lua_block { + local t = require("lib.test_admin").test + local code, body = t("/apisix/admin/services/svc-b", ngx.HTTP_PUT, [[{ + "name": "Billing", + "upstream": { + "type": "roundrobin", + "nodes": [{ + "host": "127.0.0.1", + "port": 1995, + "weight": 1 + }] + } + }]]) + if code > 300 then + ngx.say(body) + return + end + + code, body = t("/apisix/admin/upstreams/up-1", ngx.HTTP_PUT, [[{ + "type": "roundrobin", + "nodes": [{ + "host": "127.0.0.1", + "port": 1995, + "weight": 1 + }] + }]]) + if code > 300 then + ngx.say(body) + return + end + + code, body = t("/apisix/admin/stream_routes/1", ngx.HTTP_PUT, [[{ + "plugins": { + "prometheus": { + "prefer_name": true + } + }, + "upstream_id": "up-1", + "service_id": "svc-b" + }]]) + if code > 300 then + ngx.say(body) + return + end + } + } +--- response_body + + + +=== TEST 24: proxy a session through that route +--- stream_request +hello +--- stream_response +hello world + + + +=== TEST 25: the session carries the service of the route +--- request +GET /apisix/prometheus/metrics +--- response_body_like eval +qr/apisix_stream_status\{code="200",listen_addr="0\.0\.0\.0:1985",service="Billing",service_id="svc-b",node="127\.0\.0\.1:1995"\} 1$/m + + + +=== TEST 26: the connection count carries it too +The route has no name, so prefer_name leaves route on its id. +--- request +GET /apisix/prometheus/metrics +--- response_body_like eval +qr/apisix_stream_connection_total\{route="1",service="Billing",service_id="svc-b"\} 1$/m diff --git a/t/stream-plugin/prometheus.t b/t/stream-plugin/prometheus.t index 33b037cc6fbd..9be04a226b81 100644 --- a/t/stream-plugin/prometheus.t +++ b/t/stream-plugin/prometheus.t @@ -128,7 +128,7 @@ hello world --- request GET /apisix/prometheus/metrics --- response_body eval -qr/apisix_stream_connection_total\{route="mqtt"\} 1/ +qr/apisix_stream_connection_total\{route="mqtt",service="",service_id=""\} 1/ @@ -161,7 +161,7 @@ Received unexpected MQTT packet type+flags --- request GET /t --- response_body eval -qr/apisix_stream_connection_total\{route="mqtt"\} 2/ +qr/apisix_stream_connection_total\{route="mqtt",service="",service_id=""\} 2/