From 1d84df1673c65f603cec1f24db891fc88dafa2c0 Mon Sep 17 00:00:00 2001 From: Hrushikesh Yadav Date: Sun, 28 Jun 2026 16:48:17 +0530 Subject: [PATCH 1/6] [serve] Add max_request_retries to bound router retry loops Signed-off-by: Hrushikesh Yadav --- python/ray/serve/_private/constants.py | 6 +++ python/ray/serve/_private/router.py | 30 ++++++++++++++ python/ray/serve/config.py | 23 +++++++++++ python/ray/serve/tests/test_backpressure.py | 43 +++++++++++++++++++++ 4 files changed, 102 insertions(+) diff --git a/python/ray/serve/_private/constants.py b/python/ray/serve/_private/constants.py index 19395afb0c8c..218758450e12 100644 --- a/python/ray/serve/_private/constants.py +++ b/python/ray/serve/_private/constants.py @@ -536,6 +536,12 @@ "RAY_SERVE_ROUTER_RETRY_MAX_BACKOFF_S", 0.5 ) +# Maximum number of times the router retries routing a request after a replica +# rejects it. -1 means unlimited retries (default, for backwards compatibility). +RAY_SERVE_ROUTER_MAX_REQUEST_RETRIES = get_env_int( + "RAY_SERVE_ROUTER_MAX_REQUEST_RETRIES", -1 +) + # The default autoscaling policy to use if none is specified. DEFAULT_AUTOSCALING_POLICY_NAME = ( "ray.serve.autoscaling_policy:default_autoscaling_policy" diff --git a/python/ray/serve/_private/router.py b/python/ray/serve/_private/router.py index 9f3b262008d8..baacc181db68 100644 --- a/python/ray/serve/_private/router.py +++ b/python/ray/serve/_private/router.py @@ -673,6 +673,7 @@ def __init__( self._initial_backoff_s: Optional[float] = None self._backoff_multiplier: Optional[float] = None self._max_backoff_s: Optional[float] = None + self._max_request_retries: int = -1 # Initializing `self._metrics_manager` before `self.long_poll_client` is # necessary to avoid race condition where `self.update_deployment_config()` @@ -848,6 +849,9 @@ def update_deployment_config(self, deployment_config: DeploymentConfig): deployment_config.request_router_config.backoff_multiplier ) self._max_backoff_s = deployment_config.request_router_config.max_backoff_s + self._max_request_retries = ( + deployment_config.request_router_config.max_request_retries + ) if self._request_router: self._request_router.update_backoff_params( @@ -1117,6 +1121,20 @@ async def _route_and_send_request_once( return None + def _check_retry_limit(self, num_retries: int) -> None: + """Raise BackPressureError if the retry limit has been exceeded.""" + max_retries = self._max_request_retries + if max_retries >= 0 and num_retries > max_retries: + deployment_config = self._metrics_manager._deployment_config + raise BackPressureError( + num_queued_requests=self._metrics_manager.num_queued_requests, + max_queued_requests=( + deployment_config.max_queued_requests + if deployment_config is not None + else -1 + ), + ) + async def route_and_send_request( self, pr: PendingRequest, @@ -1126,11 +1144,13 @@ async def route_and_send_request( This will block indefinitely if no replicas are available to handle the request, so it's up to the caller to time out or cancel the request. + If max_request_retries is configured (>= 0), retries are bounded. """ # Wait for the router to be initialized before sending the request. await self._request_router_initialized.wait() is_retry = False + num_retries = 0 while True: result = await self._route_and_send_request_once( pr, @@ -1145,6 +1165,8 @@ async def route_and_send_request( # TODO(edoakes): this retry procedure is not perfect because it'll reset the # process of choosing candidates replicas (i.e., for locality-awareness). is_retry = True + num_retries += 1 + self._check_retry_limit(num_retries) @tracing_decorator_factory( trace_name="route_to_replica", @@ -1473,8 +1495,10 @@ async def _pick_and_reserve_replica( capacity rejection, replica actor death, or transient unavailability. Updates the queue-length cache from the replica's reported count. + If max_request_retries is configured (>= 0), retries are bounded. """ is_retry = False + num_retries = 0 while True: num_curr_replicas = len(self.request_router.curr_replicas) with self._metrics_manager.wrap_queued_request( @@ -1490,11 +1514,15 @@ async def _pick_and_reserve_replica( replica.replica_id, replica.actor_id, e ) is_retry = True + num_retries += 1 + self._check_retry_limit(num_retries) continue except ActorUnavailableError: self.request_router.on_replica_actor_unavailable(replica.replica_id) logger.warning(f"{replica.replica_id} is temporarily unavailable.") is_retry = True + num_retries += 1 + self._check_retry_limit(num_retries) continue self.request_router.on_new_queue_len_info( @@ -1504,6 +1532,8 @@ async def _pick_and_reserve_replica( return replica, slot_token is_retry = True + num_retries += 1 + self._check_retry_limit(num_retries) def _register_completion_callback( self, result: ReplicaResult, replica: RunningReplica, pr: PendingRequest diff --git a/python/ray/serve/config.py b/python/ray/serve/config.py index ba1a3e5a61aa..aea82bd33e66 100644 --- a/python/ray/serve/config.py +++ b/python/ray/serve/config.py @@ -36,6 +36,7 @@ DEFAULT_REQUEST_ROUTING_STATS_TIMEOUT_S, DEFAULT_TARGET_ONGOING_REQUESTS, DEFAULT_UVICORN_KEEP_ALIVE_TIMEOUT_S, + RAY_SERVE_ROUTER_MAX_REQUEST_RETRIES, RAY_SERVE_ROUTER_RETRY_BACKOFF_MULTIPLIER, RAY_SERVE_ROUTER_RETRY_INITIAL_BACKOFF_S, RAY_SERVE_ROUTER_RETRY_MAX_BACKOFF_S, @@ -279,6 +280,26 @@ class RequestRouterConfig(BaseModel): ), ) + max_request_retries: int = Field( + default=RAY_SERVE_ROUTER_MAX_REQUEST_RETRIES, + description=( + "Maximum number of times the router retries routing a request " + "after a replica rejects it. -1 means unlimited retries " + "(the default, for backwards compatibility). When the limit is " + "exceeded, the request is dropped with a BackPressureError (503)." + ), + ) + + @field_validator("max_request_retries") + @classmethod + def validate_max_request_retries(cls, v): + if v < -1: + raise ValueError( + "max_request_retries must be -1 (unlimited) " + "or a non-negative integer." + ) + return v + @field_validator("request_router_kwargs") @classmethod def request_router_kwargs_json_serializable(cls, v): @@ -313,6 +334,7 @@ def __eq__(self, other): and self.initial_backoff_s == other.initial_backoff_s and self.backoff_multiplier == other.backoff_multiplier and self.max_backoff_s == other.max_backoff_s + and self.max_request_retries == other.max_request_retries ) def __hash__(self): @@ -334,6 +356,7 @@ def __hash__(self): self.initial_backoff_s, self.backoff_multiplier, self.max_backoff_s, + self.max_request_retries, ) ) diff --git a/python/ray/serve/tests/test_backpressure.py b/python/ray/serve/tests/test_backpressure.py index 024d59bc4390..9f1183e36f9b 100644 --- a/python/ray/serve/tests/test_backpressure.py +++ b/python/ray/serve/tests/test_backpressure.py @@ -234,5 +234,48 @@ def send_request(): wait_for_condition(lambda: ray.get(signal_actor.cur_num_waiters.remote()) == 0) +def test_max_request_retries_config_validation(): + """max_request_retries accepts -1 and non-negative ints, rejects < -1.""" + from ray.serve.config import RequestRouterConfig + + RequestRouterConfig(max_request_retries=-1) + RequestRouterConfig(max_request_retries=0) + RequestRouterConfig(max_request_retries=10) + + with pytest.raises(ValueError): + RequestRouterConfig(max_request_retries=-2) + + +def test_handle_retry_limit(serve_instance): + """Requests should get 503 when max_request_retries is exhausted.""" + from ray.serve.config import RequestRouterConfig + + signal_actor = SignalActor.remote() + + @serve.deployment( + max_ongoing_requests=1, + max_queued_requests=-1, + request_router_config=RequestRouterConfig(max_request_retries=3), + ) + class Deployment: + async def __call__(self, msg: str) -> str: + await signal_actor.wait.remote() + return msg + + handle = serve.run(Deployment.bind()) + + first_response = handle.remote("hi-1") + wait_for_condition(lambda: ray.get(signal_actor.cur_num_waiters.remote()) == 1) + + with pytest.raises(BackPressureError): + handle.remote("hi-2").result(timeout=30) + + ray.get(signal_actor.send.remote()) + assert first_response.result() == "hi-1" + + ray.get(signal_actor.send.remote(clear=True)) + wait_for_condition(lambda: ray.get(signal_actor.cur_num_waiters.remote()) == 0) + + if __name__ == "__main__": sys.exit(pytest.main(["-v", "-s", __file__])) From 1e7f049cbcea2fcc4fa00fd4419d952656c0ab11 Mon Sep 17 00:00:00 2001 From: Hrushikesh Yadav Date: Mon, 29 Jun 2026 14:49:18 +0530 Subject: [PATCH 2/6] refactor: store max_queued_requests on AsyncioRouter directly - Add self._max_queued_requests field initialized in __init__ - Update it in update_deployment_config alongside other config fields - Use it in _check_retry_limit instead of accessing private _metrics_manager._deployment_config attribute Signed-off-by: Hrushikesh Yadav --- python/ray/serve/_private/router.py | 9 +++------ 1 file changed, 3 insertions(+), 6 deletions(-) diff --git a/python/ray/serve/_private/router.py b/python/ray/serve/_private/router.py index baacc181db68..92d631f97d48 100644 --- a/python/ray/serve/_private/router.py +++ b/python/ray/serve/_private/router.py @@ -674,6 +674,7 @@ def __init__( self._backoff_multiplier: Optional[float] = None self._max_backoff_s: Optional[float] = None self._max_request_retries: int = -1 + self._max_queued_requests: int = -1 # Initializing `self._metrics_manager` before `self.long_poll_client` is # necessary to avoid race condition where `self.update_deployment_config()` @@ -852,6 +853,7 @@ def update_deployment_config(self, deployment_config: DeploymentConfig): self._max_request_retries = ( deployment_config.request_router_config.max_request_retries ) + self._max_queued_requests = deployment_config.max_queued_requests if self._request_router: self._request_router.update_backoff_params( @@ -1125,14 +1127,9 @@ def _check_retry_limit(self, num_retries: int) -> None: """Raise BackPressureError if the retry limit has been exceeded.""" max_retries = self._max_request_retries if max_retries >= 0 and num_retries > max_retries: - deployment_config = self._metrics_manager._deployment_config raise BackPressureError( num_queued_requests=self._metrics_manager.num_queued_requests, - max_queued_requests=( - deployment_config.max_queued_requests - if deployment_config is not None - else -1 - ), + max_queued_requests=self._max_queued_requests, ) async def route_and_send_request( From 7368ff126a805fd3deb99c8c3dd638de942d616a Mon Sep 17 00:00:00 2001 From: Hrushikesh Yadav Date: Tue, 30 Jun 2026 00:27:45 +0530 Subject: [PATCH 3/6] address review: move unit test, fix imports, add docs - Move test_max_request_retries_config_validation to unit tests dir - Move inline imports to top of test_backpressure.py - Add request_router_config and max_request_retries to deployment docs Signed-off-by: Hrushikesh Yadav --- doc/source/serve/configure-serve-deployment.md | 3 ++- python/ray/serve/tests/test_backpressure.py | 15 +-------------- python/ray/serve/tests/unit/test_config.py | 10 ++++++++++ 3 files changed, 13 insertions(+), 15 deletions(-) diff --git a/doc/source/serve/configure-serve-deployment.md b/doc/source/serve/configure-serve-deployment.md index 5e2e5811cf48..255c5836355e 100644 --- a/doc/source/serve/configure-serve-deployment.md +++ b/doc/source/serve/configure-serve-deployment.md @@ -17,7 +17,8 @@ You can also refer to the [API reference](../serve/api/doc/ray.serve.deployment_ - `ray_actor_options` - Options to pass to the Ray Actor decorator, such as resource requirements. Valid options are: `accelerator_type`, `memory`, `num_cpus`, `num_gpus`, `object_store_memory`, `resources`, and `runtime_env` For more details - [Resource management in Serve](serve-cpus-gpus) - `max_ongoing_requests` - Maximum number of queries that are sent to a replica of this deployment without receiving a response. Defaults to 5 (note the default changed from 100 to 5 in Ray 2.32.0). This may be an important parameter to configure for [performance tuning](serve-perf-tuning). - `autoscaling_config` - Parameters to configure autoscaling behavior. If this is set, you can't set `num_replicas` to a number. For more details on configurable parameters for autoscaling, see [Ray Serve Autoscaling](serve-autoscaling). -- `max_queued_requests` - Maximum number of requests to this deployment that will be queued at each caller (proxy or DeploymentHandle). Once this limit is reached, subsequent requests will raise a BackPressureError (for handles) or return an HTTP 503 status code (for HTTP requests). Defaults to -1 (no limit). +- `max_queued_requests` - Maximum number of requests to this deployment that will be queued at each caller (proxy or DeploymentHandle). Once this limit is reached, subsequent requests will raise a BackPressureError (for handles) or return an HTTP 503 status code by default (configurable via `backpressure_config`) for HTTP requests. Defaults to -1 (no limit). + - `request_router_config` - Advanced configuration for the request router. Accepts a `RequestRouterConfig` object with options including `max_request_retries` (maximum number of times the router retries a request when replicas reject it; defaults to -1 for unlimited retries; set to a non-negative integer to bound retries and prevent OOM under sustained load). - `user_config` - Config to pass to the reconfigure method of the deployment. This can be updated dynamically without restarting the replicas of the deployment. The user_config must be fully JSON-serializable. For more details, see [Serve User Config](serve-user-config). - `health_check_period_s` - Duration between health check calls for the replica. Defaults to 10s. The health check is by default a no-op Actor call to the replica, but you can define your own health check using the "check_health" method in your deployment that raises an exception when unhealthy. - `health_check_timeout_s` - Duration in seconds, that replicas wait for a health check method to return before considering it as failed. Defaults to 30s. diff --git a/python/ray/serve/tests/test_backpressure.py b/python/ray/serve/tests/test_backpressure.py index 9f1183e36f9b..d5892d44d9fd 100644 --- a/python/ray/serve/tests/test_backpressure.py +++ b/python/ray/serve/tests/test_backpressure.py @@ -12,6 +12,7 @@ from ray import serve from ray._common.test_utils import SignalActor, wait_for_condition from ray.serve._private.test_utils import get_application_url +from ray.serve.config import RequestRouterConfig from ray.serve.exceptions import BackPressureError @@ -234,22 +235,8 @@ def send_request(): wait_for_condition(lambda: ray.get(signal_actor.cur_num_waiters.remote()) == 0) -def test_max_request_retries_config_validation(): - """max_request_retries accepts -1 and non-negative ints, rejects < -1.""" - from ray.serve.config import RequestRouterConfig - - RequestRouterConfig(max_request_retries=-1) - RequestRouterConfig(max_request_retries=0) - RequestRouterConfig(max_request_retries=10) - - with pytest.raises(ValueError): - RequestRouterConfig(max_request_retries=-2) - - def test_handle_retry_limit(serve_instance): """Requests should get 503 when max_request_retries is exhausted.""" - from ray.serve.config import RequestRouterConfig - signal_actor = SignalActor.remote() @serve.deployment( diff --git a/python/ray/serve/tests/unit/test_config.py b/python/ray/serve/tests/unit/test_config.py index 01a0bd48454f..79c24c9e5da5 100644 --- a/python/ray/serve/tests/unit/test_config.py +++ b/python/ray/serve/tests/unit/test_config.py @@ -1666,6 +1666,16 @@ def test_optional_field(self): assert "initial_replicas" not in result +def test_max_request_retries_config_validation(): + """max_request_retries accepts -1 and non-negative ints, rejects < -1.""" + RequestRouterConfig(max_request_retries=-1) + RequestRouterConfig(max_request_retries=0) + RequestRouterConfig(max_request_retries=10) + + with pytest.raises(ValueError): + RequestRouterConfig(max_request_retries=-2) + + if __name__ == "__main__": import sys From c0f6a52ab274143f078da384db933fb7caf4452a Mon Sep 17 00:00:00 2001 From: Hrushikesh Yadav Date: Tue, 30 Jun 2026 00:42:26 +0530 Subject: [PATCH 4/6] add max_request_retries to serve.proto for controller serialization - Add optional int32 max_request_retries field to RequestRouterConfig proto - Using optional so unset (older controller) falls back to Pydantic default (-1) - Add proto round-trip tests for explicit values and absent field Signed-off-by: Hrushikesh Yadav --- python/ray/serve/tests/unit/test_config.py | 19 +++++++++++++++++++ src/ray/protobuf/serve.proto | 6 ++++++ 2 files changed, 25 insertions(+) diff --git a/python/ray/serve/tests/unit/test_config.py b/python/ray/serve/tests/unit/test_config.py index 79c24c9e5da5..e91215f38844 100644 --- a/python/ray/serve/tests/unit/test_config.py +++ b/python/ray/serve/tests/unit/test_config.py @@ -1676,6 +1676,25 @@ def test_max_request_retries_config_validation(): RequestRouterConfig(max_request_retries=-2) +@pytest.mark.parametrize("max_retries", [-1, 0, 5]) +def test_max_request_retries_proto_roundtrip(max_retries): + """Ensure max_request_retries survives to_proto/from_proto.""" + config = DeploymentConfig( + request_router_config=RequestRouterConfig(max_request_retries=max_retries) + ) + roundtripped = DeploymentConfig.from_proto_bytes(config.to_proto_bytes()) + assert roundtripped.request_router_config.max_request_retries == max_retries + + +def test_max_request_retries_proto_absent_field(): + """Absent max_request_retries (older controller) falls back to -1 default.""" + proto = DeploymentConfig().to_proto() + proto.request_router_config.ClearField("max_request_retries") + assert not proto.request_router_config.HasField("max_request_retries") + deserialized = DeploymentConfig.from_proto(proto) + assert deserialized.request_router_config.max_request_retries == -1 + + if __name__ == "__main__": import sys diff --git a/src/ray/protobuf/serve.proto b/src/ray/protobuf/serve.proto index a92c0ac3248a..565dfa4a4208 100644 --- a/src/ray/protobuf/serve.proto +++ b/src/ray/protobuf/serve.proto @@ -135,6 +135,12 @@ message RequestRouterConfig { // Maximum backoff time (in seconds) between retries. double max_backoff_s = 8; + + // Maximum number of times the router retries routing a request after a + // replica rejects it. -1 means unlimited retries (the default). + // Declared optional so that unset (e.g. from an older controller) is + // distinguishable from 0 (which means zero retries). + optional int32 max_request_retries = 9; } //[End] ROUTING CONFIG From f6d6574cf581da7e320cf607bcedb14aa8ee2268 Mon Sep 17 00:00:00 2001 From: Hrushikesh Yadav Date: Thu, 2 Jul 2026 19:23:51 +0530 Subject: [PATCH 5/6] [serve] Replace flaky retry-limit integration test with deterministic unit test The integration test in test_backpressure.py used a single always-busy replica (max_ongoing_requests=1, blocked on a signal) to try to exhaust max_request_retries. That scenario never surfaces to the retry counter: when all replicas are at capacity, the request router's internal _choose_replicas_with_backoff loop backs off and blocks until a replica frees up, so _pick_and_reserve_replica never iterates and the request just queues. The request timed out instead of raising BackPressureError. Replace it with a unit test in test_router.py that drives the actual code path #61017 fixes: a replica that repeatedly rejects reservations (_reject_reservation=True on both the initial and retry pick). With max_request_retries set, the router now raises BackPressureError once the retry budget is exhausted. The test is deterministic and fast, following the existing test_choose_replica_retries_when_reservation_rejected harness. Signed-off-by: Hrushikesh Yadav --- python/ray/serve/tests/test_backpressure.py | 30 ------------------ python/ray/serve/tests/unit/test_router.py | 35 +++++++++++++++++++++ 2 files changed, 35 insertions(+), 30 deletions(-) diff --git a/python/ray/serve/tests/test_backpressure.py b/python/ray/serve/tests/test_backpressure.py index d5892d44d9fd..024d59bc4390 100644 --- a/python/ray/serve/tests/test_backpressure.py +++ b/python/ray/serve/tests/test_backpressure.py @@ -12,7 +12,6 @@ from ray import serve from ray._common.test_utils import SignalActor, wait_for_condition from ray.serve._private.test_utils import get_application_url -from ray.serve.config import RequestRouterConfig from ray.serve.exceptions import BackPressureError @@ -235,34 +234,5 @@ def send_request(): wait_for_condition(lambda: ray.get(signal_actor.cur_num_waiters.remote()) == 0) -def test_handle_retry_limit(serve_instance): - """Requests should get 503 when max_request_retries is exhausted.""" - signal_actor = SignalActor.remote() - - @serve.deployment( - max_ongoing_requests=1, - max_queued_requests=-1, - request_router_config=RequestRouterConfig(max_request_retries=3), - ) - class Deployment: - async def __call__(self, msg: str) -> str: - await signal_actor.wait.remote() - return msg - - handle = serve.run(Deployment.bind()) - - first_response = handle.remote("hi-1") - wait_for_condition(lambda: ray.get(signal_actor.cur_num_waiters.remote()) == 1) - - with pytest.raises(BackPressureError): - handle.remote("hi-2").result(timeout=30) - - ray.get(signal_actor.send.remote()) - assert first_response.result() == "hi-1" - - ray.get(signal_actor.send.remote(clear=True)) - wait_for_condition(lambda: ray.get(signal_actor.cur_num_waiters.remote()) == 0) - - if __name__ == "__main__": sys.exit(pytest.main(["-v", "-s", __file__])) diff --git a/python/ray/serve/tests/unit/test_router.py b/python/ray/serve/tests/unit/test_router.py index 80043c42f070..0599036f5abb 100644 --- a/python/ray/serve/tests/unit/test_router.py +++ b/python/ray/serve/tests/unit/test_router.py @@ -1857,6 +1857,41 @@ async def test_choose_replica_retries_when_reservation_rejected( assert fake_request_router.replica_queue_len_cache.get(r2_id) == 0 + @pytest.mark.parametrize( + "setup_router", + [{"enable_queue_len_cache": True}], + indirect=True, + ) + async def test_choose_replica_bounds_retries_on_repeated_rejection( + self, setup_router: Tuple[AsyncioRouter, FakeRequestRouter] + ): + """When a replica keeps rejecting reservations, the router should raise + BackPressureError after max_request_retries instead of retrying forever.""" + router, fake_request_router = setup_router + router._max_request_retries = 2 + + r1_id = ReplicaID( + unique_id="test-replica-1", deployment_id=DeploymentID(name="test") + ) + r1 = FakeReplica(r1_id) + r1._reject_reservation = True + # Both the initial and retry picks return the same always-rejecting replica, + # so no reservation ever succeeds and the retry counter drives the outcome. + fake_request_router.set_replica_to_return(r1) + fake_request_router.set_replica_to_return_on_retry(r1) + + request_metadata = RequestMetadata( + request_id="test-request-1", + internal_request_id="test-internal-request-1", + ) + + async def _enter(): + async with router.choose_replica(request_metadata): + pass + + with pytest.raises(BackPressureError): + await asyncio.wait_for(_enter(), timeout=5) + @pytest.mark.parametrize( "setup_router", [{"enable_queue_len_cache": True}], From 86a1e9a1c58aaf48498e90872d0a374f6e21b41e Mon Sep 17 00:00:00 2001 From: Hrushikesh Yadav Date: Fri, 3 Jul 2026 14:25:56 +0530 Subject: [PATCH 6/6] [serve] Add max_request_retries to expected get_serve_instance_details JSON The new max_request_retries field on RequestRouterConfig (default -1) now appears in the serialized deployment config returned by get_serve_instance_details. Both test_controller.py and test_direct_ingress.py hardcode the expected JSON and were missing the new key, so their `assert details_dict == expected_dict` failed. Add "max_request_retries": -1 to the expected request_router_config block in both. Signed-off-by: Hrushikesh Yadav --- python/ray/serve/tests/test_controller.py | 1 + python/ray/serve/tests/test_direct_ingress.py | 1 + 2 files changed, 2 insertions(+) diff --git a/python/ray/serve/tests/test_controller.py b/python/ray/serve/tests/test_controller.py index 20dc9312f28d..3e111668ca7a 100644 --- a/python/ray/serve/tests/test_controller.py +++ b/python/ray/serve/tests/test_controller.py @@ -210,6 +210,7 @@ def autoscaling_app(): "initial_backoff_s": 0.025, "backoff_multiplier": 2.0, "max_backoff_s": 0.5, + "max_request_retries": -1, }, "rolling_update_percentage": 0.2, }, diff --git a/python/ray/serve/tests/test_direct_ingress.py b/python/ray/serve/tests/test_direct_ingress.py index a542a5752503..f51ce0d214a5 100644 --- a/python/ray/serve/tests/test_direct_ingress.py +++ b/python/ray/serve/tests/test_direct_ingress.py @@ -2570,6 +2570,7 @@ def autoscaling_app(): "initial_backoff_s": 0.025, "backoff_multiplier": 2.0, "max_backoff_s": 0.5, + "max_request_retries": -1, }, "rolling_update_percentage": 0.2, },