Skip to content

Commit f3de2e5

Browse files
un-defclaude
andauthored
Restore model probe for service routers (#4322)
Since #4270, a router replica of a service with `model` and no `probes` was probed with `GET /health` instead of the chat completions request. The latter deadlocked: the router only answers it once workers are registered, and the worker sync skipped routers that were not ready. #4313 dropped the readiness check from the sync, so this deadlock is gone. `/health` passes before the router has any workers. In a rolling deployment, the replacement router was considered ready, and the old one was scaled down, before the replacement could serve requests. Now that #4320 syncs workers with every running router, the replacement passes the chat completions probe once it has workers. Workers behind a router still get no default probe, as they may not serve HTTP at all (gRPC workers). With a Dynamo router, a router rolling deployment now waits for the replacement indefinitely instead of scaling down the old router, as workers stay attached to the old router's etcd/NATS. Such deployments were broken before as well and are to be fixed separately. Co-authored-by: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
1 parent 6e36024 commit f3de2e5

3 files changed

Lines changed: 17 additions & 33 deletions

File tree

‎src/dstack/_internal/core/models/configurations.py‎

Lines changed: 0 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -72,7 +72,6 @@
7272
DEFAULT_PROBE_READY_AFTER = 1
7373
DEFAULT_PROBE_METHOD = "get"
7474
DEFAULT_PROBE_UNTIL_READY = False
75-
ROUTER_HEALTH_PROBE_URL = "/health"
7675
MAX_PROBE_URL_LEN = 2048
7776
DEFAULT_REPLICA_GROUP_NAME = "0"
7877
OPENAI_MODEL_PROBE_TIMEOUT = 30

‎src/dstack/_internal/server/services/jobs/configurators/base.py‎

Lines changed: 11 additions & 25 deletions
Original file line numberDiff line numberDiff line change
@@ -27,7 +27,6 @@
2727
DEFAULT_REPLICA_GROUP_NAME,
2828
LEGACY_REPO_DIR,
2929
OPENAI_MODEL_PROBE_TIMEOUT,
30-
ROUTER_HEALTH_PROBE_URL,
3130
HTTPHeaderSpec,
3231
NodeGroup,
3332
PortMapping,
@@ -513,17 +512,17 @@ def _probes(self) -> list[ProbeSpec]:
513512
model = conf.model
514513
if not isinstance(model, OpenAIChatModel):
515514
return []
516-
if all(group.router is None for group in conf.replica_groups):
517-
# No router: every replica serves the model itself, so a chat completions
518-
# request is a genuine end-to-end readiness check.
519-
return [_openai_model_probe_spec(model.name, model.prefix)]
520-
group = self._replica_group()
521-
if group is not None and group.router is not None:
522-
# Probe the router's own liveness endpoint, which does not depend on any
523-
# worker. Workers get no default probe: they may not serve HTTP at all
524-
# (gRPC workers), and they don't receive traffic directly.
525-
return [_router_health_probe_spec()]
526-
return []
515+
if any(group.router is not None for group in conf.replica_groups):
516+
group = self._replica_group()
517+
if group is None or group.router is None:
518+
# Workers get no default probe: they may not serve HTTP at all (gRPC
519+
# workers), and they don't receive traffic directly.
520+
return []
521+
# The router answers chat completions only once it has workers, which it gets
522+
# regardless of its own readiness. For SGLang routers, dstack registers workers
523+
# via `ServiceRouterWorkerSyncWorker`, which doesn't check router readiness.
524+
# Dynamo workers register themselves with the router via etcd/NATS.
525+
return [_openai_model_probe_spec(model.name, model.prefix)]
527526

528527

529528
def interpolate_job_volumes(
@@ -598,19 +597,6 @@ def _openai_model_probe_spec(model_name: str, prefix: str) -> ProbeSpec:
598597
)
599598

600599

601-
def _router_health_probe_spec() -> ProbeSpec:
602-
# Both supported routers (SGLang/SMG and Dynamo) serve `/health` independently of
603-
# whether any worker is registered.
604-
return ProbeSpec(
605-
type="http",
606-
method=DEFAULT_PROBE_METHOD,
607-
url=ROUTER_HEALTH_PROBE_URL,
608-
timeout=DEFAULT_PROBE_TIMEOUT,
609-
interval=DEFAULT_PROBE_INTERVAL,
610-
ready_after=DEFAULT_PROBE_READY_AFTER,
611-
)
612-
613-
614600
def _join_shell_commands(commands: List[str]) -> str:
615601
for i, cmd in enumerate(commands):
616602
cmd = cmd.strip()

‎src/tests/_internal/server/services/jobs/configurators/test_service.py‎

Lines changed: 6 additions & 7 deletions
Original file line numberDiff line numberDiff line change
@@ -6,7 +6,6 @@
66
from dstack._internal import settings
77
from dstack._internal.core.models.configurations import (
88
OPENAI_MODEL_PROBE_TIMEOUT,
9-
ROUTER_HEALTH_PROBE_URL,
109
ProbeConfig,
1110
PythonVersion,
1211
ReplicaGroup,
@@ -116,9 +115,9 @@ def _router_worker_configuration() -> ServiceConfiguration:
116115
],
117116
)
118117

119-
async def test_router_group_gets_health_probe(self):
120-
"""The router must not be probed with chat completions: it only answers those once
121-
dstack has registered workers, and registration requires the router to be ready."""
118+
async def test_router_group_gets_model_probe(self):
119+
"""The router is probed with chat completions like a replica serving the model
120+
itself: worker registration doesn't wait for the router to be ready."""
122121
run_spec = get_run_spec(
123122
run_name="run", repo_id="id", configuration=self._router_worker_configuration()
124123
)
@@ -128,9 +127,9 @@ async def test_router_group_gets_health_probe(self):
128127

129128
probes = job_specs[0].probes
130129
assert len(probes) == 1
131-
assert probes[0].url == ROUTER_HEALTH_PROBE_URL
132-
assert probes[0].method == "get"
133-
assert probes[0].body is None
130+
assert probes[0].method == "post"
131+
assert probes[0].url == "/v1/chat/completions"
132+
assert "meta-llama/Meta-Llama-3.1-8B-Instruct" in (probes[0].body or "")
134133

135134
async def test_worker_group_gets_no_derived_probe(self):
136135
"""Workers behind a router may speak gRPC, so no probe can be derived from `model`."""

0 commit comments

Comments
 (0)