runpod: a pod is ready when the model is loaded, not when the port opens (v0.3.20)

/v1/models answers the moment a vLLM process starts — minutes before it can
serve, because the checkpoint still has to be downloaded (every pod
re-downloads it without a network volume) and the KV cache built on first use.
The pool called that ready, every client read it as usable, and the whole model
load landed inside somebody's first request, where it is indistinguishable from
a hang.

That is what cost today. One OCR page took 807s end to end — ~400s image pull,
~400s weights — while the client timed out at 180s and then 300s and concluded
the serving was broken. It was not: nobody had ever waited long enough, and the
only reason we know is a probe run with a 1400s timeout, which came back 200
with text and conf 0.9019.

So the pool now sends one token against the served model before calling the pod
ready. The pod bills through the load either way; this only decides whether the
wait is visible as a pod that is not ready yet, or hidden inside a request that
looks stuck. `pods_ready` becomes a signal a client can act on, which is what
every client already assumed it was.

A failed warm-up is reported and the pod used anyway: it may still serve (an
engine that takes no chat completions, a model whose warm-up shape we guessed
wrong), and rejecting a pod we have already paid to boot over a diagnostic
request would be worse than the hidden latency this removes. Off with
warmup_on_boot=false or warmup_timeout_s=0.
Co-Authored-By: 's avatarClaude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01PxYHWbCAjjjD7A1rhFKtaW
parent 9572ef56
......@@ -16,7 +16,7 @@
# Canonical product version for CoderAI — single source of truth. Both the API
# metadata and the admin web UI read from here.
__version__ = "0.3.19"
__version__ = "0.3.20"
__tagline__ = "Complete orchestration, distribution and escalation of remotizable advanced inference."
# Configure the CUDA caching allocator BEFORE torch is imported anywhere.
......
......@@ -167,6 +167,22 @@ class RunpodModelConfig:
# concurrency needs and retires the excess, after holding the observation
# this long (a burst that dips for a moment must not cost a pod that is
# needed again seconds later, and a cold pod takes minutes to replace).
# Load the weights BEFORE calling the pod ready, by sending it one tiny
# inference. /v1/models answers as soon as the server is up, which is
# minutes before it can actually serve: a vLLM pod still has to download
# the checkpoint and build its KV cache on first use. So "ready" meant
# "reachable", every client read it as "usable", and the whole load landed
# inside somebody's first request — 807s for one OCR page, against clients
# timing out at 180-300s, and the conclusion that serving was broken.
#
# The pod bills through the load either way. This only decides whether the
# wait is visible to the operator (a pod that is not ready yet) or hidden
# inside a request that looks hung.
warmup_on_boot: bool = True
#: Budget for it. Separate from load_timeout_s: answering /v1/models and
#: having the weights resident are different milestones, and a checkpoint
#: re-downloaded on every pod (no network volume) is the slow one.
warmup_timeout_s: int = 900
scale_down_after_s: int = 180
# Which surplus pod to give back first. Off, the pool retires the least
# loaded one (and among equals the one used least recently, whose prefix
......@@ -1663,6 +1679,9 @@ def parse_model_runpod(block: Optional[dict]) -> RunpodModelConfig:
# 0 disables shedding entirely (pods then go only when idle_timeout_s says so).
cfg.scale_down_after_s = max(0, _as_int(b.get("scale_down_after_s"),
cfg.scale_down_after_s))
cfg.warmup_on_boot = _as_bool(b.get("warmup_on_boot"), cfg.warmup_on_boot)
cfg.warmup_timeout_s = max(0, _as_int(b.get("warmup_timeout_s"),
cfg.warmup_timeout_s))
cfg.scale_down_costliest_first = _as_bool(b.get("scale_down_costliest_first"),
cfg.scale_down_costliest_first)
cfg.max_inflight_per_pod = max(0, _as_int(b.get("max_inflight_per_pod"),
......@@ -2530,6 +2549,49 @@ class RunpodPodPool:
hourly_usd=float(shared_rate or 0.0))
return True
def _warmup(self, pod_id: str, url: str) -> bool:
"""Make the pod load its weights before anyone calls it ready.
One token against the served model. Cheap, and it is the only request
that can tell the difference between "the server is up" and "the model
is resident" — /v1/models answers the moment the process starts.
Returns whether it worked. A failure is NOT fatal: the pod may still
serve (a model whose warm-up shape we guessed wrong, an engine that
does not take chat completions), and refusing to use a pod we have
already paid to boot over a diagnostic request would be worse than the
hidden latency this exists to remove.
"""
if not getattr(self.mcfg, "warmup_on_boot", True):
return False
budget = int(getattr(self.mcfg, "warmup_timeout_s", 900) or 0)
if budget <= 0:
return False
t0 = time.time()
payload = {"model": self.served or str(self.model_key),
"messages": [{"role": "user", "content": "ok"}],
"max_tokens": 1, "temperature": 0.0}
headers = {"Content-Type": "application/json"}
if self.api_key:
headers["Authorization"] = f"Bearer {self.api_key}"
try:
reply = pod_http.post(url.rstrip("/") + "/v1/chat/completions",
json=payload, headers=headers, timeout=budget)
except Exception as exc:
print(f"[runpod] pod {pod_id}: warm-up did not complete "
f"({type(exc).__name__}: {exc}) — serving anyway, the first "
f"real request will pay the model load", flush=True)
return False
took = time.time() - t0
if reply.status_code >= 400:
print(f"[runpod] pod {pod_id}: warm-up answered HTTP "
f"{reply.status_code} after {took:.0f}s — serving anyway "
f"({str(reply.text)[:200]})", flush=True)
return False
print(f"[runpod] pod {pod_id}: weights resident after {took:.0f}s "
f"(warm-up inference) — ready means ready", flush=True)
return True
def _data_centers(self) -> list:
"""Regions to try, in order. Always at least one entry.
......@@ -2632,6 +2694,10 @@ class RunpodPodPool:
f"{time.time() - t_port:.0f}s more "
f"({time.time() - t_created:.0f}s total)", flush=True)
self._log_pod_catalogue(url)
# Pay the model load here, where it is visible as a pod
# that is not ready yet, rather than inside the first
# request where it looks like a hang.
self._warmup(pod_id, url)
break
time.sleep(5)
else:
......
......@@ -742,6 +742,29 @@ and the pool rents a second pod only once the first is carrying
`scale_up_inflight_per_pod` requests. The same `pool` field works on a burst
target, so two models' overflow shares one card.
### "Ready" means the model is loaded
A vLLM pod answers `/v1/models` the moment its process starts — minutes before it
can serve anything, because the checkpoint still has to be downloaded (every pod
re-downloads it without a network volume) and the KV cache built on first use. So
a pod was called ready while it was merely reachable, and the whole model load
landed inside somebody's first request, where it is indistinguishable from a hang.
Measured: one OCR page took **807 s** end to end — ~400 s image pull, ~400 s
weights — while the client timed out at 180–300 s and reported the serving as
broken. It was not; nobody had waited long enough.
So the pool sends **one token** against the served model before calling the pod
ready (`warmup_on_boot`, on by default, budget `warmup_timeout_s`). The pod bills
through the load either way; this only decides whether the wait is visible as a
pod that is not ready yet, or hidden inside a request that looks stuck. A
warm-up that fails is reported and the pod is used anyway — it may still serve,
and refusing a pod already paid for over a diagnostic request would be worse than
the latency this removes.
For a client this means `pods_ready == 1` can be trusted: poll it, then send the
document with an ordinary timeout.
### What a pod was doing
A rented pod is the one machine whose logs cannot be read from outside: RunPod
......
""""Ready" has to mean the model is loaded, not that the port is open.
A vLLM pod answers /v1/models the moment its process starts — minutes before it
can serve anything, because the checkpoint still has to be downloaded (every
pod re-downloads it when there is no network volume) and the KV cache built on
first use. So the pool called the pod ready, every client read that as usable,
and the entire model load landed inside somebody's first request.
Measured on the Digesta orchestrator: one OCR page took 807s end to end, of which
~400s was the pull and ~400s the weights, while the client timed out at 180-300s
and concluded the serving was broken. It was not: nobody had ever waited long
enough. A warm-up moves that wait to where it is visible — a pod that is not
ready yet — and the pod bills through the load either way.
"""
import pathlib
import sys
ROOT = pathlib.Path(__file__).resolve().parents[1]
sys.path.insert(0, str(ROOT))
from codai.api import runpod_worker as rw # noqa: E402
class _Reply:
def __init__(self, status=200, text="{}"):
self.status_code, self.text = status, text
def _pool(monkeypatch, reply=None, boom=None, warmup=True, budget=900):
import threading
pool = rw.RunpodPodPool.__new__(rw.RunpodPodPool)
pool._cv = threading.Condition()
pool.model_key = "datalab-to/surya-ocr-2"
pool.served = "datalab-to/surya-ocr-2"
pool.api_key = "tok"
cfg = rw.RunpodModelConfig()
cfg.warmup_on_boot = warmup
cfg.warmup_timeout_s = budget
pool.mcfg = cfg
seen = {}
def _post(url, json=None, headers=None, timeout=None, **kw):
seen["url"], seen["json"] = url, json
seen["headers"], seen["timeout"] = headers or {}, timeout
if boom:
raise boom
return reply or _Reply()
from codai.api import pod_http
monkeypatch.setattr(pod_http, "post", _post)
return pool, seen
def test_the_warmup_asks_the_served_model_for_one_token(monkeypatch):
"""Cheap on purpose: the point is to make the weights resident, not to
generate anything."""
pool, seen = _pool(monkeypatch)
assert pool._warmup("p1", "https://10.0.0.2:8000") is True
assert seen["url"].endswith("/v1/chat/completions")
assert seen["json"]["model"] == "datalab-to/surya-ocr-2"
assert seen["json"]["max_tokens"] == 1
def test_it_carries_the_pods_bearer_token(monkeypatch):
pool, seen = _pool(monkeypatch)
pool._warmup("p1", "https://10.0.0.2:8000")
assert seen["headers"]["Authorization"] == "Bearer tok"
def test_the_budget_is_its_own_setting(monkeypatch):
"""Answering /v1/models and having the weights resident are different
milestones; a checkpoint re-downloaded per pod is the slow one."""
pool, seen = _pool(monkeypatch, budget=1234)
pool._warmup("p1", "https://10.0.0.2:8000")
assert seen["timeout"] == 1234
def test_a_failed_warmup_does_not_reject_the_pod(monkeypatch):
"""We have already paid to boot it, and it may well serve — refusing it over
a diagnostic request would be worse than the latency this removes."""
pool, _ = _pool(monkeypatch, boom=OSError("connection reset"))
assert pool._warmup("p1", "https://10.0.0.2:8000") is False
def test_an_http_error_is_reported_and_tolerated(monkeypatch, capsys):
pool, _ = _pool(monkeypatch, reply=_Reply(status=400, text="bad request"))
assert pool._warmup("p1", "https://10.0.0.2:8000") is False
assert "serving anyway" in capsys.readouterr().out
def test_it_can_be_switched_off(monkeypatch):
"""An engine that does not take chat completions, or an operator who would
rather have the pod available sooner."""
pool, seen = _pool(monkeypatch, warmup=False)
assert pool._warmup("p1", "https://10.0.0.2:8000") is False
assert seen == {} # nothing was sent
def test_a_zero_budget_switches_it_off_too(monkeypatch):
pool, seen = _pool(monkeypatch, budget=0)
assert pool._warmup("p1", "https://10.0.0.2:8000") is False
assert seen == {}
def test_a_successful_warmup_says_how_long_the_load_took(monkeypatch, capsys):
"""The number nobody had: it is the difference between a slow pod and a
hung request."""
pool, _ = _pool(monkeypatch)
pool._warmup("p1", "https://10.0.0.2:8000")
out = capsys.readouterr().out
assert "weights resident after" in out
def test_provisioning_warms_up_before_calling_the_pod_ready():
"""The ordering is the whole feature: the wait has to land before 'ready',
not inside the first request."""
import inspect
src = inspect.getsource(rw.RunpodPodPool._provision_one)
assert "_warmup" in src
assert src.index("_warmup") < src.index("ready for")
def test_the_defaults_warm_up(monkeypatch):
"""A pod that is reachable but cannot yet answer is the case that cost an
evening, so this is on unless it is turned off."""
cfg = rw.RunpodModelConfig()
assert cfg.warmup_on_boot is True
assert cfg.warmup_timeout_s >= 600
Markdown is supported
0% or
You are about to add 0 people to the discussion. Proceed with caution.
Finish editing this message first!
Please register or to comment