Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
94 changes: 78 additions & 16 deletions src/main.py
Original file line number Diff line number Diff line change
Expand Up @@ -13,11 +13,14 @@
If vLLM dies during startup for a reason a restart cannot fix (CUDA OOM, a
MAX_MODEL_LEN the GPU cannot hold, a bad flag, a gated model), the worker stays
up and answers every job with the cause instead of crash-looping; see
startup_errors.py. One exception is retried in-place first: a Hugging Face
revision that no longer exists (vLLM pins refs to commit hashes since v0.28, so
a force-pushed repo invalidates the pin) gets a single relaunch against the
repo's current state before the error is declared fatal. Unrecognised failures
still exit non-zero so the platform retries them.
startup_errors.py. Two startup failures get a single relaunch first, because
retrying them can genuinely succeed: a Hugging Face revision that no longer
exists (vLLM pins refs to commit hashes since v0.28, so a force-pushed repo
invalidates the pin) is retried against the repo's current state, and a
startup OOM is retried with a smaller memory footprint (CUDA graphs disabled
and a reduced MAX_NUM_BATCHED_TOKENS — the two settings whose defaults grew at
init time in recent vLLM releases). Unrecognised failures still exit non-zero
so the platform retries them.
"""

import collections
Expand All @@ -34,7 +37,7 @@

import model_preflight
import startup_errors
from args_builder import build_vllm_args
from args_builder import TRUE_VALUES, build_vllm_args
from download_model import LOCAL_MODEL_ARGS_PATH

logging.basicConfig(level=logging.INFO, format="%(asctime)s %(levelname)s %(name)s: %(message)s")
Expand Down Expand Up @@ -172,6 +175,49 @@ def drop_pinned_revisions() -> None:
)


OOM_TOKEN_BUDGET = "8192" # vLLM doubled the max-num-batched-tokens default in v0.28


def apply_oom_recovery() -> bool:
"""Relax memory-hungry settings for a single relaunch after a startup OOM.

Returns False when there is nothing left to relax (CUDA graphs are already
off and the batched-token budget is already small), which main.py reads as
"another attempt with the same footprint would OOM the same way".
"""
relaxed = False
eager = os.getenv("ENFORCE_EAGER", "").strip().lower()
if eager not in TRUE_VALUES:
os.environ["ENFORCE_EAGER"] = "true"
if eager:
logging.warning(
"Overriding ENFORCE_EAGER=%s for the retry: graphs are how the "
"boot OOM'd, so dropping them.",
eager,
)
else:
logging.warning(
"Disabling CUDA graphs for the retry (ENFORCE_EAGER=true): their "
"memory has been reserved up front since v0.29, and the boot OOM'd."
)
relaxed = True
raw_budget = os.getenv("MAX_NUM_BATCHED_TOKENS", "").strip()
shrink = not raw_budget or raw_budget == "0"
if not shrink:
try:
shrink = int(raw_budget) > int(OOM_TOKEN_BUDGET)
except ValueError:
pass # unparsable: it is the operator's value, keep it
if shrink:
os.environ["MAX_NUM_BATCHED_TOKENS"] = OOM_TOKEN_BUDGET
logging.warning(
"Reducing the peak-activation budget for the retry (MAX_NUM_BATCHED_TOKENS=%s).",
OOM_TOKEN_BUDGET,
)
relaxed = True
return relaxed


def main() -> None:
global vllm_process

Expand All @@ -193,15 +239,15 @@ def main() -> None:
"Model pre-flight failed; answering jobs with the cause instead of starting vLLM: %s",
startup_error,
)
# At most two launch attempts: the second only for a vanished Hugging Face
# revision, the one startup failure where retrying can genuinely succeed
# because vLLM re-resolves the revision on every launch. (The pre-flight
# catches a *configured* revision that never existed; this catches one that
# existed at pre-flight time and was gone by load time, or a pin vLLM
# resolved itself.)
attempts_left = 2
# At most three launches: each of the two recoverable startup failures gets
# one relaunch — a vanished Hugging Face revision (vLLM re-resolves the pin
# on every launch; the pre-flight catches a *configured* revision that never
# existed, this catches one gone by load time or resolved by vLLM itself)
# and a startup OOM (a smaller footprint can fit where the defaults did not).
can_retry_revision = True
can_retry_oom = True
oom_retried = False
while startup_error is None:
attempts_left -= 1
vllm_process = start_vllm()
try:
wait_for_vllm(vllm_process)
Expand All @@ -210,7 +256,8 @@ def main() -> None:
logging.error("%s", e)
stop_vllm(vllm_process)
output = "".join(recent_output)
if attempts_left and startup_errors.revision_not_found(output):
if can_retry_revision and startup_errors.revision_not_found(output):
can_retry_revision = False
logging.warning(
"vLLM could not fetch the pinned Hugging Face revision "
"(vLLM pins refs to commit hashes; a force-pushed repo "
Expand All @@ -220,7 +267,22 @@ def main() -> None:
drop_pinned_revisions()
recent_output.clear()
continue
startup_error = startup_errors.classify(output, model=os.getenv("MODEL_NAME"))
if (
can_retry_oom
and startup_errors.memory_shortfall(output)
and apply_oom_recovery()
):
can_retry_oom = False
oom_retried = True
logging.warning(
"vLLM ran out of GPU memory during startup. Relaunching once "
"with a smaller memory footprint."
)
recent_output.clear()
continue
startup_error = startup_errors.classify(
output, model=os.getenv("MODEL_NAME"), oom_retried=oom_retried
)
if startup_error is None:
# Nothing recognisable, so let the platform restart us: a failed
# download or a bad host is worth another attempt.
Expand Down
36 changes: 34 additions & 2 deletions src/startup_errors.py
Original file line number Diff line number Diff line change
Expand Up @@ -79,6 +79,19 @@ def revision_not_found_message(named: str, revision: str) -> str:
)


def memory_shortfall(output: str) -> bool:
"""True when startup died for lack of GPU memory in any of its three forms.

Split out from classify() because main.py treats this like the vanished
revision: one relaunch with a smaller memory footprint can genuinely
succeed (CUDA graph memory is reserved up front since v0.29 and the
batched-tokens default doubled in v0.28, so a boot can OOM where a more
conservative setup fits), so the worker retries once before answering
jobs with the error.
"""
return bool(_OOM.search(output) or _NO_KV_MEMORY.search(output) or _KV_TOO_SMALL.search(output))


def revision_not_found(output: str) -> bool:
"""True when the failure is a Hugging Face revision that no longer exists.

Expand All @@ -90,8 +103,12 @@ def revision_not_found(output: str) -> bool:
return bool(_REVISION_NOT_FOUND.search(output))


def classify(output: str, model: Optional[str] = None) -> Optional[str]:
"""One actionable message for a known fatal failure, or None to let it retry."""
def classify(output: str, model: Optional[str] = None, oom_retried: bool = False) -> Optional[str]:
"""One actionable message for a known fatal failure, or None to let it retry.

``oom_retried`` mirrors the revision case: set it when main.py already ran
its one reduced-footprint relaunch, so the wording reflects what was tried.
"""
named = model or "The model"

# Memory problems first: an OOM traceback is often surrounded by secondary
Expand All @@ -105,6 +122,21 @@ def classify(output: str, model: Optional[str] = None) -> Optional[str]:
if estimated
else ""
)
if oom_retried:
# By the time classify sees this output, main.py's reduced-footprint
# relaunch has already run, so the advice skips knobs that were
# already applied and names the fixes that actually remain.
return (
f"{named} ran out of GPU memory during startup.{card}{hint} "
f"The worker already retried once with a smaller footprint "
f"(ENFORCE_EAGER=true and MAX_NUM_BATCHED_TOKENS reduced to 8192 "
f"where it was higher or unset) and it still does not fit. What "
f"remains: a smaller or quantized checkpoint, a larger GPU (or "
f"more GPUs with TENSOR_PARALLEL_SIZE), a lower MAX_MODEL_LEN, or, "
f"if KV_CACHE_DTYPE=fp8 is set on an Ampere/Ada GPU, "
f"KV_CACHE_DTYPE=auto (fp8 KV forces the FlashInfer backend there, "
f"which needs extra workspace memory)."
)
return (
f"{named} ran out of GPU memory during startup.{card}{hint} "
f"Lower MAX_MODEL_LEN or MAX_NUM_SEQS, set ENFORCE_EAGER=true to skip "
Expand Down
153 changes: 153 additions & 0 deletions tests/test_main_oom_retry.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,153 @@
"""main() relaunches a startup OOM once with a smaller footprint, and only once."""

import os
import sys
import types
from pathlib import Path
from unittest import mock

import pytest

sys.path.insert(0, str(Path(__file__).resolve().parent.parent / "src"))

import main # noqa: E402

TORCH_OOM = (
"torch.OutOfMemoryError: CUDA out of memory. Tried to allocate 340.00 MiB. "
"GPU 0 has a total capacity of 44.42 GiB of which 85.88 MiB is free."
)
KV_TOO_SMALL = (
"ValueError: The model's max seq len (65536) is larger than the available KV cache "
"memory (4.21 GiB). Based on the available memory, the estimated maximum model "
"length is 34480. Try increasing `gpu_memory_utilization` or decreasing "
"`max_model_len` when initializing the engine."
)
REVISION_NOT_FOUND = (
"huggingface_hub.errors.RevisionNotFoundError: 404 Client Error. Revision Not Found "
"for url https://huggingface.co/org/model/resolve/abc123def/config.json."
)
GATED = "huggingface_hub.errors.GatedRepoError: 401 Client Error: Cannot access gated repo"


@pytest.fixture
def harness(monkeypatch):
"""Run main() without vLLM, the RunPod SDK, or a GPU.

Returns a driver: call it with the outputs each launch attempt should die
with (None = the attempt becomes healthy) and it returns the handler module
stub, whose startup_error records what jobs would be answered with.
"""
monkeypatch.setenv("MODEL_NAME", "org/model")
monkeypatch.delenv("ENFORCE_EAGER", raising=False)
monkeypatch.delenv("MAX_NUM_BATCHED_TOKENS", raising=False)
# The merged pre-flight (model_preflight.py) would otherwise ask the real
# HF Hub about "org/model"; these tests are about what happens after it.
monkeypatch.setattr(main.model_preflight, "check_model_access", lambda: None)
monkeypatch.setattr(main, "stop_vllm", lambda proc: None)

handler_stub = types.ModuleType("handler")
handler_stub.handler = lambda job: None
runpod_stub = types.ModuleType("runpod")
runpod_stub.serverless = types.SimpleNamespace(start=mock.MagicMock())
monkeypatch.setitem(sys.modules, "handler", handler_stub)
monkeypatch.setitem(sys.modules, "runpod", runpod_stub)

start_vllm = mock.MagicMock(return_value=mock.MagicMock())
monkeypatch.setattr(main, "start_vllm", start_vllm)

def drive(*attempt_outputs):
outputs = iter(attempt_outputs)

def fake_wait(proc):
output = next(outputs)
if output is not None:
main.recent_output.clear()
main.recent_output.append(output)
raise RuntimeError("vLLM serve exited during startup with code 1")

monkeypatch.setattr(main, "wait_for_vllm", fake_wait)
main.recent_output.clear()
main.main()
handler_stub.launches = start_vllm.call_count
return handler_stub

return drive


class TestOomRelaunch:
def test_oom_gets_one_reduced_footprint_relaunch_that_can_succeed(self, harness):
handler = harness(TORCH_OOM, None)

assert handler.launches == 2
assert handler.startup_error is None
assert os.environ["ENFORCE_EAGER"] == "true"
assert os.environ["MAX_NUM_BATCHED_TOKENS"] == "8192"

def test_second_oom_answers_jobs_with_what_was_tried(self, harness):
handler = harness(TORCH_OOM, TORCH_OOM)

assert handler.launches == 2 # never a third attempt
assert "ran out of GPU memory" in handler.startup_error
assert "already retried once" in handler.startup_error
assert "44.42 GiB" in handler.startup_error

def test_no_kv_memory_and_kv_too_small_are_the_same_class(self, harness):
handler = harness(KV_TOO_SMALL, None)

assert handler.launches == 2
assert handler.startup_error is None

def test_fully_tightened_config_means_nothing_left_to_relax(self, harness, monkeypatch):
# Both knobs already at/over their recovery values: another attempt would
# run the same footprint and OOM identically, so don't pay for it.
monkeypatch.setenv("ENFORCE_EAGER", "true")
monkeypatch.setenv("MAX_NUM_BATCHED_TOKENS", "8192")

handler = harness(TORCH_OOM)

assert handler.launches == 1
assert "ran out of GPU memory" in handler.startup_error
assert "already retried" not in handler.startup_error

def test_partial_user_tightening_still_earns_one_relaunch(self, harness, monkeypatch):
# Eager is already on but the token budget is untouched (default 16384):
# halving it is a real change, so retry once shrinking only that knob.
monkeypatch.setenv("ENFORCE_EAGER", "true")

handler = harness(TORCH_OOM, None)

assert handler.launches == 2
assert os.environ["ENFORCE_EAGER"] == "true"
assert os.environ["MAX_NUM_BATCHED_TOKENS"] == "8192"

def test_user_tightened_token_budget_is_kept_for_the_relaunch(self, harness, monkeypatch):
monkeypatch.setenv("MAX_NUM_BATCHED_TOKENS", "4096")

handler = harness(TORCH_OOM, None)

assert handler.launches == 2
assert os.environ["MAX_NUM_BATCHED_TOKENS"] == "4096"
assert os.environ["ENFORCE_EAGER"] == "true"

def test_explicit_eager_false_is_still_retried_but_loudly_overridden(self, harness, monkeypatch):
monkeypatch.setenv("ENFORCE_EAGER", "false")

handler = harness(TORCH_OOM, None)

assert handler.launches == 2
assert os.environ["ENFORCE_EAGER"] == "true"

def test_non_memory_failures_do_not_take_the_oom_relaunch(self, harness):
handler = harness(REVISION_NOT_FOUND, GATED)

# The revision relaunch fires once; the OOM budget stays untouched, so
# the second failure is answered, not re-launched.
assert handler.launches == 2
assert "gated or private" in handler.startup_error
assert "MAX_NUM_BATCHED_TOKENS" not in os.environ

def test_revision_and_oom_each_get_their_single_retry(self, harness):
handler = harness(REVISION_NOT_FOUND, TORCH_OOM, None)

assert handler.launches == 3
assert handler.startup_error is None
12 changes: 8 additions & 4 deletions tests/test_main_revision_retry.py
Original file line number Diff line number Diff line change
@@ -1,4 +1,8 @@
"""main() relaunches once for a vanished HF revision, and only for that."""
"""main() relaunches once for a vanished HF revision, and not for other failures.

(Startup OOM gets its own single reduced-footprint relaunch; that policy lives
in test_main_oom_retry.py.)
"""

import sys
import types
Expand All @@ -15,7 +19,7 @@
"huggingface_hub.errors.RevisionNotFoundError: 404 Client Error. Revision Not Found "
"for url https://huggingface.co/org/model/resolve/abc123def/config.json."
)
TORCH_OOM = "torch.OutOfMemoryError: CUDA out of memory. Tried to allocate 108.00 MiB."
GATED = "huggingface_hub.errors.GatedRepoError: 401 Client Error: Cannot access gated repo"


@pytest.fixture
Expand Down Expand Up @@ -81,7 +85,7 @@ def test_second_revision_failure_answers_jobs_with_the_cause(self, harness):
assert "revision" in handler.startup_error.lower()

def test_other_fatal_failures_do_not_relaunch(self, harness):
handler = harness(TORCH_OOM)
handler = harness(GATED)

assert handler.launches == 1
assert "ran out of GPU memory" in handler.startup_error
assert "gated or private" in handler.startup_error
Loading
Loading