diff --git a/.runpod/hub.json b/.runpod/hub.json index 8e3a8332..fd4528a8 100644 --- a/.runpod/hub.json +++ b/.runpod/hub.json @@ -686,6 +686,16 @@ "advanced": true } }, + { + "key": "VLLM_STARTUP_TIMEOUT", + "input": { + "name": "vLLM Startup Timeout", + "type": "number", + "description": "Seconds to wait for the internal vLLM /health endpoint before failing the worker. Raise this for very large models (400GB+) where weight loading can exceed the default on contended hosts.", + "default": 1800, + "advanced": true + } + }, { "key": "ENABLE_EXPERT_PARALLEL", "input": { diff --git a/docs/configuration.md b/docs/configuration.md index bced49fd..487497fe 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -72,7 +72,7 @@ These are consumed by the wrapper itself, not passed to vLLM: | `BASE_PATH` | `/runpod-volume` | Build arg, not read at runtime: the image derives `HF_HOME` and `VLLM_CACHE_ROOT` (`$BASE_PATH/vllm-cache`) from it at build time, so both caches persist on a network volume. Setting it on an endpoint moves neither; set `HF_HUB_CACHE` (model downloads) and `VLLM_CACHE_ROOT` (compile cache) directly. | | `MAX_CONCURRENCY` | `30` | Max concurrent jobs per worker (RunPod concurrency modifier). vLLM queues internally beyond this. | | `VLLM_PORT` | `8000` | Loopback port the internal `vllm serve` binds to. | -| `VLLM_STARTUP_TIMEOUT`| `1200` | Seconds to wait for vLLM `/health` before failing the worker. | +| `VLLM_STARTUP_TIMEOUT`| `1800` | Seconds to wait for vLLM `/health` before failing the worker. | | `REQUEST_TIMEOUT` | `3600` | Per-request timeout to the vLLM server, in seconds. | | `VLLM_EXTRA_ARGS` | — | Raw extra CLI args (see above). | | `VLLM_CONFIG_FILE` | — | Path to a `vllm serve` YAML config file (`--config`). | diff --git a/src/main.py b/src/main.py index 42698da0..d206fdc7 100644 --- a/src/main.py +++ b/src/main.py @@ -47,7 +47,7 @@ VLLM_HOST = "127.0.0.1" VLLM_PORT = os.getenv("VLLM_PORT", "8000") -STARTUP_TIMEOUT = int(os.getenv("VLLM_STARTUP_TIMEOUT", "1200")) # seconds +STARTUP_TIMEOUT = int(os.getenv("VLLM_STARTUP_TIMEOUT", "1800")) # seconds HEALTH_POLL_INTERVAL = 2 # seconds # How long to wait for vLLM to exit after SIGTERM before SIGKILL. SHUTDOWN_GRACE = 30 # seconds