Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
10 changes: 10 additions & 0 deletions .runpod/hub.json
Original file line number Diff line number Diff line change
Expand Up @@ -686,6 +686,16 @@
"advanced": true
}
},
{
"key": "VLLM_STARTUP_TIMEOUT",
"input": {
"name": "vLLM Startup Timeout",
"type": "number",
"description": "Seconds to wait for the internal vLLM /health endpoint before failing the worker. Raise this for very large models (400GB+) where weight loading can exceed the default on contended hosts.",
"default": 1800,
"advanced": true
}
},
{
"key": "ENABLE_EXPERT_PARALLEL",
"input": {
Expand Down
2 changes: 1 addition & 1 deletion docs/configuration.md
Original file line number Diff line number Diff line change
Expand Up @@ -72,7 +72,7 @@ These are consumed by the wrapper itself, not passed to vLLM:
| `BASE_PATH` | `/runpod-volume` | Build arg, not read at runtime: the image derives `HF_HOME` and `VLLM_CACHE_ROOT` (`$BASE_PATH/vllm-cache`) from it at build time, so both caches persist on a network volume. Setting it on an endpoint moves neither; set `HF_HUB_CACHE` (model downloads) and `VLLM_CACHE_ROOT` (compile cache) directly. |
| `MAX_CONCURRENCY` | `30` | Max concurrent jobs per worker (RunPod concurrency modifier). vLLM queues internally beyond this. |
| `VLLM_PORT` | `8000` | Loopback port the internal `vllm serve` binds to. |
| `VLLM_STARTUP_TIMEOUT`| `1200` | Seconds to wait for vLLM `/health` before failing the worker. |
| `VLLM_STARTUP_TIMEOUT`| `1800` | Seconds to wait for vLLM `/health` before failing the worker. |
| `REQUEST_TIMEOUT` | `3600` | Per-request timeout to the vLLM server, in seconds. |
| `VLLM_EXTRA_ARGS` | — | Raw extra CLI args (see above). |
| `VLLM_CONFIG_FILE` | — | Path to a `vllm serve` YAML config file (`--config`). |
Expand Down
2 changes: 1 addition & 1 deletion src/main.py
Original file line number Diff line number Diff line change
Expand Up @@ -47,7 +47,7 @@

VLLM_HOST = "127.0.0.1"
VLLM_PORT = os.getenv("VLLM_PORT", "8000")
STARTUP_TIMEOUT = int(os.getenv("VLLM_STARTUP_TIMEOUT", "1200")) # seconds
STARTUP_TIMEOUT = int(os.getenv("VLLM_STARTUP_TIMEOUT", "1800")) # seconds
HEALTH_POLL_INTERVAL = 2 # seconds
# How long to wait for vLLM to exit after SIGTERM before SIGKILL.
SHUTDOWN_GRACE = 30 # seconds
Expand Down
Loading