From 5a4f00d77fbb75939d5d78aedb93ac7ab0d2231f Mon Sep 17 00:00:00 2001 From: amazloumi Date: Wed, 26 Aug 2026 10:28:49 -0400 Subject: [PATCH 1/3] Move VLM training configs into examples/vlm/ MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The 13 configs/train/vlm_*.toml presets move to examples/vlm/configs/ under the same filenames, so the core configs/train/ tree ships only general presets. BREAKING for any path naming the old location; no shim. The example now owns its own entry point (examples/vlm/train.py) and a README. The script is a thin argv -> load_config -> run_training wrapper: the VLM step body still lives in core, where run_training selects it from config.is_vlm, so no seam work was needed here. The two tests that loaded the shipped VLM configs move to examples/vlm/tests/ — they assert the example's contents, and a core test must not depend on examples/. The loader contract they also happened to cover (list[FreezeSpec] plus the parallel [vision_encoder]/[adapter]/[vlm] tables) stays in tests/unit/test_config.py against an inline TOML. [video].data_root in vlm_video_webvid.toml becomes a placeholder; shipped example configs must be machine-independent, and the example's tests now enforce that for every config in the directory. --- CHANGELOG.md | 1 + README.md | 16 ++- docs/how-to/train-on-video.md | 8 +- examples/vlm/README.md | 56 ++++++++ .../vlm/configs}/vlm_7b.toml | 2 +- .../vlm/configs}/vlm_7b_ac.toml | 4 +- .../vlm/configs}/vlm_7b_cross_attn.toml | 4 +- .../vlm/configs}/vlm_7b_freeze_schedule.toml | 8 +- .../vlm/configs}/vlm_7b_moma.toml | 2 +- .../vlm/configs}/vlm_7b_mot.toml | 2 +- .../vlm/configs}/vlm_7b_siglip2.toml | 2 +- .../configs}/vlm_7b_siglip2_cross_attn.toml | 4 +- .../vlm/configs}/vlm_debug.toml | 4 +- .../vlm/configs}/vlm_debug_moe.toml | 4 +- .../vlm/configs}/vlm_debug_moma.toml | 2 +- .../vlm/configs}/vlm_debug_mot.toml | 2 +- .../vlm/configs}/vlm_video_webvid.toml | 4 +- examples/vlm/tests/test_configs.py | 128 ++++++++++++++++++ examples/vlm/train.py | 39 ++++++ kempnerforge/model/vision.py | 2 +- tests/unit/test_config.py | 56 +++++--- 21 files changed, 296 insertions(+), 54 deletions(-) create mode 100644 examples/vlm/README.md rename {configs/train => examples/vlm/configs}/vlm_7b.toml (96%) rename {configs/train => examples/vlm/configs}/vlm_7b_ac.toml (95%) rename {configs/train => examples/vlm/configs}/vlm_7b_cross_attn.toml (96%) rename {configs/train => examples/vlm/configs}/vlm_7b_freeze_schedule.toml (91%) rename {configs/train => examples/vlm/configs}/vlm_7b_moma.toml (97%) rename {configs/train => examples/vlm/configs}/vlm_7b_mot.toml (96%) rename {configs/train => examples/vlm/configs}/vlm_7b_siglip2.toml (96%) rename {configs/train => examples/vlm/configs}/vlm_7b_siglip2_cross_attn.toml (96%) rename {configs/train => examples/vlm/configs}/vlm_debug.toml (93%) rename {configs/train => examples/vlm/configs}/vlm_debug_moe.toml (95%) rename {configs/train => examples/vlm/configs}/vlm_debug_moma.toml (96%) rename {configs/train => examples/vlm/configs}/vlm_debug_mot.toml (95%) rename {configs/train => examples/vlm/configs}/vlm_video_webvid.toml (93%) create mode 100644 examples/vlm/tests/test_configs.py create mode 100644 examples/vlm/train.py diff --git a/CHANGELOG.md b/CHANGELOG.md index a3b70ec..0e2e02d 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -105,6 +105,7 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ### Changed +- **BREAKING: VLM training configs moved to `examples/vlm/`.** The 13 `configs/train/vlm_*.toml` presets now live in `examples/vlm/configs/` (same filenames), alongside the example's own `train.py` entry point, `README.md`, and tests. No shim or symlink: update any path that named `configs/train/vlm_*.toml`. `[video].data_root` in `vlm_video_webvid.toml` is now a placeholder rather than a site-specific path. - **Training entry point is now a library.** `scripts/train.py:main()` (~1000 lines) is decomposed into `kempnerforge/training/` modules: `runtime.py` (`RuntimeContext`, `PipelineBundle`, `setup_distributed`), `data_pipeline.py` (`DataPipeline`, `PhaseState`, data/eval/phase builders), `loop.py` (`BatchStream`, `StepResult`, `TrainingSession`, `run_training_loop`, and the `text_step` / `vlm_step` / `pipeline_step` bodies picked by `select_step_fn`), and `entry.py` (`run_training` plus the model/checkpoint/resume builders). `scripts/train.py` is now a thin CLI wrapper — same CLI, same log lines, same metric keys, same checkpoint format. Behavior-preserving; the loop is unit-testable with a fake `CheckpointManager` (`tests/unit/test_train_entry.py`). `run_training(config, *, step_fn=None, hooks=None)` lets an experiment own the step body or register hooks without copying the build phases, and both `run_training` and `run_training_loop` tear down in `finally` now that they are library calls rather than a script about to exit. - `docs/getting-started/install.md` Prerequisites: documents `.python-version` and uv's auto-fetch behavior. - `README.md` and `kempnerforge/README.md` Prerequisites: clarify that uv auto-fetches Python 3.12 via `.python-version`. diff --git a/README.md b/README.md index 18d4867..cea7fe9 100644 --- a/README.md +++ b/README.md @@ -137,27 +137,29 @@ KempnerForge supports VLM training — images **or video** (a clip is an ordered - **Mixture-of-Transformers** (`arch = "mot"`, Liang et al. 2024 Algorithm 1): every layer carries per-modality Q/K/V/O projections plus a per-modality FFN; a single global self-attention mixes all modality streams. Image tokens prepend the text sequence (same residual layout as Joint-Decoder); per-modality residual projections are zero-initialized so a fresh MoT block is identity at construction. A warm-start helper (`mot_warm_start_from_text_stack`) translates a JD or text-only checkpoint into per-modality copies — toggle via `[model.vlm].mot_warm_start_from_text` + `mot_warm_start_path`. - **Mixture of Modality-Aware Experts** (`arch = "moma"`, Lin et al. 2024 arXiv:2407.21770): one shared set of Q/K/V/O projections feeding a global self-attention, plus per-modality MoE FFN groups (paper's optimal default 4 image + 4 text experts per layer). Tokens route deterministically to their modality group (level-1, reusing the same `modality_ids` mechanism MoT uses) and then through a learned expert-choice + Sigmoid router within the group (level-2, with Gumbel-Sigmoid noise during training, paper Eq. 5). Image tokens prepend the text sequence (same residual layout as JD/MoT). v1 supports training only — expert-choice routing is non-causal, so autoregressive generation requires auxiliary routers (paper §2.4) which are deferred to a follow-up. -**Video** works across all four archs with no arch-specific changes: a clip is decoded into frames (by a registered `sampling_policy` — default `uniform`: by timestamp at a target fps, first and last frame kept), each frame is encoded and pooled by the connector, and the `F × tokens_per_frame` visual tokens enter the backbone exactly like image tokens. The data side is **pluggable** — `[video].dataset_type` selects a registered dataset builder (`webvid` ships; `dataset_name` picks the corpus within a WebVid-style layout) and `sampling_policy` selects the frame-sampling policy, so new dataset styles / policies are additive registrations. Configure the `[video]` section (`data_root`, `dataset_type`, `dataset_name`, `sampling_policy`, `fps`, `max_frames`, `frame_size`); see `configs/train/vlm_video_webvid.toml`. Video decoding uses PyAV, an optional dependency — install it with `uv sync --group video`. +**Video** works across all four archs with no arch-specific changes: a clip is decoded into frames (by a registered `sampling_policy` — default `uniform`: by timestamp at a target fps, first and last frame kept), each frame is encoded and pooled by the connector, and the `F × tokens_per_frame` visual tokens enter the backbone exactly like image tokens. The data side is **pluggable** — `[video].dataset_type` selects a registered dataset builder (`webvid` ships; `dataset_name` picks the corpus within a WebVid-style layout) and `sampling_policy` selects the frame-sampling policy, so new dataset styles / policies are additive registrations. Configure the `[video]` section (`data_root`, `dataset_type`, `dataset_name`, `sampling_policy`, `fps`, `max_frames`, `frame_size`); see `examples/vlm/configs/vlm_video_webvid.toml`. Video decoding uses PyAV, an optional dependency — install it with `uv sync --group video`. + +The example ships in [`examples/vlm/`](examples/vlm/) — its configs and its own training entry point: ```bash # 1-GPU smoke (random encoder, Joint-Decoder) -uv run python scripts/train.py configs/train/vlm_debug.toml \ +uv run python examples/vlm/train.py examples/vlm/configs/vlm_debug.toml \ --data.hf_dataset_name= --data.tokenizer_path=gpt2 # 4-GPU SigLIP2 + 7B Joint-Decoder -uv run torchrun --nproc_per_node=4 scripts/train.py configs/train/vlm_7b_siglip2.toml +uv run torchrun --nproc_per_node=4 examples/vlm/train.py examples/vlm/configs/vlm_7b_siglip2.toml # 4-GPU 7B Cross-Attention (8 CA blocks at cadence 4) -uv run torchrun --nproc_per_node=4 scripts/train.py configs/train/vlm_7b_cross_attn.toml +uv run torchrun --nproc_per_node=4 examples/vlm/train.py examples/vlm/configs/vlm_7b_cross_attn.toml # 4-GPU 7B Mixture-of-Transformers -uv run torchrun --nproc_per_node=4 scripts/train.py configs/train/vlm_7b_mot.toml +uv run torchrun --nproc_per_node=4 examples/vlm/train.py examples/vlm/configs/vlm_7b_mot.toml # 4-GPU 7B Mixture of Modality-Aware Experts (4 text + 4 image experts per layer) -uv run torchrun --nproc_per_node=4 scripts/train.py configs/train/vlm_7b_moma.toml +uv run torchrun --nproc_per_node=4 examples/vlm/train.py examples/vlm/configs/vlm_7b_moma.toml # 4-GPU video training on WebVid (Joint-Decoder; flip [vlm].arch for cross_attention / mot / moma) -uv run torchrun --nproc_per_node=4 scripts/train.py configs/train/vlm_video_webvid.toml +uv run torchrun --nproc_per_node=4 examples/vlm/train.py examples/vlm/configs/vlm_video_webvid.toml ``` Configs set `[model.vlm]` with `arch`, the encoder registry key, the number of image tokens, and a freeze list (`FreezeSpec`). For Cross-Attention, set `cross_attention_every_n_layers` and optionally `cross_attention_n_kv_heads` (0 → MHA; positive → GQA on the cross path). For MoT, set `mot_modalities` (must include both `"image"` and `"text"`); `mot_image_n_heads` / `mot_image_n_kv_heads` are forward-looking per-modality head fields (v1 enforces equality with the text-side counts since the operator runs a single global SDPA). For MoMa, set `moma_experts_per_modality = {image = N, text = M}` as a nested TOML table (the paper's optimal balanced default is `4t4i`; unbalanced allocations like `{image = 1, text = 7}` match the paper's `moe_7t1i` ablation), and optionally `moma_capacity_factor` (defaults to `1/|E^M|` per modality — the paper's perfect-balance setting) and `moma_gumbel_noise` (`true` by default for paper-faithful EC routing). `model.num_experts` must be `0` when `arch = "moma"`; the per-modality counts supersede it, and JobConfig.validate rejects the combination. The vision encoder stays in its HF-loaded dtype; the transformer, adapter, and CA / MoT / MoMa blocks are cast to `param_dtype`. Pipeline Parallel + VLM is not supported on this branch (raises at startup); MoMa + Expert Parallelism is also rejected in v1. Video is supported across all four archs via the `[video]` section (a clip is decoded into frames, pooled by the connector, and fed like image tokens); multi-image inputs and video *grounding* (point/track outputs with per-frame timestamps) are reserved for follow-up work. diff --git a/docs/how-to/train-on-video.md b/docs/how-to/train-on-video.md index 5fc98f3..6c347d4 100644 --- a/docs/how-to/train-on-video.md +++ b/docs/how-to/train-on-video.md @@ -44,7 +44,7 @@ build- and config-time checks enforce this and fail before any GPU work. A video run adds a `[video]` section (sibling of `[vision_encoder]` / `[adapter]` / `[vlm]`) and a token-reducing connector. See -`configs/train/vlm_video_webvid.toml` for a complete example; the key parts: +`examples/vlm/configs/vlm_video_webvid.toml` for a complete example; the key parts: ```toml [adapter] @@ -55,7 +55,7 @@ pool_window = 2 # 14×14 grid -> 7×7 = 49 tokens/frame arch = "joint_decoder" # also: cross_attention | mot | moma [video] -data_root = "/path/to/webvid-10m" +data_root = "path-to-webvid-10m" dataset_type = "webvid" # registry key; add styles via @registry.register_video_dataset dataset_name = "webvid-10M" # corpus dir under raw//data (WebVid style) sampling_policy = "uniform" # registry key; the frame-sampling policy @@ -85,10 +85,10 @@ requires it. ```bash # 4-GPU video training (Joint-Decoder) -uv run torchrun --nproc_per_node=4 scripts/train.py configs/train/vlm_video_webvid.toml +uv run torchrun --nproc_per_node=4 examples/vlm/train.py examples/vlm/configs/vlm_video_webvid.toml # Quick smoke: no SigLIP download, a few clips, few steps -uv run torchrun --nproc_per_node=2 scripts/train.py configs/train/vlm_video_webvid.toml \ +uv run torchrun --nproc_per_node=2 examples/vlm/train.py examples/vlm/configs/vlm_video_webvid.toml \ --vision_encoder.type=random --vision_encoder.num_tokens=196 \ --vision_encoder.feature_dim=768 --video.max_samples=256 --train.max_steps=20 ``` diff --git a/examples/vlm/README.md b/examples/vlm/README.md new file mode 100644 index 0000000..5461c92 --- /dev/null +++ b/examples/vlm/README.md @@ -0,0 +1,56 @@ +# VLM example + +Vision-language training — images **or** video — on the core `Transformer`. A +frozen HF vision encoder produces visual tokens, a connector projects (and +optionally pools) them, and an arch-specific path feeds the backbone. Everything +here is configuration plus an entry point; `kempnerforge/` never imports it, so +this directory can be deleted without touching the core. + +## Configs + +`vlm_debug*` are 1-GPU smoke presets — tiny backbone, `random` encoder, so they +run on a fresh clone with no download. The `vlm_7b*` presets are 4-8 GPU +starting points. + +| Config | Arch | Encoder | For | +| --- | --- | --- | --- | +| `vlm_debug.toml` | joint_decoder | random | 1-GPU smoke | +| `vlm_debug_mot.toml` | mot | random | 1-GPU smoke | +| `vlm_debug_moma.toml` | moma | random | 1-GPU smoke | +| `vlm_debug_moe.toml` | cross_attention | random | 1-GPU smoke, MoE FFN | +| `vlm_7b.toml` | joint_decoder | random | 7B, AC off (VRAM stress) | +| `vlm_7b_ac.toml` | joint_decoder | random | 7B, AC full + longer seq | +| `vlm_7b_mot.toml` | mot | random | 7B | +| `vlm_7b_moma.toml` | moma | random | 7B | +| `vlm_7b_cross_attn.toml` | cross_attention | random | 7B | +| `vlm_7b_freeze_schedule.toml` | cross_attention | random | multi-stage `FreezeStage` schedule | +| `vlm_7b_siglip2.toml` | joint_decoder | siglip2 | real-run starting point | +| `vlm_7b_siglip2_cross_attn.toml` | cross_attention | siglip2 | real-run starting point | +| `vlm_video_webvid.toml` | joint_decoder | siglip2 | video (WebVid-10M) | + +Paths in these configs are placeholders (`data_root = "path-to-webvid-10m"`) — +point them at your own data and output directories, or override on the CLI. + +## Run it + +```bash +# 1-GPU smoke +uv run python examples/vlm/train.py examples/vlm/configs/vlm_debug.toml + +# 4 GPUs, single node +uv run torchrun --nproc_per_node=4 examples/vlm/train.py \ + examples/vlm/configs/vlm_7b_siglip2.toml + +# Override anything on the CLI +uv run python examples/vlm/train.py examples/vlm/configs/vlm_debug.toml \ + --train.max_steps=20 --checkpoint.dir=/your/run/dir +``` + +Video needs PyAV: `uv sync --group video`. + +Tests: `uv run pytest examples/vlm/tests/ -v` (they are outside the core +`testpaths`, so run them by path). + +## Evaluation + +Benchmark evaluation of the resulting checkpoints lives in `eval/`. diff --git a/configs/train/vlm_7b.toml b/examples/vlm/configs/vlm_7b.toml similarity index 96% rename from configs/train/vlm_7b.toml rename to examples/vlm/configs/vlm_7b.toml index 70c67ec..e0e0095 100644 --- a/configs/train/vlm_7b.toml +++ b/examples/vlm/configs/vlm_7b.toml @@ -16,7 +16,7 @@ # throughput: ~41k tok/s, ~46% MFU # # Usage: -# uv run torchrun --nproc_per_node=4 scripts/train.py configs/train/vlm_7b.toml +# uv run torchrun --nproc_per_node=4 examples/vlm/train.py examples/vlm/configs/vlm_7b.toml # # Default points at a 30-sample COCO val substitute (sayakpaul/coco-30-val-2014) # so a fresh clone runs without external setup. For real training, override diff --git a/configs/train/vlm_7b_ac.toml b/examples/vlm/configs/vlm_7b_ac.toml similarity index 95% rename from configs/train/vlm_7b_ac.toml rename to examples/vlm/configs/vlm_7b_ac.toml index 66fb377..41d8638 100644 --- a/configs/train/vlm_7b_ac.toml +++ b/examples/vlm/configs/vlm_7b_ac.toml @@ -16,8 +16,8 @@ # throughput: ~31k tok/s, ~35% MFU # # Usage: -# uv run torchrun --nproc_per_node=4 scripts/train.py \ -# configs/train/vlm_7b_ac.toml +# uv run torchrun --nproc_per_node=4 examples/vlm/train.py \ +# examples/vlm/configs/vlm_7b_ac.toml # # Default points at a 30-sample COCO val substitute (sayakpaul/coco-30-val-2014) # so a fresh clone runs without external setup. For real training, override diff --git a/configs/train/vlm_7b_cross_attn.toml b/examples/vlm/configs/vlm_7b_cross_attn.toml similarity index 96% rename from configs/train/vlm_7b_cross_attn.toml rename to examples/vlm/configs/vlm_7b_cross_attn.toml index 5b729f1..cc2fdee 100644 --- a/configs/train/vlm_7b_cross_attn.toml +++ b/examples/vlm/configs/vlm_7b_cross_attn.toml @@ -26,8 +26,8 @@ # of compute per step. # # Usage: -# uv run torchrun --nproc_per_node=4 scripts/train.py \ -# configs/train/vlm_7b_cross_attn.toml +# uv run torchrun --nproc_per_node=4 examples/vlm/train.py \ +# examples/vlm/configs/vlm_7b_cross_attn.toml # # Default points at a 30-sample COCO val substitute (sayakpaul/coco-30-val-2014) # so a fresh clone runs without external setup. For real training, override diff --git a/configs/train/vlm_7b_freeze_schedule.toml b/examples/vlm/configs/vlm_7b_freeze_schedule.toml similarity index 91% rename from configs/train/vlm_7b_freeze_schedule.toml rename to examples/vlm/configs/vlm_7b_freeze_schedule.toml index b007a31..feec21f 100644 --- a/configs/train/vlm_7b_freeze_schedule.toml +++ b/examples/vlm/configs/vlm_7b_freeze_schedule.toml @@ -6,21 +6,21 @@ # max_seq_len allocation (CA): max_text_len only. The residual stream is # text-only; image features flow as K/V into separate CrossAttentionBlocks. # -# Exercises the FreezeStage hook in scripts/train.py: +# Exercises the FreezeStage hook: # - step 0..9: vision encoder frozen, everything else trainable. # - step 10: freeze adapter (typical "pretrain CA blocks first" recipe). # - step 20: unfreeze adapter (typical "now align embeddings" recipe). # # checkpoint.interval matches one transition step (10) to exercise -# the async-save fence: the FreezeStage hook in scripts/train.py +# the async-save fence: the FreezeStage hook # calls flush_pending_save() before applying the transition, so the # in-flight save's metadata.json lands with the pre-transition spec # (adapter trainable) and only the next save records the post- # transition spec (adapter frozen). # # Usage: -# uv run torchrun --nproc_per_node=4 scripts/train.py \ -# configs/train/vlm_7b_freeze_schedule.toml +# uv run torchrun --nproc_per_node=4 examples/vlm/train.py \ +# examples/vlm/configs/vlm_7b_freeze_schedule.toml # # Default points at a 30-sample COCO val substitute (sayakpaul/coco-30-val-2014) # so a fresh clone runs without external setup. For real training, override diff --git a/configs/train/vlm_7b_moma.toml b/examples/vlm/configs/vlm_7b_moma.toml similarity index 97% rename from configs/train/vlm_7b_moma.toml rename to examples/vlm/configs/vlm_7b_moma.toml index 0a1b36f..a6f0840 100644 --- a/configs/train/vlm_7b_moma.toml +++ b/examples/vlm/configs/vlm_7b_moma.toml @@ -30,7 +30,7 @@ # must cover both modalities. # # Usage: -# uv run torchrun --nproc_per_node=4 scripts/train.py configs/train/vlm_7b_moma.toml +# uv run torchrun --nproc_per_node=4 examples/vlm/train.py examples/vlm/configs/vlm_7b_moma.toml # # Default points at a 30-sample COCO val substitute (sayakpaul/coco-30-val-2014) # so a fresh clone runs without external setup. For real training override: diff --git a/configs/train/vlm_7b_mot.toml b/examples/vlm/configs/vlm_7b_mot.toml similarity index 96% rename from configs/train/vlm_7b_mot.toml rename to examples/vlm/configs/vlm_7b_mot.toml index b282cad..a4b5126 100644 --- a/configs/train/vlm_7b_mot.toml +++ b/examples/vlm/configs/vlm_7b_mot.toml @@ -29,7 +29,7 @@ # loss: 11.6 -> 6.4 over 50 steps (clear convergence on real data) # # Usage: -# uv run torchrun --nproc_per_node=4 scripts/train.py configs/train/vlm_7b_mot.toml +# uv run torchrun --nproc_per_node=4 examples/vlm/train.py examples/vlm/configs/vlm_7b_mot.toml # # Default points at a 30-sample COCO val substitute (sayakpaul/coco-30-val-2014) # so a fresh clone runs without external setup. For real training, override diff --git a/configs/train/vlm_7b_siglip2.toml b/examples/vlm/configs/vlm_7b_siglip2.toml similarity index 96% rename from configs/train/vlm_7b_siglip2.toml rename to examples/vlm/configs/vlm_7b_siglip2.toml index 261b84a..c673a6e 100644 --- a/configs/train/vlm_7b_siglip2.toml +++ b/examples/vlm/configs/vlm_7b_siglip2.toml @@ -23,7 +23,7 @@ # Leaves ~90 GB / H100 for activations and grad accumulation headroom. # # Usage: -# uv run torchrun --nproc_per_node=4 scripts/train.py configs/train/vlm_7b_siglip2.toml +# uv run torchrun --nproc_per_node=4 examples/vlm/train.py examples/vlm/configs/vlm_7b_siglip2.toml # # Data: # Default hf_dataset_name is a 30-sample COCO val substitute diff --git a/configs/train/vlm_7b_siglip2_cross_attn.toml b/examples/vlm/configs/vlm_7b_siglip2_cross_attn.toml similarity index 96% rename from configs/train/vlm_7b_siglip2_cross_attn.toml rename to examples/vlm/configs/vlm_7b_siglip2_cross_attn.toml index a70e3bf..79db7fc 100644 --- a/configs/train/vlm_7b_siglip2_cross_attn.toml +++ b/examples/vlm/configs/vlm_7b_siglip2_cross_attn.toml @@ -24,8 +24,8 @@ # carries text only. # # Usage: -# uv run torchrun --nproc_per_node=4 scripts/train.py \ -# configs/train/vlm_7b_siglip2_cross_attn.toml +# uv run torchrun --nproc_per_node=4 examples/vlm/train.py \ +# examples/vlm/configs/vlm_7b_siglip2_cross_attn.toml # # Default points at a 30-sample COCO val substitute (sayakpaul/coco-30-val-2014) # so a fresh clone runs without external setup. For real training, override diff --git a/configs/train/vlm_debug.toml b/examples/vlm/configs/vlm_debug.toml similarity index 93% rename from configs/train/vlm_debug.toml rename to examples/vlm/configs/vlm_debug.toml index 108c37b..fa04e4f 100644 --- a/configs/train/vlm_debug.toml +++ b/examples/vlm/configs/vlm_debug.toml @@ -5,14 +5,14 @@ # Runs end-to-end in <1 minute on 1 GPU. Uses RandomVisionEncoder so no # HF download is needed; pair with any HF image-text dataset (the default # below is a placeholder). For a real vision encoder, see -# configs/train/vlm_7b_siglip2.toml. +# examples/vlm/configs/vlm_7b_siglip2.toml. # # max_seq_len allocation (JD/MoT): residual_image_tokens + max_text_len. # Image tokens prepend the text sequence in the residual stream, so the # budget must cover both modalities. # # Usage: -# uv run python scripts/train.py configs/train/vlm_debug.toml \ +# uv run python examples/vlm/train.py examples/vlm/configs/vlm_debug.toml \ # --data.hf_dataset_name=... --data.tokenizer_path=gpt2 [model] diff --git a/configs/train/vlm_debug_moe.toml b/examples/vlm/configs/vlm_debug_moe.toml similarity index 95% rename from configs/train/vlm_debug_moe.toml rename to examples/vlm/configs/vlm_debug_moe.toml index 6ee0cd5..b1e89be 100644 --- a/configs/train/vlm_debug_moe.toml +++ b/examples/vlm/configs/vlm_debug_moe.toml @@ -19,8 +19,8 @@ # through the VLMWrapper without exposing them as wrapper attrs. # # Usage: -# uv run torchrun --nproc_per_node=2 scripts/train.py \ -# configs/train/vlm_debug_moe.toml \ +# uv run torchrun --nproc_per_node=2 examples/vlm/train.py \ +# examples/vlm/configs/vlm_debug_moe.toml \ # --data.hf_dataset_name= --data.tokenizer_path=gpt2 [model] diff --git a/configs/train/vlm_debug_moma.toml b/examples/vlm/configs/vlm_debug_moma.toml similarity index 96% rename from configs/train/vlm_debug_moma.toml rename to examples/vlm/configs/vlm_debug_moma.toml index 5575fb6..cdd5a6d 100644 --- a/configs/train/vlm_debug_moma.toml +++ b/examples/vlm/configs/vlm_debug_moma.toml @@ -17,7 +17,7 @@ # (paper §2.4), deferred to a follow-up. # # Usage: -# uv run python scripts/train.py configs/train/vlm_debug_moma.toml \ +# uv run python examples/vlm/train.py examples/vlm/configs/vlm_debug_moma.toml \ # --data.hf_dataset_name=... --data.tokenizer_path=gpt2 [model] diff --git a/configs/train/vlm_debug_mot.toml b/examples/vlm/configs/vlm_debug_mot.toml similarity index 95% rename from configs/train/vlm_debug_mot.toml rename to examples/vlm/configs/vlm_debug_mot.toml index de4bab6..4b70aaa 100644 --- a/configs/train/vlm_debug_mot.toml +++ b/examples/vlm/configs/vlm_debug_mot.toml @@ -11,7 +11,7 @@ # budget must cover both modalities. # # Usage: -# uv run python scripts/train.py configs/train/vlm_debug_mot.toml \ +# uv run python examples/vlm/train.py examples/vlm/configs/vlm_debug_mot.toml \ # --data.hf_dataset_name=... --data.tokenizer_path=gpt2 [model] diff --git a/configs/train/vlm_video_webvid.toml b/examples/vlm/configs/vlm_video_webvid.toml similarity index 93% rename from configs/train/vlm_video_webvid.toml rename to examples/vlm/configs/vlm_video_webvid.toml index 785b48b..b776233 100644 --- a/configs/train/vlm_video_webvid.toml +++ b/examples/vlm/configs/vlm_video_webvid.toml @@ -12,7 +12,7 @@ # 8 frames -> 8*49 = 392 visual + 64 text = 456 <= 576. # # Launch (single node, 4 GPUs): -# uv run torchrun --nproc_per_node=4 scripts/train.py configs/train/vlm_video_webvid.toml +# uv run torchrun --nproc_per_node=4 examples/vlm/train.py examples/vlm/configs/vlm_video_webvid.toml # # Quick smoke (no SigLIP download, a few clips; pair with a small step count): # ... --vision_encoder.type=random --vision_encoder.num_tokens=196 \ @@ -44,7 +44,7 @@ max_text_len = 64 freeze = [{module = "vision_encoder", frozen = true}] [video] -data_root = "/n/holylfs06/LABS/kempner_shared/Everyone/testbed/video/webvid-10m" +data_root = "path-to-webvid-10m" split = "train" fps = 2.0 max_frames = 8 diff --git a/examples/vlm/tests/test_configs.py b/examples/vlm/tests/test_configs.py new file mode 100644 index 0000000..91dc221 --- /dev/null +++ b/examples/vlm/tests/test_configs.py @@ -0,0 +1,128 @@ +"""Tests for what this example ships: its configs and its entry point.""" + +from __future__ import annotations + +import importlib.util +import sys +import tomllib +from collections.abc import Iterator +from pathlib import Path +from typing import Any + +import pytest + +from kempnerforge.config.loader import load_config + +EXAMPLE_ROOT = Path(__file__).resolve().parents[1] +CONFIG_DIR = EXAMPLE_ROOT / "configs" +CONFIGS = sorted(CONFIG_DIR.glob("*.toml")) +CONFIG_IDS = [p.name for p in CONFIGS] + + +def _strings(node: Any, key_path: str = "") -> Iterator[tuple[str, str]]: + if isinstance(node, dict): + for key, value in node.items(): + yield from _strings(value, f"{key_path}.{key}" if key_path else key) + elif isinstance(node, list): + for index, value in enumerate(node): + yield from _strings(value, f"{key_path}[{index}]") + elif isinstance(node, str): + yield key_path, node + + +def _assert_no_machine_specific_paths(path: Path) -> None: + raw = tomllib.loads(path.read_text()) + for key, value in _strings(raw): + assert not value.startswith("/"), f"{path.name}: {key} is an absolute path: {value!r}" + assert "~" not in value, f"{path.name}: {key} references a home directory: {value!r}" + + +def test_config_dir_is_populated() -> None: + """Refuse rather than pass vacuously: an empty glob would make every + parametrized case below disappear silently if the directory moved.""" + assert CONFIGS, f"no configs found under {CONFIG_DIR}" + + +@pytest.mark.parametrize("path", CONFIGS, ids=CONFIG_IDS) +def test_config_loads_and_validates(path: Path) -> None: + config = load_config(str(path), cli_args=[]) + assert config.is_vlm is True + config.validate(world_size=4) + + +@pytest.mark.parametrize("path", CONFIGS, ids=CONFIG_IDS) +def test_config_has_no_machine_specific_paths(path: Path) -> None: + _assert_no_machine_specific_paths(path) + + +def test_machine_specific_path_check_fires(tmp_path: Path) -> None: + """Self-test for the scan above, so a passing sweep means something.""" + bad = tmp_path / "bad.toml" + bad.write_text('[video]\ndata_root = "/absolute/cluster/share/webvid-10m"\n') + with pytest.raises(AssertionError, match="absolute path"): + _assert_no_machine_specific_paths(bad) + + home = tmp_path / "home.toml" + home.write_text('[checkpoint]\ndir = "~/scratch/run"\n') + with pytest.raises(AssertionError, match="home directory"): + _assert_no_machine_specific_paths(home) + + +def test_vlm_debug_toml() -> None: + """Regression: parallel [vision_encoder] / [adapter] / [vlm] tables load + correctly, and list[FreezeSpec] inside VLMConfig instantiates each freeze + entry via __post_init__.""" + config = load_config(str(CONFIG_DIR / "vlm_debug.toml"), cli_args=[]) + assert config.is_vlm is True + assert config.vision_encoder is not None + assert config.vlm is not None + assert config.vision_encoder.type == "random" + assert config.vision_encoder.num_tokens == 64 + assert len(config.vlm.freeze) == 1 + assert config.vlm.freeze[0].module == "vision_encoder" + assert config.vlm.freeze[0].frozen is True + config.validate(world_size=1) + + +def test_vlm_7b_siglip2_toml() -> None: + config = load_config(str(CONFIG_DIR / "vlm_7b_siglip2.toml"), cli_args=[]) + assert config.is_vlm is True + assert config.vision_encoder is not None + assert config.vlm is not None + assert config.vision_encoder.type == "siglip2" + # num_tokens defaults to 0 = "infer from encoder at build time". The + # encoder probes 196 (14x14 patches, no CLS) for this path; the build-time + # max_seq_len cross-check in build_vlm_wrapper enforces + # 196 + 2048 = 2244 <= max_seq_len=2304. + assert config.vision_encoder.num_tokens == 0 + config.validate(world_size=4) + + +def _load_entry_point() -> Any: + spec = importlib.util.spec_from_file_location("_vlm_example_train", EXAMPLE_ROOT / "train.py") + assert spec is not None and spec.loader is not None + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + return module + + +def test_entry_point_passes_the_loaded_config_through(monkeypatch: pytest.MonkeyPatch) -> None: + module = _load_entry_point() + seen: dict[str, Any] = {} + monkeypatch.setattr(module, "run_training", lambda config: seen.update(config=config)) + monkeypatch.setattr( + sys, + "argv", + ["train.py", str(CONFIG_DIR / "vlm_debug.toml"), "--train.max_steps=3"], + ) + module.main() + assert seen["config"].is_vlm is True + assert seen["config"].train.max_steps == 3 + + +def test_entry_point_requires_a_config(monkeypatch: pytest.MonkeyPatch) -> None: + module = _load_entry_point() + monkeypatch.setattr(sys, "argv", ["train.py"]) + with pytest.raises(SystemExit) as exc: + module.main() + assert exc.value.code == 1 diff --git a/examples/vlm/train.py b/examples/vlm/train.py new file mode 100644 index 0000000..8ee6f13 --- /dev/null +++ b/examples/vlm/train.py @@ -0,0 +1,39 @@ +#!/usr/bin/env python3 +"""Training entry point for the VLM example. + +Parses argv, loads the config, and hands off to the core training scaffold. +The VLM step body still lives in core, so ``run_training`` selects it from +``config.is_vlm``; when it moves out here it becomes a ``step_fn=`` argument. + +Usage: + # Single GPU + uv run python examples/vlm/train.py examples/vlm/configs/vlm_debug.toml + + # Multi-GPU (single node, via torchrun) + uv run torchrun --nproc_per_node=4 examples/vlm/train.py \ + examples/vlm/configs/vlm_7b_siglip2.toml + + # With overrides + uv run python examples/vlm/train.py examples/vlm/configs/vlm_debug.toml \ + --train.max_steps=20 +""" + +from __future__ import annotations + +import sys + +from kempnerforge.config.loader import load_config +from kempnerforge.training import run_training + + +def main() -> None: + if len(sys.argv) < 2: + print("Usage: train.py [--section.key=value ...]") + sys.exit(1) + + config = load_config(sys.argv[1], cli_args=sys.argv[2:]) + run_training(config) + + +if __name__ == "__main__": + main() diff --git a/kempnerforge/model/vision.py b/kempnerforge/model/vision.py index fa3bc66..c785aef 100644 --- a/kempnerforge/model/vision.py +++ b/kempnerforge/model/vision.py @@ -47,7 +47,7 @@ class RandomVisionEncoder(VisionEncoder): same image produces the same tokens across calls; independent of model weights so it works under FSDP2 without sharding a real encoder. - Used in tests and the ``vlm_debug.toml`` smoke config. + Used in tests and in VLM smoke configs. """ def __init__(self, num_tokens: int = 16, feature_dim: int = 768, seed: int = 0) -> None: diff --git a/tests/unit/test_config.py b/tests/unit/test_config.py index b46919b..cd71b98 100644 --- a/tests/unit/test_config.py +++ b/tests/unit/test_config.py @@ -740,34 +740,50 @@ def test_toml_field_typo_raises(self, tmp_path): with pytest.raises(ValueError, match="Unknown config keys.*dimm"): load_config(str(bad_toml), cli_args=[]) - def test_load_vlm_debug_toml(self): - """Regression: parallel [vision_encoder] / [adapter] / [vlm] tables - load correctly, and list[FreezeSpec] inside VLMConfig instantiates - each freeze entry via __post_init__.""" - config = load_config("configs/train/vlm_debug.toml", cli_args=[]) + def test_vlm_freeze_list_instantiates_specs(self, tmp_path): + """``VLMConfig.freeze`` is ``list[FreezeSpec]``; the loader must + instantiate each entry so ``FreezeSpec.__post_init__`` runs, and the + parallel [vision_encoder] / [adapter] / [vlm] tables must all land.""" + toml = tmp_path / "vlm.toml" + toml.write_text( + """ +[model] +dim = 64 +n_layers = 2 +n_heads = 4 +n_kv_heads = 4 +vocab_size = 256 +max_seq_len = 96 + +[train] +seq_len = 96 +batch_size = 2 + +[vision_encoder] +type = "random" +num_tokens = 16 + +[adapter] +type = "mlp_2layer" + +[vlm] +arch = "joint_decoder" +max_text_len = 64 +freeze = [{module = "vision_encoder", frozen = true}] +""" + ) + config = load_config(str(toml), cli_args=[]) assert config.is_vlm is True assert config.vision_encoder is not None - assert config.vlm is not None assert config.vision_encoder.type == "random" - assert config.vision_encoder.num_tokens == 64 + assert config.adapter is not None + assert config.adapter.type == "mlp_2layer" + assert config.vlm is not None assert len(config.vlm.freeze) == 1 assert config.vlm.freeze[0].module == "vision_encoder" assert config.vlm.freeze[0].frozen is True config.validate(world_size=1) - def test_load_vlm_7b_siglip2_toml(self): - config = load_config("configs/train/vlm_7b_siglip2.toml", cli_args=[]) - assert config.is_vlm is True - assert config.vision_encoder is not None - assert config.vlm is not None - assert config.vision_encoder.type == "siglip2" - # num_tokens defaults to 0 = "infer from encoder at build time". - # The encoder probes 196 (14x14 patches, no CLS) for this path; the - # build-time max_seq_len cross-check in build_vlm_wrapper enforces - # 196 + 2048 = 2244 <= max_seq_len=2304. - assert config.vision_encoder.num_tokens == 0 - config.validate(world_size=4) - def test_vlm_freeze_schedule_loads_variadic_tuple(self, tmp_path): """``FreezeStage.specs`` is ``tuple[FreezeSpec, ...]``; the loader's variadic-tuple path must instantiate each spec dict via From 511de8c3839c57a87d5be7d95891a7391bb431af Mon Sep 17 00:00:00 2001 From: amazloumi Date: Wed, 26 Aug 2026 13:32:28 -0400 Subject: [PATCH 2/3] Assert the freeze list against a config that declares one The vlm_debug preset has no [adapter] table and no freeze key, so the freeze assertions there restated a dataclass default and the docstring described coverage the test did not have. vlm_video_webvid is the only preset with both, so the list[FreezeSpec] check moves there. Also repoints the eval how-to and harness usage strings at the moved config paths; vlm_jd.toml has never existed, so those examples name vlm_7b.toml. --- examples/vlm/eval/README.md | 6 +++--- examples/vlm/eval/vlm_eval_harness.py | 4 ++-- examples/vlm/tests/test_configs.py | 17 +++++++++++------ 3 files changed, 16 insertions(+), 11 deletions(-) diff --git a/examples/vlm/eval/README.md b/examples/vlm/eval/README.md index 84e521d..d6ed743 100644 --- a/examples/vlm/eval/README.md +++ b/examples/vlm/eval/README.md @@ -71,14 +71,14 @@ required. (Image-only evaluation does not need this group.) ```bash # One task, write results JSON uv run python examples/vlm/eval/vlm_eval_harness.py \ - --config configs/train/vlm_jd.toml \ + --config examples/vlm/configs/vlm_7b.toml \ --checkpoint checkpoints/vlm/step_10000 \ --tasks mmmu_val \ --output results/vlm_step_10000.json # Several tasks, quick partial run (4 examples per task) uv run python examples/vlm/eval/vlm_eval_harness.py \ - --config configs/train/vlm_jd.toml \ + --config examples/vlm/configs/vlm_7b.toml \ --checkpoint checkpoints/vlm/step_10000 \ --tasks mmmu_val,mmbench_en_dev,scienceqa_img \ --limit 4 @@ -115,7 +115,7 @@ video group (see [Install lmms-eval](#install-lmms-eval)). ```bash uv run python examples/vlm/eval/vlm_eval_harness.py \ - --config configs/train/vlm_video_webvid.toml \ + --config examples/vlm/configs/vlm_video_webvid.toml \ --checkpoint checkpoints/vlm_video/step_10000 \ --tasks \ --limit 4 diff --git a/examples/vlm/eval/vlm_eval_harness.py b/examples/vlm/eval/vlm_eval_harness.py index f1bc6cb..0f25df3 100644 --- a/examples/vlm/eval/vlm_eval_harness.py +++ b/examples/vlm/eval/vlm_eval_harness.py @@ -21,14 +21,14 @@ Usage: uv run python examples/vlm/eval/vlm_eval_harness.py \ - --config configs/train/vlm_jd.toml \ + --config examples/vlm/configs/vlm_7b.toml \ --checkpoint checkpoints/vlm/step_10000 \ --tasks mmmu_val \ --output results/vlm_step_10000.json # Quick partial run (4 examples per task) uv run python examples/vlm/eval/vlm_eval_harness.py \ - --config configs/train/vlm_jd.toml \ + --config examples/vlm/configs/vlm_7b.toml \ --checkpoint checkpoints/vlm/step_10000 \ --tasks mmmu_val,mmbench_en_dev \ --limit 4 diff --git a/examples/vlm/tests/test_configs.py b/examples/vlm/tests/test_configs.py index 91dc221..98d15bf 100644 --- a/examples/vlm/tests/test_configs.py +++ b/examples/vlm/tests/test_configs.py @@ -69,21 +69,26 @@ def test_machine_specific_path_check_fires(tmp_path: Path) -> None: def test_vlm_debug_toml() -> None: - """Regression: parallel [vision_encoder] / [adapter] / [vlm] tables load - correctly, and list[FreezeSpec] inside VLMConfig instantiates each freeze - entry via __post_init__.""" + """Parallel [vision_encoder] / [vlm] tables each load into their own config.""" config = load_config(str(CONFIG_DIR / "vlm_debug.toml"), cli_args=[]) assert config.is_vlm is True assert config.vision_encoder is not None assert config.vlm is not None assert config.vision_encoder.type == "random" assert config.vision_encoder.num_tokens == 64 - assert len(config.vlm.freeze) == 1 - assert config.vlm.freeze[0].module == "vision_encoder" - assert config.vlm.freeze[0].frozen is True config.validate(world_size=1) +def test_vlm_video_webvid_toml() -> None: + """The only preset declaring both an [adapter] table and a TOML freeze list, + so it is the one that exercises list[FreezeSpec] instantiation.""" + config = load_config(str(CONFIG_DIR / "vlm_video_webvid.toml"), cli_args=[]) + assert config.adapter is not None + assert config.vlm is not None + # Attribute access fails if the entries stayed raw dicts. + assert [(f.module, f.frozen) for f in config.vlm.freeze] == [("vision_encoder", True)] + + def test_vlm_7b_siglip2_toml() -> None: config = load_config(str(CONFIG_DIR / "vlm_7b_siglip2.toml"), cli_args=[]) assert config.is_vlm is True From 0d8961c643f67dcfb0a91372be830953ecae32d6 Mon Sep 17 00:00:00 2001 From: amazloumi Date: Wed, 26 Aug 2026 15:03:40 -0400 Subject: [PATCH 3/3] Assert a freeze list the field default cannot produce The core loader test declared freeze = [{vision_encoder, true}], which is exactly VLMConfig.freeze's default_factory, so it passed whether or not the loader read the TOML. It now declares two entries with non-default modules and flags. FreezeSpec has no __post_init__, so the docstring no longer claims one runs; what the test proves is that the entries become FreezeSpec instances rather than staying raw dicts. --- tests/unit/test_config.py | 20 +++++++++++++------- 1 file changed, 13 insertions(+), 7 deletions(-) diff --git a/tests/unit/test_config.py b/tests/unit/test_config.py index cd71b98..35ad244 100644 --- a/tests/unit/test_config.py +++ b/tests/unit/test_config.py @@ -741,9 +741,10 @@ def test_toml_field_typo_raises(self, tmp_path): load_config(str(bad_toml), cli_args=[]) def test_vlm_freeze_list_instantiates_specs(self, tmp_path): - """``VLMConfig.freeze`` is ``list[FreezeSpec]``; the loader must - instantiate each entry so ``FreezeSpec.__post_init__`` runs, and the - parallel [vision_encoder] / [adapter] / [vlm] tables must all land.""" + """``VLMConfig.freeze`` is ``list[FreezeSpec]``; the loader must turn + each TOML table into a ``FreezeSpec``, and the parallel + [vision_encoder] / [adapter] / [vlm] tables must all land. The entries + differ from the field default, so a no-op loader fails here.""" toml = tmp_path / "vlm.toml" toml.write_text( """ @@ -769,7 +770,10 @@ def test_vlm_freeze_list_instantiates_specs(self, tmp_path): [vlm] arch = "joint_decoder" max_text_len = 64 -freeze = [{module = "vision_encoder", frozen = true}] +freeze = [ + {module = "adapter", frozen = true}, + {module = "transformer", frozen = false}, +] """ ) config = load_config(str(toml), cli_args=[]) @@ -779,9 +783,11 @@ def test_vlm_freeze_list_instantiates_specs(self, tmp_path): assert config.adapter is not None assert config.adapter.type == "mlp_2layer" assert config.vlm is not None - assert len(config.vlm.freeze) == 1 - assert config.vlm.freeze[0].module == "vision_encoder" - assert config.vlm.freeze[0].frozen is True + # Attribute access fails if the entries stayed raw dicts. + assert [(f.module, f.frozen) for f in config.vlm.freeze] == [ + ("adapter", True), + ("transformer", False), + ] config.validate(world_size=1) def test_vlm_freeze_schedule_loads_variadic_tuple(self, tmp_path):