Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
23 commits
Select commit Hold shift + click to select a range
8fcfa3e
Add 256k Qwen3.8-27B definition and multi-node Miles trainer support
micahtyong Sep 18, 2026
5696e7a
Require TP groups to tile a node in multi-node configs
micahtyong Sep 18, 2026
7fa4d1d
Fix multi-node checkpoint capture publication
micahtyong Sep 18, 2026
71e3897
Refresh multi-node trainer assets before Ray joins
micahtyong Sep 18, 2026
c6c4267
Widen the 256k trainer's collective timeout
micahtyong Sep 18, 2026
49d8175
Cap coalesced forward_backward batches on the 256k rung
micahtyong Sep 18, 2026
4d3c22b
Align 256k sequences to the context/tensor-parallel layout
micahtyong Sep 18, 2026
cbed492
Load actor module under its package name in test so relative profilin…
micahtyong Sep 18, 2026
55e941e
Elect checkpoint-commit representatives by Modal task id
micahtyong Sep 21, 2026
aef7148
Publish multi-node checkpoints with every node's shards
micahtyong Sep 21, 2026
c467cb4
Fail checkpoint volume sync on every rank instead of hanging
micahtyong Sep 21, 2026
192b95e
Give multi-node trainer actors the checkpoint volume name
micahtyong Sep 22, 2026
3ece760
Publish checkpoints from the volume's committed state
micahtyong Sep 22, 2026
faaaecb
Stop refreshing a node's volume view after committing checkpoints
micahtyong Sep 22, 2026
186c33d
Keep rank zero's own files in its view when publishing a checkpoint
micahtyong Sep 22, 2026
0f5aa78
Fail a checkpoint publish that is missing a peer's shards
micahtyong Sep 22, 2026
b7d53c0
Install a checkpoint capture from committed volume state
micahtyong Sep 22, 2026
035d3e4
Add the two-node 256k Qwen3.8-27B definition
micahtyong Sep 22, 2026
f2c12ab
Trim checkpoint publication docstrings
micahtyong Sep 22, 2026
c795591
Check node count when waiting for the Ray cluster
micahtyong Sep 22, 2026
52753d5
Finalize the two-node 256k definition registry
micahtyong Sep 22, 2026
1607559
Rename the two-node 256k definition symbols
micahtyong Sep 22, 2026
04a8c7a
Shorten _align_row docstring per review
micahtyong Sep 22, 2026
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
23 changes: 20 additions & 3 deletions src/lilo/backends/miles_config.py
Original file line number Diff line number Diff line change
Expand Up @@ -27,6 +27,7 @@ class MilesBackendConfig:
hf_checkpoint: str
model_type: str
actor_num_gpus_per_node: int
actor_num_nodes: int = 1
tensor_model_parallel_size: int = 1
context_parallel_size: int = 1
expert_model_parallel_size: int = 1
Expand All @@ -43,18 +44,25 @@ class MilesBackendConfig:
"output_layer",
)
max_tokens_per_gpu: int = 8192
align_sequences_to_parallel_layout: bool = False
Comment thread
devin-ai-integration[bot] marked this conversation as resolved.
extra_args: tuple[str, ...] = ()

@property
def world_size(self) -> int:
return self.actor_num_gpus_per_node
return self.actor_num_nodes * self.actor_num_gpus_per_node

@property
def data_parallel_size(self) -> int:
return self.world_size // (
self.tensor_model_parallel_size * self.context_parallel_size
)

@property
def sequence_alignment(self) -> int:
if not self.align_sequences_to_parallel_layout:
return 1
return 2 * self.context_parallel_size * self.tensor_model_parallel_size

@property
def peft_target_modules(self) -> tuple[str, ...]:
targets: list[str] = []
Expand All @@ -68,6 +76,7 @@ def peft_target_modules(self) -> tuple[str, ...]:
def validate(self) -> None:
positive = {
"actor_num_gpus_per_node": self.actor_num_gpus_per_node,
"actor_num_nodes": self.actor_num_nodes,
"tensor_model_parallel_size": self.tensor_model_parallel_size,
"context_parallel_size": self.context_parallel_size,
"expert_model_parallel_size": self.expert_model_parallel_size,
Expand Down Expand Up @@ -98,9 +107,17 @@ def validate(self) -> None:
!= 0
):
raise ValueError(
"actor_num_gpus_per_node must be a multiple of "
"actor_num_nodes * actor_num_gpus_per_node must be a multiple of "
"tensor_model_parallel_size * context_parallel_size"
)
if (
self.actor_num_nodes > 1
and self.actor_num_gpus_per_node % self.tensor_model_parallel_size != 0
):
raise ValueError(
"tensor_model_parallel_size must evenly divide "
"actor_num_gpus_per_node on each node"
)

def miles_arguments(self) -> list[str]:
"""Arguments owned by the integration, excluding model architecture flags."""
Expand All @@ -120,7 +137,7 @@ def miles_arguments(self) -> list[str]:
"--rollout-num-gpus",
"0",
"--actor-num-nodes",
"1",
str(self.actor_num_nodes),
"--actor-num-gpus-per-node",
str(self.actor_num_gpus_per_node),
"--multi-lora-n-adapters",
Expand Down
60 changes: 58 additions & 2 deletions src/lilo/backends/miles_lora.py
Original file line number Diff line number Diff line change
Expand Up @@ -171,7 +171,11 @@ def forward_backward(
):
self._stop_profiling()
with self._timer.phase("prepare_batch", step, model_id=batch.items[0].model_id):
prepared = prepare_batch(batch, self.job_to_slot)
prepared = prepare_batch(
batch,
self.job_to_slot,
sequence_alignment=self.config.sequence_alignment,
)
phase_name = "forward_only" if batch.forward_only else "forward_backward"
with (
self._timer.phase(phase_name, step, model_id=batch.items[0].model_id),
Expand Down Expand Up @@ -332,7 +336,12 @@ def persist_checkpoint(
),
self._record("lilo/persist_checkpoint"),
):
_install_directory(capture["path"], target, overwrite=overwrite)
_install_capture(
capture["path"],
target,
overwrite=overwrite,
world_size=self.config.world_size,
)
_commit_volume(os.environ.get("LILO_CHECKPOINT_VOLUME"))
return str(target)
finally:
Expand Down Expand Up @@ -643,6 +652,53 @@ def _adam_parameters(adam: AdamParams) -> dict[str, float]:
return values


def _install_capture(
source: Path, target: Path, *, overwrite: bool, world_size: int
) -> None:
"""Install a capture whose shards were written across several nodes.

Each trainer node writes its shards into its own view of the checkpoint
volume, so copying the capture out of this container's filesystem would
install only the shards of the node it shares — a checkpoint that loads
on no rank but the ones that wrote it. The committed volume state holds
every node's shards, and a server-side copy from it needs no local view.
"""
Comment thread
micahtyong marked this conversation as resolved.
volume_name = os.environ.get("LILO_CHECKPOINT_VOLUME")
root = Path(os.environ.get("LILO_CHECKPOINT_ROOT") or "/checkpoints")
try:
relative = (str(source.relative_to(root)), str(target.relative_to(root)))
except ValueError:
relative = None
if volume_name is None or relative is None:
_install_directory(source, target, overwrite=overwrite)
return

import modal

volume = modal.Volume.from_name(volume_name)
volume.commit()
entries = [entry.path for entry in volume.listdir(relative[0])]
shards = [path for path in entries if path.endswith(".distcp")]
if shards and len(shards) < world_size:
raise RuntimeError(
f"capture {relative[0]} holds {len(shards)} of {world_size} shards; "
"a checkpoint short of a node's shards fails only on the resume "
"that needs it"
)
if target.exists():
if not overwrite:
raise FileExistsError(f"checkpoint already exists: {target}")
shutil.rmtree(target)
target.mkdir(parents=True, exist_ok=True)
volume.commit()
volume.copy_files(entries, relative[1], recursive=True)
print(
f"lilo_checkpoint_install path={relative[1]} files={len(entries)} "
f"shards={len(shards)}",
flush=True,
)


def _install_directory(source: Path, target: Path, *, overwrite: bool) -> None:
target.parent.mkdir(parents=True, exist_ok=True)
temporary = target.parent / f".{target.name}.{uuid.uuid4().hex}.tmp"
Expand Down
Loading
Loading