From d0525b63ea4ca98a10b9f11c04667cf3394a27e1 Mon Sep 17 00:00:00 2001 From: micah Date: Wed, 23 Sep 2026 18:59:11 +0000 Subject: [PATCH] Remove forward_backward batch cap from 256k Qwen3.8-27B definition Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../modal/definitions/qwen3_8_27b_miles_lora_256k.py | 6 ------ 1 file changed, 6 deletions(-) diff --git a/src/lilo/providers/modal/definitions/qwen3_8_27b_miles_lora_256k.py b/src/lilo/providers/modal/definitions/qwen3_8_27b_miles_lora_256k.py index 00bd69c..85bc2ea 100644 --- a/src/lilo/providers/modal/definitions/qwen3_8_27b_miles_lora_256k.py +++ b/src/lilo/providers/modal/definitions/qwen3_8_27b_miles_lora_256k.py @@ -47,11 +47,6 @@ # all-to-all runs behind a straggler's recompute, so a rank can sit in one # collective far longer than ten minutes without anything being wrong. DISTRIBUTED_TIMEOUT_MINUTES = 120 -# Coalescing a whole rollout's datums into one Miles call turns a step into a -# single multi-thousand-collective forward_backward across both nodes, where -# one desynchronized rank wedges every process group. One datum per call keeps -# the collective chains short; gradients still accumulate until optim_step. -MAX_FORWARD_BACKWARD_BATCH = 1 ROLLOUT_GPU_TYPE = "H200" ROLLOUT_GPUS = 4 @@ -222,7 +217,6 @@ def run_trainer( nproc=1, max_models=max_models, sampler_persistence_concurrency=8, - max_forward_backward_batch=MAX_FORWARD_BACKWARD_BATCH, )