From 145af1222dd3449cd94c096fcfc0f37748fc516a Mon Sep 17 00:00:00 2001 From: Kevin Heye Date: Thu, 17 Sep 2026 10:17:17 +0200 Subject: [PATCH 01/12] Let probe_batch_size cover a probe that builds a model probe_batch_size was written for a measurement that is a single call, and says so. Every training probe needs two things it does not offer, so all three of them -- dfine_node, yolov5_node and classification_node -- reimplement the same composition of reserve_margin, measured_fits and find_batch_size instead. The one caller left is dfine_node's detection pass. Forward on_out_of_memory, which measured_fits already accepts, so a caller can drop what a failed trial left on the card. Add minimum, for a step that cannot run on a single sample at all: BatchNorm over a 1x1 feature map, or a training whose validation halves the batch. The search then runs in units of the minimum, and a failure at that size reports it rather than one. yolov5_node and classification_node had each grown their own copy of that rescaling. A minimum of one searches exactly as before, so nothing that calls this today changes. Co-Authored-By: Claude Opus 5 --- AGENTS.md | 16 ++++-- learning_loop_node/tests/unit/test_cuda.py | 67 ++++++++++++++++++++++ learning_loop_node/trainer/cuda.py | 29 +++++++--- 3 files changed, 99 insertions(+), 13 deletions(-) diff --git a/AGENTS.md b/AGENTS.md index 716853ff..629b148f 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -66,12 +66,16 @@ brings its own way of running a step. `macro_f1` scores the confusion matrix about a batch-size probe that torch has to answer. `usable_memory_bytes` and `limit_cuda_memory` turn a `--vram-limit-gb` setting into the budget a probe measures against and the cap that holds the process to it, and capping an allocator has no NVML equivalent. `probe_batch_size` is the -whole probe for a node whose measurement is a single call — it resolves the limit, falls back -without a card, holds the safety margin and runs the search. A node that must build a throwaway -model first reserves the margin before building it, and so composes the same pieces itself: -`reserve_margin`, `measured_fits` and `find_batch_size`. `measured_fits` is where an -out-of-memory failure is told from a bug — both arrive as the same exception types, and a probe -that confuses them reports the smallest batch size as the card's fault. +whole probe except the step itself — it resolves the limit, falls back without a card, holds the +safety margin, runs the search and releases what the trials left behind. `on_out_of_memory` is how +a node that builds a throwaway model drops an optimizer's gradients after a failed trial, and +`minimum` is for a step that cannot run on a single sample at all: BatchNorm over a 1x1 feature +map, or a training whose validation halves the batch. Only a node that needs the margin claimed +*before* it builds its model still composes `reserve_margin`, `measured_fits` and +`find_batch_size` itself — `dfine_node` does, so that a model too large for the budget fails while +it is being built. `measured_fits` is where an out-of-memory failure is told from a bug — both +arrive as the same exception types, and a probe that confuses them reports the smallest batch size +as the card's fault. It imports torch, the package does **not** declare it, and only a trainer imports the module — so the library keeps working where nothing trains. Its unit test installs a stand-in under the name diff --git a/learning_loop_node/tests/unit/test_cuda.py b/learning_loop_node/tests/unit/test_cuda.py index c700f3f2..b822746f 100644 --- a/learning_loop_node/tests/unit/test_cuda.py +++ b/learning_loop_node/tests/unit/test_cuda.py @@ -133,6 +133,73 @@ def test_a_probe_without_a_gpu_does_not_run_the_step(load): assert not fake.allocated, 'and no margin claimed on a card that is not there' +# --- a minimum above one --- + +def test_a_minimum_keeps_the_search_off_the_sizes_below_it(load): + cuda, fake = load() + ran: list[int] = [] + assert cuda.probe_batch_size(_fits_up_to(16, fake, ran), limit=64, minimum=2) == 16 + assert ran == [2, 4, 8, 16, 32], 'the smallest trial is the minimum, not one' + + +def test_a_minimum_is_rounded_down_to_a_power_of_two(load): + cuda, fake = load() + ran: list[int] = [] + cuda.probe_batch_size(_fits_up_to(64, fake, ran), limit=64, minimum=3) + assert ran[0] == 2 + + +def test_a_minimum_of_one_searches_exactly_as_before(load): + cuda, fake = load() + ran: list[int] = [] + assert cuda.probe_batch_size(_fits_up_to(16, fake, ran), limit=64, minimum=1) == 16 + assert ran == [1, 2, 4, 8, 16, 32] + + +def test_a_minimum_that_does_not_fit_reports_its_own_size(load): + cuda, fake = load() + ran: list[int] = [] + with pytest.raises(InsufficientMemoryError, match='batch size 4'): + cuda.probe_batch_size(_fits_up_to(2, fake, ran), limit=32, minimum=4) + + +def test_a_minimum_above_the_limit_still_gets_tried(load): + cuda, fake = load() + ran: list[int] = [] + assert cuda.probe_batch_size(_fits_up_to(64, fake, ran), limit=2, minimum=8) == 8 + assert ran == [8] + + +def test_without_a_gpu_the_fallback_respects_the_minimum(load): + cuda, fake = load(cuda_available=False) + ran: list[int] = [] + assert cuda.probe_batch_size(_fits_up_to(1024, fake, ran), limit=64, minimum=16) == 16 + assert not ran + + +# --- cleaning up after a trial that did not fit --- + +def test_the_out_of_memory_hook_runs_after_every_failed_trial(load): + cuda, fake = load() + ran: list[int] = [] + dropped: list[int] = [] + cuda.probe_batch_size(_fits_up_to(4, fake, ran), limit=32, + on_out_of_memory=lambda: dropped.append(len(ran))) + assert dropped == [4], 'once, after the single trial that went over' + + +def test_the_out_of_memory_hook_does_not_run_for_a_bug(load): + cuda, _ = load() + dropped: list[int] = [] + + def run_batch(_: int) -> None: + raise RuntimeError('a real bug') + + with pytest.raises(RuntimeError, match='a real bug'): + cuda.probe_batch_size(run_batch, limit=32, on_out_of_memory=lambda: dropped.append(1)) + assert not dropped + + def test_a_batch_size_of_one_that_does_not_fit_is_an_error(load): cuda, fake = load() ran: list[int] = [] diff --git a/learning_loop_node/trainer/cuda.py b/learning_loop_node/trainer/cuda.py index 1b126bb0..e59e38b6 100644 --- a/learning_loop_node/trainer/cuda.py +++ b/learning_loop_node/trainer/cuda.py @@ -16,6 +16,7 @@ import torch from .batch_size import MAX_BATCH_SIZE, find_batch_size, is_out_of_memory, no_gpu_batch_size, smaller_pot +from .exceptions import InsufficientMemoryError logger = logging.getLogger(__name__) @@ -24,27 +25,41 @@ def probe_batch_size(run_batch: Callable[[int], str | None], *, probe: str = 'batch-size probe', - limit: int = 0, vram_limit_gb: float = 0) -> int: + limit: int = 0, minimum: int = 1, vram_limit_gb: float = 0, + on_out_of_memory: Callable[[], None] | None = None) -> int: """Find the largest power-of-two batch size ``run_batch`` fits into. - For a probe whose measurement is one call. A probe that has to build a throwaway model first - composes :func:`reserve_margin`, :func:`measured_fits` and ``find_batch_size`` itself. + This is the whole of a probe except the step itself: the margin, the search, telling an + out-of-memory failure from a bug, and releasing what the trials left behind. A caller that + builds a throwaway model supplies ``on_out_of_memory`` to drop what a failed trial left on the + card; only a caller that needs the margin claimed *before* it builds that model has to compose + :func:`reserve_margin`, :func:`measured_fits` and ``find_batch_size`` itself. :param run_batch: Runs the batch; may return a detail to append to the log line. :param probe: Names this probe in the log, so a node running several stays readable. :param limit: Caps the search, rounded down to a power of two; 0 means :data:`~learning_loop_node.trainer.batch_size.MAX_BATCH_SIZE`. + :param minimum: Smallest batch size to try, rounded down to a power of two. Raise it above one + for a step that cannot run on a single sample at all — BatchNorm over a 1x1 feature map, a + validation pass that halves the batch — where a failure at one says nothing about memory. :param vram_limit_gb: The budget the safety margin is a share of; 0 means the whole card. - :raises InsufficientMemoryError: If not even a batch size of 1 fits. + :param on_out_of_memory: Runs after a trial ran out of memory, to drop what it left behind + (an optimizer's gradients, say). + :raises InsufficientMemoryError: If not even ``minimum`` fits. """ - limit = smaller_pot(limit or MAX_BATCH_SIZE) + minimum = smaller_pot(max(1, minimum)) + limit = max(smaller_pot(limit or MAX_BATCH_SIZE), minimum) if not torch.cuda.is_available(): - return no_gpu_batch_size(limit, probe) + return max(minimum, no_gpu_batch_size(limit, probe)) margin = reserve_margin(vram_limit_gb, probe=probe) try: - chosen = find_batch_size(measured_fits(run_batch, probe=probe), limit=limit) + fits = measured_fits(run_batch, probe=probe, on_out_of_memory=on_out_of_memory) + # Searched in units of the minimum, so the smallest trial is the minimum rather than one. + chosen = minimum * find_batch_size(lambda n: fits(minimum * n), limit=limit // minimum) + except InsufficientMemoryError as exc: + raise InsufficientMemoryError(f'batch size {minimum} does not fit in memory') from exc finally: del margin free_cuda_memory() From a2a707b43eeba39283bf50807ca32a6d09f49613 Mon Sep 17 00:00:00 2001 From: Kevin Heye Date: Fri, 18 Sep 2026 14:41:03 +0200 Subject: [PATCH 02/12] Settle the batch size from the hyperparameters in one call probe_batch_size leaves each trainer the same handful of lines around it: read the bound, clamp by the dataset, probe. Three nodes spell that three ways, and with it the hyperparameter name, which is a contract with the loop rather than a local choice. measure_batch_size takes it over, so a trainer is left with the one thing only it can supply: the step. One key, not two. batch_size is the largest batch a training may use, and it is measured rather than trusted. A size that does not fit backs off to the largest power of two below it, instead of starting a training that runs out of memory at epoch 30; classification_node returned such a size as given. A size that does fit is used as named even when it is not a power of two, which is what the new candidate is for: it is tried once the doubling has reached its ceiling, so the only non-power-of-two this can return is one somebody asked for and it then measured. A bound derived from the dataset is never a candidate -- samples // 8 is a heuristic, not a size anyone named. minimum and candidate sit in find_batch_size rather than in probe_batch_size, so dfine_node gets both without composing them itself; it reserves its margin before it builds its model and so cannot use probe_batch_size. The settled size is returned, not written back. Writing it into batch_size would leave a second measurement in the same process bounded by the first, which would cost classification_node the whole of its inference gain. Co-Authored-By: Claude Opus 5 --- AGENTS.md | 21 ++++- .../tests/unit/test_batch_size.py | 67 ++++++++++++- learning_loop_node/tests/unit/test_cuda.py | 93 +++++++++++++++++++ learning_loop_node/trainer/batch_size.py | 42 +++++++-- learning_loop_node/trainer/cuda.py | 72 ++++++++++---- 5 files changed, 264 insertions(+), 31 deletions(-) diff --git a/AGENTS.md b/AGENTS.md index 629b148f..0923c9cc 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -65,12 +65,25 @@ brings its own way of running a step. `macro_f1` scores the confusion matrix `trainer/cuda.py` is the one exception to that framework independence, and holds everything about a batch-size probe that torch has to answer. `usable_memory_bytes` and `limit_cuda_memory` turn a `--vram-limit-gb` setting into the budget a probe measures against and the cap that holds -the process to it, and capping an allocator has no NVML equivalent. `probe_batch_size` is the -whole probe except the step itself — it resolves the limit, falls back without a card, holds the +the process to it, and capping an allocator has no NVML equivalent. + +`measure_batch_size` is what every trainer calls, whatever shape its hyperparameters have: it +takes the requested size as an `int`, so a node parsing into a dataclass enters the same door as +one keeping a dict. `batch_size` — the key named by the `BATCH_SIZE` constant, so the three nodes +agree on the spelling — is the largest batch the training may use, and it is measured rather than +trusted: a size that fits is used as named, whether or not it is a power of two, and one that does +not becomes the largest power of two below it that does. A training that starts small beats one +that runs out of memory at epoch 30. The settled size is returned and **not** stored anywhere the +next measurement would read it, which would otherwise leave inference bounded by training. Below +it, +`probe_batch_size` is the whole probe except the step itself — it resolves the limit, falls back without a card, holds the safety margin, runs the search and releases what the trials left behind. `on_out_of_memory` is how a node that builds a throwaway model drops an optimizer's gradients after a failed trial, and -`minimum` is for a step that cannot run on a single sample at all: BatchNorm over a 1x1 feature -map, or a training whose validation halves the batch. Only a node that needs the margin claimed +`minimum` and `candidate` belong to `find_batch_size` itself, so a node composing by hand gets +them too: `minimum` is for a step that cannot run on a single sample at all — BatchNorm over a 1x1 +feature map, or a training whose validation halves the batch — and `candidate` is the one way the +search returns a size that is not a power of two, and only ever one that was named and then +measured. Only a node that needs the margin claimed *before* it builds its model still composes `reserve_margin`, `measured_fits` and `find_batch_size` itself — `dfine_node` does, so that a model too large for the budget fails while it is being built. `measured_fits` is where an out-of-memory failure is told from a bug — both diff --git a/learning_loop_node/tests/unit/test_batch_size.py b/learning_loop_node/tests/unit/test_batch_size.py index 9ad3fd3e..a724c7a5 100644 --- a/learning_loop_node/tests/unit/test_batch_size.py +++ b/learning_loop_node/tests/unit/test_batch_size.py @@ -11,6 +11,7 @@ no_gpu_batch_size, smaller_pot, ) +from ...trainer.exceptions import InsufficientMemoryError def test_the_search_doubles_up_to_the_limit(): @@ -30,6 +31,57 @@ def test_a_limit_that_is_not_a_power_of_two_is_rounded_down(): assert find_batch_size(_fits_up_to(1024), limit=1) == 1 +def test_a_minimum_keeps_the_search_off_the_sizes_below_it(): + ran: list[int] = [] + assert find_batch_size(_fits_up_to(16, ran), limit=64, minimum=2) == 16 + assert ran == [2, 4, 8, 16, 32], 'the smallest trial is the minimum, not one' + + +def test_a_minimum_is_rounded_down_to_a_power_of_two(): + ran: list[int] = [] + find_batch_size(_fits_up_to(64, ran), limit=64, minimum=3) + assert ran[0] == 2 + + +def test_a_minimum_that_does_not_fit_reports_its_own_size(): + with pytest.raises(InsufficientMemoryError, match='batch size 4'): + find_batch_size(_fits_up_to(2, []), limit=32, minimum=4) + + +def test_a_minimum_above_the_limit_still_gets_tried(): + ran: list[int] = [] + assert find_batch_size(_fits_up_to(64, ran), limit=2, minimum=8) == 8 + assert ran == [8] + + +def test_a_named_candidate_that_fits_is_used_as_named(): + ran: list[int] = [] + assert find_batch_size(_fits_up_to(24, ran), limit=24, candidate=24) == 24 + assert ran == [1, 2, 4, 8, 16, 24], 'tried only after the doubling reached its ceiling' + + +def test_a_named_candidate_that_does_not_fit_leaves_the_power_of_two(): + assert find_batch_size(_fits_up_to(16, []), limit=24, candidate=24) == 16 + + +def test_a_candidate_is_not_tried_when_memory_stopped_the_doubling_earlier(): + ran: list[int] = [] + assert find_batch_size(_fits_up_to(4, ran), limit=24, candidate=24) == 4 + assert 24 not in ran, 'if 16 does not fit, 24 cannot' + + +def test_a_candidate_above_the_limit_is_ignored(): + ran: list[int] = [] + assert find_batch_size(_fits_up_to(64, ran), limit=16, candidate=24) == 16 + assert 24 not in ran + + +def test_a_candidate_that_is_already_a_power_of_two_changes_nothing(): + ran: list[int] = [] + assert find_batch_size(_fits_up_to(64, ran), limit=16, candidate=16) == 16 + assert ran.count(16) == 1, 'the doubling already tried it' + + def test_a_machine_that_cannot_take_one_sample_is_an_error(): fits, calls = _recording(0) with pytest.raises(RuntimeError, match='batch size 1 does not fit'): @@ -110,6 +162,15 @@ def fits(batch_size: int) -> bool: return fits, calls -def _fits_up_to(capacity: int) -> Callable[[int], bool]: - fits, _ = _recording(capacity) - return fits +def _fits_up_to(capacity: int, ran: list[int] | None = None) -> Callable[[int], bool]: + """A `fits` predicate for a machine of `capacity`, appending every size asked about to `ran`.""" + fits, calls = _recording(capacity) + if ran is None: + return fits + + def recording_fits(batch_size: int) -> bool: + result = fits(batch_size) + ran[:] = calls + return result + + return recording_fits diff --git a/learning_loop_node/tests/unit/test_cuda.py b/learning_loop_node/tests/unit/test_cuda.py index b822746f..908cdbc3 100644 --- a/learning_loop_node/tests/unit/test_cuda.py +++ b/learning_loop_node/tests/unit/test_cuda.py @@ -133,6 +133,99 @@ def test_a_probe_without_a_gpu_does_not_run_the_step(load): assert not fake.allocated, 'and no margin claimed on a card that is not there' +# --- settling a batch size from the hyperparameters --- + +def test_a_requested_size_that_fits_is_used_as_named(load): + cuda, fake = load() + ran: list[int] = [] + assert cuda.measure_batch_size(_fits_up_to(64, fake, ran), batch_size=16) == 16 + assert ran[-1] == 16, 'and it was measured, not taken on trust' + + +def test_a_requested_size_that_does_not_fit_backs_off_instead_of_running_out_of_memory(load): + cuda, fake = load() + assert cuda.measure_batch_size(_fits_up_to(8, fake, []), batch_size=32) == 8 + + +def test_a_requested_size_is_tried_even_when_it_is_not_a_power_of_two(load): + cuda, fake = load() + ran: list[int] = [] + assert cuda.measure_batch_size(_fits_up_to(24, fake, ran), batch_size=24) == 24 + assert ran == [1, 2, 4, 8, 16, 24], 'the named size only after the search reached its ceiling' + + +def test_a_named_size_that_does_not_fit_falls_back_to_the_power_of_two_below_it(load): + cuda, fake = load() + ran: list[int] = [] + assert cuda.measure_batch_size(_fits_up_to(16, fake, ran), batch_size=24) == 16 + assert ran[-1] == 24, 'it was tried, and it did not fit' + + +def test_a_named_size_is_not_tried_when_memory_stopped_the_search_earlier(load): + cuda, fake = load() + ran: list[int] = [] + assert cuda.measure_batch_size(_fits_up_to(4, fake, ran), batch_size=24) == 4 + assert 24 not in ran, 'if 16 does not fit, 24 cannot' + + +def test_an_absent_batch_size_leaves_the_bound_to_the_library(load): + cuda, fake = load() + assert cuda.measure_batch_size(_fits_up_to(2048, fake, [])) == MAX_BATCH_SIZE + + +def test_a_batch_size_of_zero_means_measure(load): + cuda, fake = load() + ran: list[int] = [] + assert cuda.measure_batch_size(_fits_up_to(16, fake, ran), batch_size=0) == 16 + assert ran + + +def test_the_settled_size_is_only_returned(load): + cuda, fake = load() + settled = cuda.measure_batch_size(_fits_up_to(16, fake, []), batch_size=64) + assert settled == 16, 'the caller reports it; nothing here stores it where a second call would read it' + + +def test_the_dataset_bounds_the_search_as_well(load): + cuda, fake = load() + # 80 samples leave room for 10 per step, rounded down to a power of two + assert cuda.measure_batch_size(_fits_up_to(1024, fake, []), sample_count=80) == 8 + + +def test_a_dataset_bound_is_never_used_as_a_candidate(load): + cuda, fake = load() + ran: list[int] = [] + cuda.measure_batch_size(_fits_up_to(1024, fake, ran), sample_count=80) + assert 10 not in ran, 'samples // 8 is a heuristic, not a size anyone asked for' + + +def test_the_tighter_of_the_request_and_the_dataset_wins(load): + cuda, fake = load() + assert cuda.measure_batch_size(_fits_up_to(1024, fake, []), batch_size=4, + sample_count=8000) == 4 + assert cuda.measure_batch_size(_fits_up_to(1024, fake, []), batch_size=512, + sample_count=80) == 8 + + +def test_memory_still_decides_below_both_bounds(load): + cuda, fake = load() + assert cuda.measure_batch_size(_fits_up_to(2, fake, []), batch_size=64, + sample_count=8000) == 2 + + +def test_a_negative_batch_size_is_a_mistake_not_a_sentinel(load): + cuda, fake = load() + with pytest.raises(ValueError, match='batch_size'): + cuda.measure_batch_size(_fits_up_to(64, fake, []), batch_size=-1) + + +def test_the_minimum_reaches_the_probe(load): + cuda, fake = load() + ran: list[int] = [] + cuda.measure_batch_size(_fits_up_to(64, fake, ran), minimum=4) + assert ran[0] == 4 + + # --- a minimum above one --- def test_a_minimum_keeps_the_search_off_the_sizes_below_it(load): diff --git a/learning_loop_node/trainer/batch_size.py b/learning_loop_node/trainer/batch_size.py index 8fb0418c..001f281a 100644 --- a/learning_loop_node/trainer/batch_size.py +++ b/learning_loop_node/trainer/batch_size.py @@ -16,6 +16,13 @@ logger = logging.getLogger(__name__) +BATCH_SIZE = 'batch_size' +"""The hyperparameter every node reads its `measure_batch_size` argument out of. + +Named here so the nodes agree on the spelling, and here rather than in :mod:`.cuda` so that a +hyperparameter parser can read it without pulling torch in. 0 or absent means the card decides. +""" + MAX_BATCH_SIZE = 1024 """Where a search stops when its caller sets no bound of its own.""" @@ -26,20 +33,39 @@ """Fewest optimizer steps an epoch must have; :func:`dataset_limit` is derived from it.""" -def find_batch_size(fits: Callable[[int], bool], *, limit: int) -> int: - """Return the largest power-of-two batch size that fits, never exceeding ``limit``. +def find_batch_size(fits: Callable[[int], bool], *, limit: int, minimum: int = 1, + candidate: int = 0) -> int: + """Return the largest batch size that fits, never exceeding ``limit``. + + Powers of two, so equal hardware and equal hyperparameters yield an equal recipe — plus + ``candidate``, which is the one size outside that set this will return, and only when somebody + named it and it then measured. :param fits: Runs a representative probe; ``False`` on out-of-memory. - :raises InsufficientMemoryError: If not even a batch size of 1 fits. + :param limit: Upper bound; the doubling stops at the largest power of two within it. + :param minimum: Smallest size to try, rounded down to a power of two. Raise it above one for a + step that cannot run on a single sample at all — BatchNorm over a 1x1 feature map, a + validation pass that halves the batch — where a failure at one says nothing about memory. + :param candidate: An exact size, tried once the doubling has reached its ceiling, so a size + that was asked for is used as asked for rather than rounded down. Ignored unless it lies + between that ceiling and ``limit``; a bound nobody named — one derived from the dataset, + say — must not be passed here. + :raises InsufficientMemoryError: If not even ``minimum`` fits. """ - limit = smaller_pot(limit) - if not fits(1): - raise InsufficientMemoryError('batch size 1 does not fit in memory') + minimum = smaller_pot(max(1, minimum)) + bound = max(limit, minimum) + ceiling = max(smaller_pot(bound), minimum) + + if not fits(minimum): + raise InsufficientMemoryError(f'batch size {minimum} does not fit in memory') - size = 1 - while size < limit and fits(size * 2): + size = minimum + while size < ceiling and fits(size * 2): size *= 2 + if size == ceiling and ceiling < candidate <= bound and fits(candidate): + size = candidate # the doubling was not what stopped it, so the named size is reachable + return size diff --git a/learning_loop_node/trainer/cuda.py b/learning_loop_node/trainer/cuda.py index e59e38b6..6bd0ccaf 100644 --- a/learning_loop_node/trainer/cuda.py +++ b/learning_loop_node/trainer/cuda.py @@ -15,19 +15,63 @@ import torch -from .batch_size import MAX_BATCH_SIZE, find_batch_size, is_out_of_memory, no_gpu_batch_size, smaller_pot -from .exceptions import InsufficientMemoryError +from .batch_size import ( + BATCH_SIZE, + MAX_BATCH_SIZE, + dataset_limit, + find_batch_size, + is_out_of_memory, + no_gpu_batch_size, + smaller_pot, +) logger = logging.getLogger(__name__) SAFETY_MARGIN = 0.05 """Share of the budget held back while probing, against allocator fragmentation later on.""" +def measure_batch_size(run_batch: Callable[[int], str | None], *, batch_size: int = 0, + sample_count: int | None = None, probe: str = 'batch-size probe', + minimum: int = 1, vram_limit_gb: float = 0, + on_out_of_memory: Callable[[], None] | None = None) -> int: + """Settle a training's batch size against what it asked for and what the card allows. + + The whole of the decision, so that a trainer is left with only the step. Every training enters + here, whatever shape its hyperparameters have; :func:`probe_batch_size` is for a probe with no + requested size to honour, such as a detection pass bounded only by how many images there are. + + :param run_batch: Runs the batch; may return a detail to append to the log line. + :param batch_size: What the training asked for, as carried in the :data:`BATCH_SIZE` + hyperparameter: the largest batch it may use, measured rather than trusted. A size that + fits is used as asked for, whether or not it is a power of two; one that does not becomes + the largest power of two below it that does, rather than a training that runs out of memory + partway through. 0 means the card decides alone. + :param sample_count: Samples in the training split, when the caller knows it. The search is + then bounded so an epoch keeps enough optimizer steps to mean something, and so a loader + that drops its last partial batch cannot end up with no batch at all. This bound is never + used as the exact candidate -- it is a heuristic, not a size anyone named. + :param minimum: Smallest size to try; see :func:`probe_batch_size`. + :param vram_limit_gb: The budget the safety margin is a share of; 0 means the whole card. + :param on_out_of_memory: Runs after a trial ran out of memory, to drop what it left behind. + :raises InsufficientMemoryError: If not even ``minimum`` fits. + :raises ValueError: If the training asked for a negative batch size. + """ + if batch_size < 0: + raise ValueError(f'{BATCH_SIZE} must be >= 0, got {batch_size}') + + limit = batch_size + if sample_count is not None: + limit = min(limit or MAX_BATCH_SIZE, dataset_limit(sample_count)) + + return probe_batch_size(run_batch, probe=probe, limit=limit, candidate=batch_size, + minimum=minimum, vram_limit_gb=vram_limit_gb, + on_out_of_memory=on_out_of_memory) + def probe_batch_size(run_batch: Callable[[int], str | None], *, probe: str = 'batch-size probe', - limit: int = 0, minimum: int = 1, vram_limit_gb: float = 0, + limit: int = 0, candidate: int = 0, minimum: int = 1, vram_limit_gb: float = 0, on_out_of_memory: Callable[[], None] | None = None) -> int: - """Find the largest power-of-two batch size ``run_batch`` fits into. + """Run :func:`~learning_loop_node.trainer.batch_size.find_batch_size` against a real card. This is the whole of a probe except the step itself: the margin, the search, telling an out-of-memory failure from a bug, and releasing what the trials left behind. A caller that @@ -37,33 +81,29 @@ def probe_batch_size(run_batch: Callable[[int], str | None], *, probe: str = 'ba :param run_batch: Runs the batch; may return a detail to append to the log line. :param probe: Names this probe in the log, so a node running several stays readable. - :param limit: Caps the search, rounded down to a power of two; 0 means + :param limit: Caps the search; 0 means :data:`~learning_loop_node.trainer.batch_size.MAX_BATCH_SIZE`. - :param minimum: Smallest batch size to try, rounded down to a power of two. Raise it above one - for a step that cannot run on a single sample at all — BatchNorm over a 1x1 feature map, a - validation pass that halves the batch — where a failure at one says nothing about memory. + :param candidate: An exact size to try once the doubling has reached its ceiling; see + ``find_batch_size``. + :param minimum: Smallest size to try; see ``find_batch_size``. :param vram_limit_gb: The budget the safety margin is a share of; 0 means the whole card. :param on_out_of_memory: Runs after a trial ran out of memory, to drop what it left behind (an optimizer's gradients, say). :raises InsufficientMemoryError: If not even ``minimum`` fits. """ - minimum = smaller_pot(max(1, minimum)) - limit = max(smaller_pot(limit or MAX_BATCH_SIZE), minimum) + bound = max(limit or MAX_BATCH_SIZE, minimum) if not torch.cuda.is_available(): - return max(minimum, no_gpu_batch_size(limit, probe)) + return max(smaller_pot(max(1, minimum)), no_gpu_batch_size(bound, probe)) margin = reserve_margin(vram_limit_gb, probe=probe) try: fits = measured_fits(run_batch, probe=probe, on_out_of_memory=on_out_of_memory) - # Searched in units of the minimum, so the smallest trial is the minimum rather than one. - chosen = minimum * find_batch_size(lambda n: fits(minimum * n), limit=limit // minimum) - except InsufficientMemoryError as exc: - raise InsufficientMemoryError(f'batch size {minimum} does not fit in memory') from exc + chosen = find_batch_size(fits, limit=bound, minimum=minimum, candidate=candidate) finally: del margin free_cuda_memory() - logger.info('%s: selected batch size %d (upper bound %d)', probe, chosen, limit) + logger.info('%s: selected batch size %d (upper bound %d)', probe, chosen, bound) return chosen From eeb04573f98af03dfba6297c8c37fbb92dfea4eb Mon Sep 17 00:00:00 2001 From: Kevin Heye Date: Wed, 23 Sep 2026 15:34:19 +0200 Subject: [PATCH 03/12] Own the settings a probing trainer repeats Three nodes now probe their batch size against this library, and each carries the same wiring around the probe in its own copy. `--vram-limit-gb` stood in three `main.py` files with the same help text, and again in four spawned scripts with another copy of it. The flag name is what `_NodeArgumentParser` derives `VRAM_LIMIT_GB` from, so the environment variable an operator writes into a `.env` was defined by copy-paste. `node_parser` takes it now, opt-in because it means nothing on a detector node, and `add_vram_limit_argument` gives a spawned script the same flag with the same words. `int(hyperparameters.get(BATCH_SIZE, 0) or 0)` stood in two nodes. The `or 0` is load-bearing -- it swallows the `None` and the `''` the loop sends for a field nobody filled in -- and it is exactly the kind of coercion that drifts, so `requested_batch_size` owns it next to the constant it reads. Co-Authored-By: Claude Opus 5 --- learning_loop_node/helpers/entrypoint.py | 17 ++++++++++++++++- .../tests/unit/test_batch_size.py | 17 +++++++++++++++++ learning_loop_node/trainer/batch_size.py | 15 ++++++++++++++- learning_loop_node/trainer/cuda.py | 12 ++++++++++++ 4 files changed, 59 insertions(+), 2 deletions(-) diff --git a/learning_loop_node/helpers/entrypoint.py b/learning_loop_node/helpers/entrypoint.py index 4c42602e..cde5831e 100644 --- a/learning_loop_node/helpers/entrypoint.py +++ b/learning_loop_node/helpers/entrypoint.py @@ -14,19 +14,34 @@ logger = logging.getLogger(__name__) +VRAM_LIMIT_GB_FLAG = '--vram-limit-gb' +"""The flag a trainer takes its GPU budget from; ``VRAM_LIMIT_GB`` follows from the name.""" -def node_parser(*, description: str, legacy_env_prefix: str = '') -> configargparse.ArgumentParser: +VRAM_LIMIT_GB_HELP = ('Gigabytes of GPU memory a training may use. The batch size is probed against this limit ' + 'instead of the whole card, so a lower limit yields a smaller batch size rather than an ' + 'out-of-memory error. Use it to share a GPU or to keep headroom against fragmentation. ' + "The limit is relative to the card's total memory, not to what is currently free. " + '0 (default) means no limit.') +"""Written once here because an operator reads it as the contract for what ``VRAM_LIMIT_GB`` does.""" + + +def node_parser(*, description: str, legacy_env_prefix: str = '', + vram_limit: bool = False) -> configargparse.ArgumentParser: """Build the parser for a node, pre-loaded with the settings every node has. :param legacy_env_prefix: A prefix an earlier version of this node required, e.g. ``'MY_DETECTOR_'``. Both spellings keep working, with a warning: the prefix on the current name (``MY_DETECTOR_NODE_HOST``) and the prefix on the flag it was originally applied to (``MY_DETECTOR_HOST``). Leave empty for a node that never used one. + :param vram_limit: Add :data:`VRAM_LIMIT_GB_FLAG`. Opt-in, because only a node that probes a + batch size has anything to do with it; on a detector node the setting means nothing. """ parser = _NodeArgumentParser(description=description, legacy_env_prefix=legacy_env_prefix) parser.add_argument('--host', default='0.0.0.0', env_var='NODE_HOST', help='Host interface to bind to') parser.add_argument('--port', type=int, default=80, env_var='NODE_PORT', help='Port to bind to') + if vram_limit: + parser.add_argument(VRAM_LIMIT_GB_FLAG, type=float, default=0, help=VRAM_LIMIT_GB_HELP) return parser diff --git a/learning_loop_node/tests/unit/test_batch_size.py b/learning_loop_node/tests/unit/test_batch_size.py index a724c7a5..43dcef5d 100644 --- a/learning_loop_node/tests/unit/test_batch_size.py +++ b/learning_loop_node/tests/unit/test_batch_size.py @@ -3,12 +3,14 @@ import pytest from ...trainer.batch_size import ( + BATCH_SIZE, MIN_TRAIN_STEPS_PER_EPOCH, batch_count, dataset_limit, find_batch_size, is_out_of_memory, no_gpu_batch_size, + requested_batch_size, smaller_pot, ) from ...trainer.exceptions import InsufficientMemoryError @@ -112,6 +114,21 @@ def test_batch_count_covers_the_whole_set_without_overshooting_by_a_batch(): assert covered - sample_count < batch_size +@pytest.mark.parametrize('empty', [{}, {BATCH_SIZE: None}, {BATCH_SIZE: ''}, {BATCH_SIZE: 0}]) +def test_an_unfilled_batch_size_means_no_bound(empty: dict): + assert requested_batch_size(empty) == 0 + + +@pytest.mark.parametrize(('value', 'expected'), [('64', 64), (64, 64), (48.0, 48)]) +def test_a_batch_size_is_read_whatever_the_loop_spelled_it_as(value: object, expected: int): + assert requested_batch_size({BATCH_SIZE: value}) == expected + + +def test_a_batch_size_that_is_not_a_number_is_an_error(): + with pytest.raises(ValueError): + requested_batch_size({BATCH_SIZE: 'lots'}) + + def test_the_dataset_limit_keeps_enough_steps_per_epoch(): for sample_count in (8, 20, 47, 100, 1000, 118_000): limit = smaller_pot(dataset_limit(sample_count)) diff --git a/learning_loop_node/trainer/batch_size.py b/learning_loop_node/trainer/batch_size.py index 001f281a..f8debcf5 100644 --- a/learning_loop_node/trainer/batch_size.py +++ b/learning_loop_node/trainer/batch_size.py @@ -10,7 +10,8 @@ """ import logging -from collections.abc import Callable +from collections.abc import Callable, Mapping +from typing import Any from .exceptions import InsufficientMemoryError @@ -69,6 +70,18 @@ def find_batch_size(fits: Callable[[int], bool], *, limit: int, minimum: int = 1 return size +def requested_batch_size(hyperparameters: Mapping[str, Any]) -> int: + """The bound a training asked for, read out of the hyperparameters the loop sent. + + A field nobody filled in arrives as absent, ``None`` or ``''`` depending on where it came + from, and all three mean the same thing: no bound of its own, the card decides alone. Read it + through here rather than reaching into the dict, so every node agrees on that. + + :raises ValueError: If the value is there but is not a number. + """ + return int(hyperparameters.get(BATCH_SIZE, 0) or 0) + + def dataset_limit(sample_count: int) -> int: """The batch size ceiling that still leaves ``MIN_TRAIN_STEPS_PER_EPOCH`` steps per epoch.""" if sample_count < 1: diff --git a/learning_loop_node/trainer/cuda.py b/learning_loop_node/trainer/cuda.py index 6bd0ccaf..9e5d6f2f 100644 --- a/learning_loop_node/trainer/cuda.py +++ b/learning_loop_node/trainer/cuda.py @@ -11,10 +11,12 @@ import gc import logging +from argparse import ArgumentParser from collections.abc import Callable import torch +from ..helpers.entrypoint import VRAM_LIMIT_GB_FLAG, VRAM_LIMIT_GB_HELP from .batch_size import ( BATCH_SIZE, MAX_BATCH_SIZE, @@ -163,6 +165,16 @@ def usable_memory_bytes(vram_limit_gb: float) -> int: return min(total_bytes, int(vram_limit_gb * 1024**3)) +def add_vram_limit_argument(parser: ArgumentParser) -> None: + """Give a spawned training script the same GPU budget flag its node has. + + The cap does not survive a spawn, so a node that probes against a budget has to hand the + number to whatever it spawns, and that process has to call :func:`limit_cuda_memory` itself. + This is the parsing half of that, spelled and documented exactly as on the node. + """ + parser.add_argument(VRAM_LIMIT_GB_FLAG, type=float, default=0, help=VRAM_LIMIT_GB_HELP) + + def limit_cuda_memory(vram_limit_gb: float) -> None: """Cap how much of the GPU this process may allocate, to ``vram_limit_gb`` gigabytes. From f6640cbb8e9e1a40e1ea39798ffa7a5f926e8f99 Mon Sep 17 00:00:00 2001 From: Kevin Heye Date: Thu, 24 Sep 2026 11:51:03 +0200 Subject: [PATCH 04/12] Read the requested batch size from max_batch_size Trainers report the batch size they settled on as `batch_size`, and the hyperparameters are stored with the training and handed to the next one. With `batch_size` also being the input, a resumed or follow-up training read an earlier card's measurement back as its own upper bound, so a card that once had to settle for 16 capped every later training at 16. The input is now `max_batch_size`, named by REQUESTED_BATCH_SIZE (renamed from BATCH_SIZE, which was not released yet), and `batch_size` stays free for reporting. Co-Authored-By: Claude Opus 5.5 --- learning_loop_node/tests/unit/test_batch_size.py | 16 ++++++++++++---- learning_loop_node/trainer/batch_size.py | 9 +++++++-- learning_loop_node/trainer/cuda.py | 14 +++++++------- 3 files changed, 26 insertions(+), 13 deletions(-) diff --git a/learning_loop_node/tests/unit/test_batch_size.py b/learning_loop_node/tests/unit/test_batch_size.py index 43dcef5d..e8660ede 100644 --- a/learning_loop_node/tests/unit/test_batch_size.py +++ b/learning_loop_node/tests/unit/test_batch_size.py @@ -3,8 +3,8 @@ import pytest from ...trainer.batch_size import ( - BATCH_SIZE, MIN_TRAIN_STEPS_PER_EPOCH, + REQUESTED_BATCH_SIZE, batch_count, dataset_limit, find_batch_size, @@ -114,19 +114,27 @@ def test_batch_count_covers_the_whole_set_without_overshooting_by_a_batch(): assert covered - sample_count < batch_size -@pytest.mark.parametrize('empty', [{}, {BATCH_SIZE: None}, {BATCH_SIZE: ''}, {BATCH_SIZE: 0}]) +@pytest.mark.parametrize('empty', [{}, {REQUESTED_BATCH_SIZE: None}, {REQUESTED_BATCH_SIZE: ''}, + {REQUESTED_BATCH_SIZE: 0}]) def test_an_unfilled_batch_size_means_no_bound(empty: dict): assert requested_batch_size(empty) == 0 +def test_the_reported_batch_size_is_never_read_as_a_bound(): + """A trainer reports its settled size as `batch_size`; the next training must not inherit it.""" + assert REQUESTED_BATCH_SIZE != 'batch_size' + assert requested_batch_size({'batch_size': 16}) == 0 + assert requested_batch_size({'batch_size': 16, REQUESTED_BATCH_SIZE: 64}) == 64 + + @pytest.mark.parametrize(('value', 'expected'), [('64', 64), (64, 64), (48.0, 48)]) def test_a_batch_size_is_read_whatever_the_loop_spelled_it_as(value: object, expected: int): - assert requested_batch_size({BATCH_SIZE: value}) == expected + assert requested_batch_size({REQUESTED_BATCH_SIZE: value}) == expected def test_a_batch_size_that_is_not_a_number_is_an_error(): with pytest.raises(ValueError): - requested_batch_size({BATCH_SIZE: 'lots'}) + requested_batch_size({REQUESTED_BATCH_SIZE: 'lots'}) def test_the_dataset_limit_keeps_enough_steps_per_epoch(): diff --git a/learning_loop_node/trainer/batch_size.py b/learning_loop_node/trainer/batch_size.py index f8debcf5..9db8d25b 100644 --- a/learning_loop_node/trainer/batch_size.py +++ b/learning_loop_node/trainer/batch_size.py @@ -17,11 +17,16 @@ logger = logging.getLogger(__name__) -BATCH_SIZE = 'batch_size' +REQUESTED_BATCH_SIZE = 'max_batch_size' """The hyperparameter every node reads its `measure_batch_size` argument out of. Named here so the nodes agree on the spelling, and here rather than in :mod:`.cuda` so that a hyperparameter parser can read it without pulling torch in. 0 or absent means the card decides. + +It is an input only. A trainer reports the size it settled on under a different key — +conventionally ``batch_size`` — because the hyperparameters are stored with the training and +handed to the next one: were the result written back here, a resumed or follow-up training would +read an earlier card's measurement as its own bound. """ MAX_BATCH_SIZE = 1024 @@ -79,7 +84,7 @@ def requested_batch_size(hyperparameters: Mapping[str, Any]) -> int: :raises ValueError: If the value is there but is not a number. """ - return int(hyperparameters.get(BATCH_SIZE, 0) or 0) + return int(hyperparameters.get(REQUESTED_BATCH_SIZE, 0) or 0) def dataset_limit(sample_count: int) -> int: diff --git a/learning_loop_node/trainer/cuda.py b/learning_loop_node/trainer/cuda.py index 9e5d6f2f..8af6fa54 100644 --- a/learning_loop_node/trainer/cuda.py +++ b/learning_loop_node/trainer/cuda.py @@ -18,8 +18,8 @@ from ..helpers.entrypoint import VRAM_LIMIT_GB_FLAG, VRAM_LIMIT_GB_HELP from .batch_size import ( - BATCH_SIZE, MAX_BATCH_SIZE, + REQUESTED_BATCH_SIZE, dataset_limit, find_batch_size, is_out_of_memory, @@ -43,11 +43,11 @@ def measure_batch_size(run_batch: Callable[[int], str | None], *, batch_size: in requested size to honour, such as a detection pass bounded only by how many images there are. :param run_batch: Runs the batch; may return a detail to append to the log line. - :param batch_size: What the training asked for, as carried in the :data:`BATCH_SIZE` - hyperparameter: the largest batch it may use, measured rather than trusted. A size that - fits is used as asked for, whether or not it is a power of two; one that does not becomes - the largest power of two below it that does, rather than a training that runs out of memory - partway through. 0 means the card decides alone. + :param batch_size: What the training asked for, as carried in the + :data:`~.batch_size.REQUESTED_BATCH_SIZE` hyperparameter: the largest batch it may use, + measured rather than trusted. A size that fits is used as asked for, whether or not it is a + power of two; one that does not becomes the largest power of two below it that does, rather + than a training that runs out of memory partway through. 0 means the card decides alone. :param sample_count: Samples in the training split, when the caller knows it. The search is then bounded so an epoch keeps enough optimizer steps to mean something, and so a loader that drops its last partial batch cannot end up with no batch at all. This bound is never @@ -59,7 +59,7 @@ def measure_batch_size(run_batch: Callable[[int], str | None], *, batch_size: in :raises ValueError: If the training asked for a negative batch size. """ if batch_size < 0: - raise ValueError(f'{BATCH_SIZE} must be >= 0, got {batch_size}') + raise ValueError(f'{REQUESTED_BATCH_SIZE} must be >= 0, got {batch_size}') limit = batch_size if sample_count is not None: From 16d25b0ca7437a962a31319c165d8a8a29f85a24 Mon Sep 17 00:00:00 2001 From: Kevin Heye Date: Thu, 24 Sep 2026 11:51:03 +0200 Subject: [PATCH 05/12] Test the VRAM limit flag on node and spawned script node_parser(vram_limit=True) and add_vram_limit_argument had no tests: default, env var, flag precedence, the legacy prefix, and the flag being absent without the opt-in. Co-Authored-By: Claude Opus 5.5 --- learning_loop_node/tests/unit/test_cuda.py | 9 +++++++ .../tests/unit/test_entrypoint.py | 27 ++++++++++++++++++- 2 files changed, 35 insertions(+), 1 deletion(-) diff --git a/learning_loop_node/tests/unit/test_cuda.py b/learning_loop_node/tests/unit/test_cuda.py index 908cdbc3..b9c76dc0 100644 --- a/learning_loop_node/tests/unit/test_cuda.py +++ b/learning_loop_node/tests/unit/test_cuda.py @@ -6,6 +6,7 @@ """ from __future__ import annotations +import argparse import importlib import logging import sys @@ -59,6 +60,14 @@ def test_nothing_is_capped_without_cuda(load): assert fake.capped == [] +def test_a_spawned_script_takes_the_same_budget_flag_as_its_node(load): + cuda, _ = load() + parser = argparse.ArgumentParser() + cuda.add_vram_limit_argument(parser) + assert parser.parse_args([]).vram_limit_gb == 0 + assert parser.parse_args(['--vram-limit-gb', '6.5']).vram_limit_gb == 6.5 + + def test_a_limit_the_card_cannot_reach_warns_instead_of_capping(load, caplog): cuda, fake = load(total_gb=8.0) with caplog.at_level(logging.WARNING): diff --git a/learning_loop_node/tests/unit/test_entrypoint.py b/learning_loop_node/tests/unit/test_entrypoint.py index 9eed399a..21d4dade 100644 --- a/learning_loop_node/tests/unit/test_entrypoint.py +++ b/learning_loop_node/tests/unit/test_entrypoint.py @@ -3,7 +3,8 @@ from ...helpers.entrypoint import node_parser MANAGED = ('WEIGHT_TYPE', 'MY_DETECTOR_WEIGHT_TYPE', 'HOST', 'NODE_HOST', 'NODE_PORT', 'PORT', - 'MY_DETECTOR_HOST', 'MY_DETECTOR_PORT', 'MY_DETECTOR_NODE_HOST') + 'MY_DETECTOR_HOST', 'MY_DETECTOR_PORT', 'MY_DETECTOR_NODE_HOST', 'VRAM_LIMIT_GB', + 'MY_DETECTOR_VRAM_LIMIT_GB') def test_every_node_gets_a_host_and_a_port(): @@ -92,6 +93,30 @@ def test_the_loop_own_host_is_not_adopted_by_a_prefixed_node(monkeypatch: pytest assert _parser(legacy_env_prefix='MY_DETECTOR_').parse_args([]).host == '0.0.0.0' +def test_a_node_that_does_not_probe_has_no_vram_limit(): + assert not hasattr(_parser().parse_args([]), 'vram_limit_gb') + + +def test_the_vram_limit_defaults_to_the_whole_card(): + assert _parser(vram_limit=True).parse_args([]).vram_limit_gb == 0 + + +def test_the_vram_limit_is_read_from_its_variable(monkeypatch: pytest.MonkeyPatch): + monkeypatch.setenv('VRAM_LIMIT_GB', '6') + assert _parser(vram_limit=True).parse_args([]).vram_limit_gb == 6.0 + + +def test_the_vram_limit_flag_beats_its_variable(monkeypatch: pytest.MonkeyPatch): + monkeypatch.setenv('VRAM_LIMIT_GB', '6') + assert _parser(vram_limit=True).parse_args(['--vram-limit-gb', '4.5']).vram_limit_gb == 4.5 + + +def test_the_vram_limit_is_still_read_under_a_legacy_prefix(monkeypatch: pytest.MonkeyPatch): + monkeypatch.setenv('MY_DETECTOR_VRAM_LIMIT_GB', '6') + args = _parser(vram_limit=True, legacy_env_prefix='MY_DETECTOR_').parse_args([]) + assert args.vram_limit_gb == 6.0 + + @pytest.fixture(autouse=True) def clean_env(monkeypatch: pytest.MonkeyPatch): """Every test starts without the variables it is about to set.""" From f27a9c61d335cfd0189fa103e2affee9598341fb Mon Sep 17 00:00:00 2001 From: Kevin Heye Date: Thu, 24 Sep 2026 11:51:03 +0200 Subject: [PATCH 06/12] Document max_batch_size and the VRAM limit flag AGENTS.md now names the input hyperparameter, why the settled size must not be written back under it, and how a trainer and its spawned script share the VRAM limit. README lists VRAM_LIMIT_GB. The paragraph no longer describes how individual node repositories compose the probe. Co-Authored-By: Claude Opus 5.5 --- AGENTS.md | 47 +++++++++++++++++++++++++++-------------------- README.md | 1 + 2 files changed, 28 insertions(+), 20 deletions(-) diff --git a/AGENTS.md b/AGENTS.md index 0923c9cc..577731fe 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -69,26 +69,33 @@ the process to it, and capping an allocator has no NVML equivalent. `measure_batch_size` is what every trainer calls, whatever shape its hyperparameters have: it takes the requested size as an `int`, so a node parsing into a dataclass enters the same door as -one keeping a dict. `batch_size` — the key named by the `BATCH_SIZE` constant, so the three nodes -agree on the spelling — is the largest batch the training may use, and it is measured rather than -trusted: a size that fits is used as named, whether or not it is a power of two, and one that does -not becomes the largest power of two below it that does. A training that starts small beats one -that runs out of memory at epoch 30. The settled size is returned and **not** stored anywhere the -next measurement would read it, which would otherwise leave inference bounded by training. Below -it, -`probe_batch_size` is the whole probe except the step itself — it resolves the limit, falls back without a card, holds the -safety margin, runs the search and releases what the trials left behind. `on_out_of_memory` is how -a node that builds a throwaway model drops an optimizer's gradients after a failed trial, and -`minimum` and `candidate` belong to `find_batch_size` itself, so a node composing by hand gets -them too: `minimum` is for a step that cannot run on a single sample at all — BatchNorm over a 1x1 -feature map, or a training whose validation halves the batch — and `candidate` is the one way the -search returns a size that is not a power of two, and only ever one that was named and then -measured. Only a node that needs the margin claimed -*before* it builds its model still composes `reserve_margin`, `measured_fits` and -`find_batch_size` itself — `dfine_node` does, so that a model too large for the budget fails while -it is being built. `measured_fits` is where an out-of-memory failure is told from a bug — both -arrive as the same exception types, and a probe that confuses them reports the smallest batch size -as the card's fault. +one keeping a dict. `max_batch_size` — the hyperparameter named by the `REQUESTED_BATCH_SIZE` +constant and read through `requested_batch_size`, so every node agrees on the spelling — is the +largest batch the training may use, and it is measured rather than trusted: a size that fits is +used as named, whether or not it is a power of two, and one that does not becomes the largest +power of two below it that does. A training that starts small beats one that runs out of memory +partway through. The settled size is returned and **not** stored anywhere the next measurement +would read it: a trainer reports it as `batch_size`, never under `max_batch_size`, because the +hyperparameters are stored with the training and handed to the next one — a resumed or follow-up +training would otherwise inherit an earlier card's measurement as its bound. + +Below it, `probe_batch_size` is the whole probe except the step itself — it resolves the limit, +falls back without a card, holds the safety margin, runs the search and releases what the trials +left behind. `on_out_of_memory` is how a node that builds a throwaway model drops an optimizer's +gradients after a failed trial, and `minimum` and `candidate` belong to `find_batch_size` itself, +so a node composing by hand gets them too: `minimum` is for a step that cannot run on a single +sample at all — BatchNorm over a 1x1 feature map, or a training whose validation halves the batch +— and `candidate` is the one way the search returns a size that is not a power of two, and only +ever one that was named and then measured. A node that needs the margin claimed *before* it builds +its model — so that a model too large for the budget fails while it is being built — can still +compose `reserve_margin`, `measured_fits` and `find_batch_size` itself. `measured_fits` is where an +out-of-memory failure is told from a bug — both arrive as the same exception types, and a probe +that confuses them reports the smallest batch size as the card's fault. + +A trainer that probes opts into the budget flag with `node_parser(vram_limit=True)`, which adds +`--vram-limit-gb` / `VRAM_LIMIT_GB`. The cap does not survive a spawn, so a script the trainer +spawns declares the same flag with `add_vram_limit_argument`, and the node passes the value on +explicitly; that script calls `limit_cuda_memory` itself. It imports torch, the package does **not** declare it, and only a trainer imports the module — so the library keeps working where nothing trains. Its unit test installs a stand-in under the name diff --git a/README.md b/README.md index 45ba8693..c695ec8c 100644 --- a/README.md +++ b/README.md @@ -29,6 +29,7 @@ You can configure connection to our Learning Loop by specifying the following en | MAX_UNCERTAIN_THRESHOLD | - | largest confidence (float) at which auto-upload will happen | Detector (opt.) | 0.6 | | EXCLUSIVE_MODEL_BUILD | - | Reject detections during update to save VRAM (set to 1) | Detector (opt.) | 0 | | INFERENCE_BATCH_SIZE | - | Batch size of trainer when calculating detections | Trainer (opt.) | 10 | +| VRAM_LIMIT_GB | - | GPU memory (GB) a training may use; the batch size is probed against it (`--vram-limit-gb`, trainers built with `node_parser(vram_limit=True)`) | Trainer (opt.) | 0 (whole card) | | RESTART_AFTER_TRAINING | - | Restart the trainer after training (set to 1) | Trainer (opt.) | 0 | | KEEP_OLD_TRAININGS | - | Do not delete old trainings (set to 1) | Trainer (opt.) | 0 | | TRAINER_IDLE_TIMEOUT_SEC | - | Automatically shutdown trainer after timeout (in seconds) | Trainer (opt.) | 0 (disabled) | From 5898fe621bbbc781e739e5bbfbacfb7a7967d6c1 Mon Sep 17 00:00:00 2001 From: Kevin Heye Date: Thu, 24 Sep 2026 12:45:10 +0200 Subject: [PATCH 07/12] Trim the new comments to what the code does not say CONTRIBUTING asks for as few and as short comments as possible, stating what is needed to understand the code rather than why it was written that way. Gone: why the help text lives in one place, why the flag is opt-in, why a caller should go through `requested_batch_size`. Kept: what an unfilled hyperparameter looks like, and that a spawned process still has to call `limit_cuda_memory` itself. Co-Authored-By: Claude Opus 5 --- learning_loop_node/helpers/entrypoint.py | 5 ++--- learning_loop_node/trainer/batch_size.py | 4 +--- learning_loop_node/trainer/cuda.py | 4 +--- 3 files changed, 4 insertions(+), 9 deletions(-) diff --git a/learning_loop_node/helpers/entrypoint.py b/learning_loop_node/helpers/entrypoint.py index cde5831e..2c80097d 100644 --- a/learning_loop_node/helpers/entrypoint.py +++ b/learning_loop_node/helpers/entrypoint.py @@ -22,7 +22,6 @@ 'out-of-memory error. Use it to share a GPU or to keep headroom against fragmentation. ' "The limit is relative to the card's total memory, not to what is currently free. " '0 (default) means no limit.') -"""Written once here because an operator reads it as the contract for what ``VRAM_LIMIT_GB`` does.""" def node_parser(*, description: str, legacy_env_prefix: str = '', @@ -33,8 +32,8 @@ def node_parser(*, description: str, legacy_env_prefix: str = '', ``'MY_DETECTOR_'``. Both spellings keep working, with a warning: the prefix on the current name (``MY_DETECTOR_NODE_HOST``) and the prefix on the flag it was originally applied to (``MY_DETECTOR_HOST``). Leave empty for a node that never used one. - :param vram_limit: Add :data:`VRAM_LIMIT_GB_FLAG`. Opt-in, because only a node that probes a - batch size has anything to do with it; on a detector node the setting means nothing. + :param vram_limit: Add :data:`VRAM_LIMIT_GB_FLAG`; only a node that probes a batch size has + anything to do with it. """ parser = _NodeArgumentParser(description=description, legacy_env_prefix=legacy_env_prefix) parser.add_argument('--host', default='0.0.0.0', env_var='NODE_HOST', diff --git a/learning_loop_node/trainer/batch_size.py b/learning_loop_node/trainer/batch_size.py index 9db8d25b..9c1d909c 100644 --- a/learning_loop_node/trainer/batch_size.py +++ b/learning_loop_node/trainer/batch_size.py @@ -78,9 +78,7 @@ def find_batch_size(fits: Callable[[int], bool], *, limit: int, minimum: int = 1 def requested_batch_size(hyperparameters: Mapping[str, Any]) -> int: """The bound a training asked for, read out of the hyperparameters the loop sent. - A field nobody filled in arrives as absent, ``None`` or ``''`` depending on where it came - from, and all three mean the same thing: no bound of its own, the card decides alone. Read it - through here rather than reaching into the dict, so every node agrees on that. + An unfilled field arrives as absent, ``None`` or ``''``, and all three mean no bound. :raises ValueError: If the value is there but is not a number. """ diff --git a/learning_loop_node/trainer/cuda.py b/learning_loop_node/trainer/cuda.py index 8af6fa54..9655bcb1 100644 --- a/learning_loop_node/trainer/cuda.py +++ b/learning_loop_node/trainer/cuda.py @@ -168,9 +168,7 @@ def usable_memory_bytes(vram_limit_gb: float) -> int: def add_vram_limit_argument(parser: ArgumentParser) -> None: """Give a spawned training script the same GPU budget flag its node has. - The cap does not survive a spawn, so a node that probes against a budget has to hand the - number to whatever it spawns, and that process has to call :func:`limit_cuda_memory` itself. - This is the parsing half of that, spelled and documented exactly as on the node. + The spawned process still has to call :func:`limit_cuda_memory` with it. """ parser.add_argument(VRAM_LIMIT_GB_FLAG, type=float, default=0, help=VRAM_LIMIT_GB_HELP) From ef150ebb08d0570710a048f0558088f2431bf8c6 Mon Sep 17 00:00:00 2001 From: Kevin Heye Date: Fri, 25 Sep 2026 09:53:04 +0200 Subject: [PATCH 08/12] Name the resumed training as the reason to keep batch_size apart The docs said the loop hands a training's hyperparameters to the next one. It does not: it builds each training from the project configuration and the job's override, taking only the resolution from a base training. What does read them back is the node itself, which saves the training to last_training__.json and restores it when it resumes after a restart. Co-Authored-By: Claude Opus 5.5 --- AGENTS.md | 7 +++++-- learning_loop_node/tests/unit/test_batch_size.py | 2 +- learning_loop_node/trainer/batch_size.py | 6 +++--- 3 files changed, 9 insertions(+), 6 deletions(-) diff --git a/AGENTS.md b/AGENTS.md index 577731fe..e8beaa34 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -76,8 +76,11 @@ used as named, whether or not it is a power of two, and one that does not become power of two below it that does. A training that starts small beats one that runs out of memory partway through. The settled size is returned and **not** stored anywhere the next measurement would read it: a trainer reports it as `batch_size`, never under `max_batch_size`, because the -hyperparameters are stored with the training and handed to the next one — a resumed or follow-up -training would otherwise inherit an earlier card's measurement as its bound. +node saves the hyperparameters with the training (`LastTrainingIO`) and a training resumed after a +restart reads them back — it would otherwise take its first run's measurement as its bound instead +of measuring again. The loop does not hand reported values to a later training: it builds each one +from the project configuration and the job's override, taking only `resolution` from a base +training. Below it, `probe_batch_size` is the whole probe except the step itself — it resolves the limit, falls back without a card, holds the safety margin, runs the search and releases what the trials diff --git a/learning_loop_node/tests/unit/test_batch_size.py b/learning_loop_node/tests/unit/test_batch_size.py index e8660ede..eb503c69 100644 --- a/learning_loop_node/tests/unit/test_batch_size.py +++ b/learning_loop_node/tests/unit/test_batch_size.py @@ -121,7 +121,7 @@ def test_an_unfilled_batch_size_means_no_bound(empty: dict): def test_the_reported_batch_size_is_never_read_as_a_bound(): - """A trainer reports its settled size as `batch_size`; the next training must not inherit it.""" + """A trainer reports its settled size as `batch_size`; a resumed training must not read it as its bound.""" assert REQUESTED_BATCH_SIZE != 'batch_size' assert requested_batch_size({'batch_size': 16}) == 0 assert requested_batch_size({'batch_size': 16, REQUESTED_BATCH_SIZE: 64}) == 64 diff --git a/learning_loop_node/trainer/batch_size.py b/learning_loop_node/trainer/batch_size.py index 9c1d909c..66ed335c 100644 --- a/learning_loop_node/trainer/batch_size.py +++ b/learning_loop_node/trainer/batch_size.py @@ -24,9 +24,9 @@ hyperparameter parser can read it without pulling torch in. 0 or absent means the card decides. It is an input only. A trainer reports the size it settled on under a different key — -conventionally ``batch_size`` — because the hyperparameters are stored with the training and -handed to the next one: were the result written back here, a resumed or follow-up training would -read an earlier card's measurement as its own bound. +conventionally ``batch_size`` — because the node saves the hyperparameters with the training and +a training resumed after a restart reads them back: were the result written back here, the resumed +run would take its first run's measurement as its bound instead of measuring again. """ MAX_BATCH_SIZE = 1024 From f397056e2c07ee79f756971d18197cb796e43c96 Mon Sep 17 00:00:00 2001 From: Kevin Heye Date: Fri, 25 Sep 2026 10:19:11 +0200 Subject: [PATCH 09/12] Drop the hand-composed probe route from the docs No trainer composes reserve_margin, measured_fits and find_batch_size itself any more: dfine_node now enters through measure_batch_size like the others. Also restore the missing blank line before measure_batch_size (E302). Co-Authored-By: Claude Opus 5.5 --- AGENTS.md | 16 +++++++--------- learning_loop_node/trainer/cuda.py | 4 ++-- 2 files changed, 9 insertions(+), 11 deletions(-) diff --git a/AGENTS.md b/AGENTS.md index e8beaa34..ea32117c 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -85,15 +85,13 @@ training. Below it, `probe_batch_size` is the whole probe except the step itself — it resolves the limit, falls back without a card, holds the safety margin, runs the search and releases what the trials left behind. `on_out_of_memory` is how a node that builds a throwaway model drops an optimizer's -gradients after a failed trial, and `minimum` and `candidate` belong to `find_batch_size` itself, -so a node composing by hand gets them too: `minimum` is for a step that cannot run on a single -sample at all — BatchNorm over a 1x1 feature map, or a training whose validation halves the batch -— and `candidate` is the one way the search returns a size that is not a power of two, and only -ever one that was named and then measured. A node that needs the margin claimed *before* it builds -its model — so that a model too large for the budget fails while it is being built — can still -compose `reserve_margin`, `measured_fits` and `find_batch_size` itself. `measured_fits` is where an -out-of-memory failure is told from a bug — both arrive as the same exception types, and a probe -that confuses them reports the smallest batch size as the card's fault. +gradients after a failed trial. `minimum` is for a step that cannot run on a single sample at all +— BatchNorm over a 1x1 feature map, or a training whose validation halves the batch — and +`candidate` is the one way the search returns a size that is not a power of two, and only ever +one that was named and then measured. A node does not compose `reserve_margin`, `measured_fits` +and `find_batch_size` itself; they are the pieces `probe_batch_size` is built from. `measured_fits` +is where an out-of-memory failure is told from a bug — both arrive as the same exception types, +and a probe that confuses them reports the smallest batch size as the card's fault. A trainer that probes opts into the budget flag with `node_parser(vram_limit=True)`, which adds `--vram-limit-gb` / `VRAM_LIMIT_GB`. The cap does not survive a spawn, so a script the trainer diff --git a/learning_loop_node/trainer/cuda.py b/learning_loop_node/trainer/cuda.py index 9655bcb1..d5f86bdb 100644 --- a/learning_loop_node/trainer/cuda.py +++ b/learning_loop_node/trainer/cuda.py @@ -32,6 +32,7 @@ SAFETY_MARGIN = 0.05 """Share of the budget held back while probing, against allocator fragmentation later on.""" + def measure_batch_size(run_batch: Callable[[int], str | None], *, batch_size: int = 0, sample_count: int | None = None, probe: str = 'batch-size probe', minimum: int = 1, vram_limit_gb: float = 0, @@ -78,8 +79,7 @@ def probe_batch_size(run_batch: Callable[[int], str | None], *, probe: str = 'ba This is the whole of a probe except the step itself: the margin, the search, telling an out-of-memory failure from a bug, and releasing what the trials left behind. A caller that builds a throwaway model supplies ``on_out_of_memory`` to drop what a failed trial left on the - card; only a caller that needs the margin claimed *before* it builds that model has to compose - :func:`reserve_margin`, :func:`measured_fits` and ``find_batch_size`` itself. + card. :param run_batch: Runs the batch; may return a detail to append to the log line. :param probe: Names this probe in the log, so a node running several stays readable. From 3a5577730e8dca1f6e10bcff88fa72c4030112bf Mon Sep 17 00:00:00 2001 From: Kevin Heye Date: Fri, 25 Sep 2026 10:31:23 +0200 Subject: [PATCH 10/12] Keep the VRAM flag beside the probe, not in the entrypoint trainer/cuda.py imported helpers/entrypoint.py only for the flag name and its help text, so the torch side of the probe depended on the node's server boilerplate. The two constants now live in the torch-free batch_size.py, which both node_parser and add_vram_limit_argument read from. Co-Authored-By: Claude Opus 5.5 --- learning_loop_node/helpers/entrypoint.py | 15 ++++----------- learning_loop_node/trainer/batch_size.py | 13 +++++++++++++ learning_loop_node/trainer/cuda.py | 3 ++- 3 files changed, 19 insertions(+), 12 deletions(-) diff --git a/learning_loop_node/helpers/entrypoint.py b/learning_loop_node/helpers/entrypoint.py index 2c80097d..08f1d254 100644 --- a/learning_loop_node/helpers/entrypoint.py +++ b/learning_loop_node/helpers/entrypoint.py @@ -12,16 +12,9 @@ import configargparse import uvicorn -logger = logging.getLogger(__name__) - -VRAM_LIMIT_GB_FLAG = '--vram-limit-gb' -"""The flag a trainer takes its GPU budget from; ``VRAM_LIMIT_GB`` follows from the name.""" +from ..trainer.batch_size import VRAM_LIMIT_GB_FLAG, VRAM_LIMIT_GB_HELP -VRAM_LIMIT_GB_HELP = ('Gigabytes of GPU memory a training may use. The batch size is probed against this limit ' - 'instead of the whole card, so a lower limit yields a smaller batch size rather than an ' - 'out-of-memory error. Use it to share a GPU or to keep headroom against fragmentation. ' - "The limit is relative to the card's total memory, not to what is currently free. " - '0 (default) means no limit.') +logger = logging.getLogger(__name__) def node_parser(*, description: str, legacy_env_prefix: str = '', @@ -32,8 +25,8 @@ def node_parser(*, description: str, legacy_env_prefix: str = '', ``'MY_DETECTOR_'``. Both spellings keep working, with a warning: the prefix on the current name (``MY_DETECTOR_NODE_HOST``) and the prefix on the flag it was originally applied to (``MY_DETECTOR_HOST``). Leave empty for a node that never used one. - :param vram_limit: Add :data:`VRAM_LIMIT_GB_FLAG`; only a node that probes a batch size has - anything to do with it. + :param vram_limit: Add :data:`~learning_loop_node.trainer.batch_size.VRAM_LIMIT_GB_FLAG`; + only a node that probes a batch size has anything to do with it. """ parser = _NodeArgumentParser(description=description, legacy_env_prefix=legacy_env_prefix) parser.add_argument('--host', default='0.0.0.0', env_var='NODE_HOST', diff --git a/learning_loop_node/trainer/batch_size.py b/learning_loop_node/trainer/batch_size.py index 66ed335c..71a65105 100644 --- a/learning_loop_node/trainer/batch_size.py +++ b/learning_loop_node/trainer/batch_size.py @@ -29,6 +29,19 @@ run would take its first run's measurement as its bound instead of measuring again. """ +VRAM_LIMIT_GB_FLAG = '--vram-limit-gb' +"""The flag a trainer takes its GPU budget from; ``VRAM_LIMIT_GB`` follows from the name. + +Here rather than beside ``node_parser``, so that :mod:`.cuda` declares the same flag for a spawned +script without depending on the node's entrypoint. +""" + +VRAM_LIMIT_GB_HELP = ('Gigabytes of GPU memory a training may use. The batch size is probed against this limit ' + 'instead of the whole card, so a lower limit yields a smaller batch size rather than an ' + 'out-of-memory error. Use it to share a GPU or to keep headroom against fragmentation. ' + "The limit is relative to the card's total memory, not to what is currently free. " + '0 (default) means no limit.') + MAX_BATCH_SIZE = 1024 """Where a search stops when its caller sets no bound of its own.""" diff --git a/learning_loop_node/trainer/cuda.py b/learning_loop_node/trainer/cuda.py index d5f86bdb..26cb68ea 100644 --- a/learning_loop_node/trainer/cuda.py +++ b/learning_loop_node/trainer/cuda.py @@ -16,10 +16,11 @@ import torch -from ..helpers.entrypoint import VRAM_LIMIT_GB_FLAG, VRAM_LIMIT_GB_HELP from .batch_size import ( MAX_BATCH_SIZE, REQUESTED_BATCH_SIZE, + VRAM_LIMIT_GB_FLAG, + VRAM_LIMIT_GB_HELP, dataset_limit, find_batch_size, is_out_of_memory, From 3d16e57a99115ac18a62b143b6049408c5f1fd95 Mon Sep 17 00:00:00 2001 From: Kevin Heye Date: Fri, 25 Sep 2026 10:31:50 +0200 Subject: [PATCH 11/12] Log the bound the training set puts on the batch size D-FINE's own probe logged the training sample count beside the chosen size; since it enters through measure_batch_size that line was gone, and a size capped by a small dataset looked the same in the log as one capped by memory. measure_batch_size now logs it for every trainer that passes sample_count. Co-Authored-By: Claude Opus 5.5 --- learning_loop_node/tests/unit/test_cuda.py | 7 +++++++ learning_loop_node/trainer/cuda.py | 2 ++ 2 files changed, 9 insertions(+) diff --git a/learning_loop_node/tests/unit/test_cuda.py b/learning_loop_node/tests/unit/test_cuda.py index b9c76dc0..172ac30e 100644 --- a/learning_loop_node/tests/unit/test_cuda.py +++ b/learning_loop_node/tests/unit/test_cuda.py @@ -208,6 +208,13 @@ def test_a_dataset_bound_is_never_used_as_a_candidate(load): assert 10 not in ran, 'samples // 8 is a heuristic, not a size anyone asked for' +def test_the_log_says_when_the_dataset_is_what_bounds_the_search(load, caplog): + cuda, fake = load() + with caplog.at_level(logging.INFO): + cuda.measure_batch_size(_fits_up_to(1024, fake, []), sample_count=80) + assert '80 training samples allow at most 10 per batch' in caplog.text + + def test_the_tighter_of_the_request_and_the_dataset_wins(load): cuda, fake = load() assert cuda.measure_batch_size(_fits_up_to(1024, fake, []), batch_size=4, diff --git a/learning_loop_node/trainer/cuda.py b/learning_loop_node/trainer/cuda.py index 26cb68ea..c3e6ec71 100644 --- a/learning_loop_node/trainer/cuda.py +++ b/learning_loop_node/trainer/cuda.py @@ -66,6 +66,8 @@ def measure_batch_size(run_batch: Callable[[int], str | None], *, batch_size: in limit = batch_size if sample_count is not None: limit = min(limit or MAX_BATCH_SIZE, dataset_limit(sample_count)) + logger.info('%s: %d training samples allow at most %d per batch', probe, sample_count, + dataset_limit(sample_count)) return probe_batch_size(run_batch, probe=probe, limit=limit, candidate=batch_size, minimum=minimum, vram_limit_gb=vram_limit_gb, From 230568553bd89b1fe87f169a327af95200a5dffe Mon Sep 17 00:00:00 2001 From: jan Date: Mon, 28 Sep 2026 15:18:05 +0200 Subject: [PATCH 12/12] Drop the motivation from the new docstrings MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit CONTRIBUTING §4 keeps a comment to what is needed to understand the code, not the motivation behind it. Gone: where REQUESTED_BATCH_SIZE and VRAM_LIMIT_GB_FLAG live and why, why the search uses powers of two, why a minimum above one exists, why the dataset bound is not a candidate, why the vram_limit flag is opt-in, and the comment on why the candidate is reachable. The reason to keep batch_size apart from max_batch_size stays in AGENTS.md and the PR description. --- learning_loop_node/helpers/entrypoint.py | 3 +-- learning_loop_node/trainer/batch_size.py | 26 ++++++++---------------- learning_loop_node/trainer/cuda.py | 15 +++++++------- 3 files changed, 16 insertions(+), 28 deletions(-) diff --git a/learning_loop_node/helpers/entrypoint.py b/learning_loop_node/helpers/entrypoint.py index 08f1d254..0d56d45d 100644 --- a/learning_loop_node/helpers/entrypoint.py +++ b/learning_loop_node/helpers/entrypoint.py @@ -25,8 +25,7 @@ def node_parser(*, description: str, legacy_env_prefix: str = '', ``'MY_DETECTOR_'``. Both spellings keep working, with a warning: the prefix on the current name (``MY_DETECTOR_NODE_HOST``) and the prefix on the flag it was originally applied to (``MY_DETECTOR_HOST``). Leave empty for a node that never used one. - :param vram_limit: Add :data:`~learning_loop_node.trainer.batch_size.VRAM_LIMIT_GB_FLAG`; - only a node that probes a batch size has anything to do with it. + :param vram_limit: Add :data:`~learning_loop_node.trainer.batch_size.VRAM_LIMIT_GB_FLAG`. """ parser = _NodeArgumentParser(description=description, legacy_env_prefix=legacy_env_prefix) parser.add_argument('--host', default='0.0.0.0', env_var='NODE_HOST', diff --git a/learning_loop_node/trainer/batch_size.py b/learning_loop_node/trainer/batch_size.py index 71a65105..c921f27c 100644 --- a/learning_loop_node/trainer/batch_size.py +++ b/learning_loop_node/trainer/batch_size.py @@ -20,21 +20,12 @@ REQUESTED_BATCH_SIZE = 'max_batch_size' """The hyperparameter every node reads its `measure_batch_size` argument out of. -Named here so the nodes agree on the spelling, and here rather than in :mod:`.cuda` so that a -hyperparameter parser can read it without pulling torch in. 0 or absent means the card decides. - -It is an input only. A trainer reports the size it settled on under a different key — -conventionally ``batch_size`` — because the node saves the hyperparameters with the training and -a training resumed after a restart reads them back: were the result written back here, the resumed -run would take its first run's measurement as its bound instead of measuring again. +0 or absent means the card decides. It is an input only: a trainer reports the size it settled on +under a different key, conventionally ``batch_size``, and never writes it back here. """ VRAM_LIMIT_GB_FLAG = '--vram-limit-gb' -"""The flag a trainer takes its GPU budget from; ``VRAM_LIMIT_GB`` follows from the name. - -Here rather than beside ``node_parser``, so that :mod:`.cuda` declares the same flag for a spawned -script without depending on the node's entrypoint. -""" +"""The flag a trainer takes its GPU budget from; ``VRAM_LIMIT_GB`` follows from the name.""" VRAM_LIMIT_GB_HELP = ('Gigabytes of GPU memory a training may use. The batch size is probed against this limit ' 'instead of the whole card, so a lower limit yields a smaller batch size rather than an ' @@ -56,15 +47,14 @@ def find_batch_size(fits: Callable[[int], bool], *, limit: int, minimum: int = 1 candidate: int = 0) -> int: """Return the largest batch size that fits, never exceeding ``limit``. - Powers of two, so equal hardware and equal hyperparameters yield an equal recipe — plus - ``candidate``, which is the one size outside that set this will return, and only when somebody - named it and it then measured. + Powers of two, plus ``candidate``, which is the one size outside that set this will return, and + only when somebody named it and it then measured. :param fits: Runs a representative probe; ``False`` on out-of-memory. :param limit: Upper bound; the doubling stops at the largest power of two within it. :param minimum: Smallest size to try, rounded down to a power of two. Raise it above one for a - step that cannot run on a single sample at all — BatchNorm over a 1x1 feature map, a - validation pass that halves the batch — where a failure at one says nothing about memory. + step that cannot run on a single sample at all: BatchNorm over a 1x1 feature map, a + validation pass that halves the batch. :param candidate: An exact size, tried once the doubling has reached its ceiling, so a size that was asked for is used as asked for rather than rounded down. Ignored unless it lies between that ceiling and ``limit``; a bound nobody named — one derived from the dataset, @@ -83,7 +73,7 @@ def find_batch_size(fits: Callable[[int], bool], *, limit: int, minimum: int = 1 size *= 2 if size == ceiling and ceiling < candidate <= bound and fits(candidate): - size = candidate # the doubling was not what stopped it, so the named size is reachable + size = candidate return size diff --git a/learning_loop_node/trainer/cuda.py b/learning_loop_node/trainer/cuda.py index c3e6ec71..f217bcaa 100644 --- a/learning_loop_node/trainer/cuda.py +++ b/learning_loop_node/trainer/cuda.py @@ -40,20 +40,19 @@ def measure_batch_size(run_batch: Callable[[int], str | None], *, batch_size: in on_out_of_memory: Callable[[], None] | None = None) -> int: """Settle a training's batch size against what it asked for and what the card allows. - The whole of the decision, so that a trainer is left with only the step. Every training enters - here, whatever shape its hyperparameters have; :func:`probe_batch_size` is for a probe with no - requested size to honour, such as a detection pass bounded only by how many images there are. + Every training enters here, whatever shape its hyperparameters have; :func:`probe_batch_size` is + for a probe with no requested size to honour, such as a detection pass bounded only by how many + images there are. :param run_batch: Runs the batch; may return a detail to append to the log line. :param batch_size: What the training asked for, as carried in the :data:`~.batch_size.REQUESTED_BATCH_SIZE` hyperparameter: the largest batch it may use, measured rather than trusted. A size that fits is used as asked for, whether or not it is a - power of two; one that does not becomes the largest power of two below it that does, rather - than a training that runs out of memory partway through. 0 means the card decides alone. + power of two; one that does not becomes the largest power of two below it that does. 0 means + the card decides alone. :param sample_count: Samples in the training split, when the caller knows it. The search is - then bounded so an epoch keeps enough optimizer steps to mean something, and so a loader - that drops its last partial batch cannot end up with no batch at all. This bound is never - used as the exact candidate -- it is a heuristic, not a size anyone named. + then bounded by :func:`~.batch_size.dataset_limit`; that bound is never used as the exact + candidate. :param minimum: Smallest size to try; see :func:`probe_batch_size`. :param vram_limit_gb: The budget the safety margin is a share of; 0 means the whole card. :param on_out_of_memory: Runs after a trial ran out of memory, to drop what it left behind.