From 069cf9c1cd09bb7af8c995b45de95f77fefcc3b6 Mon Sep 17 00:00:00 2001 From: Akshey D <131929364+aksheyd@users.noreply.github.com> Date: Mon, 28 Sep 2026 04:28:23 +0000 Subject: [PATCH 1/2] docs: make the README's adaptive and np.savez recipes work on every layer The adaptive line stopped at the first layer with outliers, where a block needs more than 8 bits and ToleranceTooTightError is raised, and its np.std(weights) failed on PyTorch tensors. It now uses weights.std(), which works on arrays and tensors, and a short retry with the error's smallest_tolerance follows the list, saying it loosens every block. The np.savez recipe saved bits=None for an adaptive tensor, which np.load refuses, so it now leaves out whichever part is None, and the test runs that exact recipe for every kind and scale type. --- python/README.md | 18 ++++++++++++++---- python/tests/test_quantize.py | 9 ++++----- 2 files changed, 18 insertions(+), 9 deletions(-) diff --git a/python/README.md b/python/README.md index d4f1326..eee3e69 100644 --- a/python/README.md +++ b/python/README.md @@ -25,18 +25,28 @@ the scales count toward the size: 4-bit codes with one f16 scale per 32 values c the other schemes return the same `Quantized` type: - `asymmetric.quantize(weights, bits=8, block=32)` adds a zero-point per block, for values that aren't centered on zero -- `adaptive.quantize(weights, tolerance=0.1 * np.std(weights))` gives each block the fewest bits, from 2 to 8, that round every weight within `tolerance`, in the weights' own units. a tenth of their standard deviation gives about 5 bits a block +- `adaptive.quantize(weights, tolerance=0.1 * weights.std())` gives each block the fewest bits, from 2 to 8, that round every weight within `tolerance`, in the weights' own units. a tenth of their standard deviation gives about 5 bits a block - `learned.refine(q, weights)` refits each block's scale, and its zero-point if it has one, to lower the error. it changes `q` in place, so call `q.copy()` first to keep the original - `learned.alternate(q, weights)` refits too, then rounds each value to the nearest code on its block's new line, and repeats until no code moves. it also changes `q` in place - `Scheme.Q4_32.quantize(weights)` picks a scheme at run time +a block with outliers can need more than 8 bits, which raises `ToleranceTooTightError`. retrying with its `smallest_tolerance` works, but loosens every block, not just that one: + +```python +try: + q = adaptive.quantize(weights, tolerance=0.1 * weights.std()) +except ToleranceTooTightError as error: + q = adaptive.quantize(weights, tolerance=error.smallest_tolerance) +``` + quantized values can be pickled, and compared with `==`. `q.to_bytes()` saves one as bytes, in the same format as the rust crate, and `Quantized.from_bytes(data)` loads it back. to keep it in an `np.savez` or safetensors file, store `np.frombuffer(q.to_bytes(), np.uint8)`. -to save its parts as plain arrays instead, like with `np.savez`, pass them back by name to `Quantized.from_parts`. an adaptive tensor keeps `block_bits` instead of `bits`: +to save its parts as plain arrays instead, like with `np.savez`, pass them back by name to `Quantized.from_parts`. leave out the one that's `None`: `bits` for an adaptive tensor, or `block_bits` for the others: ```python -np.savez("layer.npz", kind=q.kind, shape=q.shape, block=q.block, bits=q.bits, - codes=q.codes, scales=q.scales, zero_points=q.zero_points, scale=q.scale.name) +parts = dict(kind=q.kind, shape=q.shape, block=q.block, bits=q.bits, block_bits=q.block_bits, + codes=q.codes, scales=q.scales, zero_points=q.zero_points, scale=q.scale.name) +np.savez("layer.npz", **{name: part for name, part in parts.items() if part is not None}) q = Quantized.from_parts(**np.load("layer.npz")) ``` diff --git a/python/tests/test_quantize.py b/python/tests/test_quantize.py index 95c1acc..f1ff14f 100644 --- a/python/tests/test_quantize.py +++ b/python/tests/test_quantize.py @@ -528,16 +528,15 @@ def test_from_parts_rebuilds_parts_saved_with_numpy_as_the_readme_says(): "kind": quantized.kind, "shape": quantized.shape, "block": quantized.block, + "bits": quantized.bits, + "block_bits": quantized.block_bits, "codes": quantized.codes, "scales": quantized.scales, "zero_points": quantized.zero_points, "scale": quantized.scale.name, } - if quantized.kind == "adaptive": - parts["block_bits"] = quantized.block_bits - else: - parts["bits"] = quantized.bits - rebuilt = Quantized.from_parts(**saved_and_loaded_with_numpy(parts)) + saved = {name: part for name, part in parts.items() if part is not None} + rebuilt = Quantized.from_parts(**saved_and_loaded_with_numpy(saved)) assert rebuilt == quantized From eb42e08b1807f97e897f07ffc540b1c7f059b04a Mon Sep 17 00:00:00 2001 From: Akshey D <131929364+aksheyd@users.noreply.github.com> Date: Mon, 28 Sep 2026 04:47:14 +0000 Subject: [PATCH 2/2] docs: say a list needs np.std for the adaptive tolerance --- python/README.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/python/README.md b/python/README.md index 0877c96..3e5ecc5 100644 --- a/python/README.md +++ b/python/README.md @@ -25,7 +25,7 @@ the scales count toward the size: 4-bit codes with one f16 scale per 32 values c the other schemes return the same `Quantized` type: - `asymmetric.quantize(weights, bits=8, block=32)` adds a zero-point per block, for values that aren't centered on zero -- `adaptive.quantize(weights, tolerance=0.1 * weights.std())` gives each block the fewest bits, from 2 to 8, that round every weight within `tolerance`, in the weights' own units. a tenth of their standard deviation gives about 5 bits a block +- `adaptive.quantize(weights, tolerance=0.1 * weights.std())` gives each block the fewest bits, from 2 to 8, that round every weight within `tolerance`, in the weights' own units. a tenth of their standard deviation gives about 5 bits a block. for a list, use `np.std(weights)` - `learned.refine(q, weights)` refits each block's scale, and its zero-point if it has one, to lower the mean squared error. it changes `q` in place, so call `q.copy()` first to keep the original - `learned.alternate(q, weights)` refits too, then rounds each value to the nearest code on its block's new line, and repeats until no code moves. it also changes `q` in place. both can raise the worst error past an adaptive tensor's tolerance - `Scheme.Q4_32.quantize(weights)` picks a scheme at run time