diff --git a/auto_round/auto_scheme/delta_loss.py b/auto_round/auto_scheme/delta_loss.py index 3f405fb3e7..f2eb503015 100644 --- a/auto_round/auto_scheme/delta_loss.py +++ b/auto_round/auto_scheme/delta_loss.py @@ -895,6 +895,9 @@ def get_score_for_scheme( """Wrap every quantizable layer in ``quant_layer_names`` with a scoring wrapper, run forward(+backward, unless RTN-only) calibration over ``nsamples`` examples from ``dataset``/``dataloader``, then unwrap and return each layer's ``[bits, loss]``. + + The caller has already applied one scheme to every layer, so each layer reports the + loss of that single scheme. Returns ``{name: [bits, loss]}``. """ scores_dict = {} # Key=name,Val=[quant_total_bits, loss] # Include the visual block(s) when scoring VLMs with ``--quant_nontext_module`` @@ -1251,6 +1254,7 @@ def _run_forward_loop(loader): ) layer_bits, _ = compute_layer_bits(m.orig_layer, ignore_scale_zp_bits=ignore_scale_zp_bits) scores_dict[n] = [layer_bits, m.mix_score] + _fill_inactive_expert_scores(scores_dict, block_names) _log_score_summary_by_block_and_nonblock( scores_dict, @@ -2876,9 +2880,31 @@ def _select_embedding_scheme_index(): ) dp_started = time.perf_counter() - best_loss, best_path = choose_bits_per_layer_with_path(total_scores, target_params_cnt) + + # ------------------------------------------------------------------ # + # Bit allocation. Both solvers optimise the same objective over the same score + # table; they differ only in how the knapsack is solved. The DP is the historical + # default; the Lagrangian dual reaches the same allocation far faster because it + # needs no discretised state space. + # ------------------------------------------------------------------ # + if auto_scheme.solver == "lagrangian": + from auto_round.auto_scheme.solver import solve_lagrangian + + assign = solve_lagrangian(total_scores, target_params_cnt) + if assign is None: + # Infeasible under the dual -- fall back to the DP so a solver problem can + # never make AutoScheme worse than before. + logger.warning("AutoScheme: Lagrangian solver found no feasible allocation; falling back to DP.") + best_loss, best_path = choose_bits_per_layer_with_path(total_scores, target_params_cnt) + else: + best_path = [(opt[3], opt[0]) for opt in assign.values()] + best_loss = sum(opt[2] for opt in assign.values()) + else: + best_loss, best_path = choose_bits_per_layer_with_path(total_scores, target_params_cnt) + logger.info( - "AutoScheme post-scoring: DP selection took %.2fs (layers=%d)", + "AutoScheme post-scoring: %s selection took %.2fs (layers=%d)", + auto_scheme.solver, time.perf_counter() - dp_started, len(total_scores), ) @@ -2953,7 +2979,7 @@ def gen_layer_config( """Public AutoScheme entry. This wrapper performs model loading/dispatch and environment preparation, - then delegates to `_gen_layer_config` for staged scoring + DP selection. + then delegates to `_gen_layer_config` for scoring + solver-based selection. """ model_name = None is_vlm = False diff --git a/auto_round/auto_scheme/gen_auto_scheme.py b/auto_round/auto_scheme/gen_auto_scheme.py index e86e1e52bc..5ce81b3507 100644 --- a/auto_round/auto_scheme/gen_auto_scheme.py +++ b/auto_round/auto_scheme/gen_auto_scheme.py @@ -41,11 +41,26 @@ class AutoScheme: low_gpu_mem_usage: bool = True low_cpu_mem_usage: bool = True + # ------------------------------------------------------------------ # + # Bit-allocation solver. + # + # The default reproduces the historical behaviour exactly: a knapsack DP + # over scores measured on uniform-scheme models. + # ------------------------------------------------------------------ # + solver: str = "dp" + """Allocation solver: ``"dp"`` (knapsack DP) or ``"lagrangian"`` (shadow-price + bisection). Both optimise the same objective and in practice produce the same + allocation, but the Lagrangian solver hits a *fractional* avg_bits target exactly + and needs no discretised state space. See docs/auto_scheme_solver.md for measured + per-task accuracy.""" + def __post_init__(self): if isinstance(self.options, str): options = self.options.upper().replace(" ", "") self.options = options.split(",") self.options = self._deduplicate_options(self.options) + if self.solver not in ("dp", "lagrangian"): + raise ValueError(f"AutoScheme.solver must be 'dp' or 'lagrangian', got {self.solver!r}") @staticmethod def _deduplicate_options( diff --git a/auto_round/auto_scheme/solver.py b/auto_round/auto_scheme/solver.py new file mode 100644 index 0000000000..a465b0e800 --- /dev/null +++ b/auto_round/auto_scheme/solver.py @@ -0,0 +1,235 @@ +# Copyright (c) 2025 Intel Corporation +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Bit-allocation solvers for AutoScheme. + +Two solvers are available, selected through ``AutoScheme.solver``: + +``"dp"`` + The historical knapsack dynamic program (:func:`choose_bits_per_layer_with_path`). + Exact on a discretised bit grid, but the state space grows with the budget, which + dominates the runtime on large models. + +``"lagrangian"`` + Solves the same knapsack through its Lagrangian dual. For a price ``lam`` (units: + loss per bit) every layer independently picks ``argmin_s loss_s + lam * bits_s``. + Total bits decrease monotonically in ``lam``, so a bisection on ``lam`` drives the + solution onto the budget. This hits a *fractional* avg_bits target exactly and needs + no discretised state space, so it is typically an order of magnitude faster than the + DP while producing the same allocation. + +The dual only ever lands on the *convex hull* of each layer's (bits, loss) curve, so two +primal repair passes close the integrality gap: :func:`_greedy_repair` spends leftover +budget, and :func:`_swap_local_search` fixes cases where one downgrade funds one upgrade. + +This module is deliberately free of model/torch dependencies so it can be unit-tested in +isolation. +""" + +from __future__ import annotations + +from typing import Optional + +from auto_round.logger import logger + +__all__ = [ + "solve_lagrangian", + "solve_allocation", +] + +_EPS = 1e-12 + + +# ----------------------------------------------------------------------------------- # +# Score-table helpers +# +# ``total_scores`` maps a DP key (the first layer name of a shared-layer group) to a list +# of candidate options, each option being ``[scheme_index, bits_cost, loss_cost, names]``. +# ----------------------------------------------------------------------------------- # +def _option_cost(opt, lam: float) -> float: + """Lagrangian cost ``loss + lam * bits`` of a single option.""" + return opt[2] + lam * opt[1] + + +def _pick_option(opts, lam: float): + """Pick the option minimising the Lagrangian cost; ties broken toward fewer bits.""" + return min(opts, key=lambda o: (_option_cost(o, lam), o[1])) + + +def _total_bits(assign: dict) -> int: + """Total bit cost of an assignment (``key -> option``).""" + return sum(opt[1] for opt in assign.values()) + + +def _total_loss(assign: dict) -> float: + """Total predicted loss of an assignment (``key -> option``).""" + return sum(opt[2] for opt in assign.values()) + + +# ----------------------------------------------------------------------------------- # +# Lagrangian (shadow-price) solver +# ----------------------------------------------------------------------------------- # +def solve_lagrangian(total_scores: dict, budget: int, max_iter: int = 80) -> Optional[dict]: + """Solve the bit-allocation knapsack through its Lagrangian dual. + + Args: + total_scores: Score table. + budget: Upper bound on total bits. + max_iter: Bisection iterations. + + Returns: + The assignment mapping key -> option, or ``None`` when even the cheapest + configuration exceeds ``budget`` (i.e. the target is infeasible). + """ + if not total_scores: + return {} + + cheapest = {key: min(opts, key=lambda o: o[1]) for key, opts in total_scores.items()} + if _total_bits(cheapest) > budget: + return None + + # lam = 0 -> pure loss minimisation. If it already fits, nothing to trade off. + assign = {key: _pick_option(opts, 0.0) for key, opts in total_scores.items()} + if _total_bits(assign) <= budget: + return assign + + # Bracket the price: grow the upper bound until the budget is satisfied. + lo, hi = 0.0, 1e-9 + for _ in range(200): + probe = {key: _pick_option(opts, hi) for key, opts in total_scores.items()} + if _total_bits(probe) <= budget: + break + lo, hi = hi, hi * 4.0 + else: # pragma: no cover - defensive + logger.warning("AutoScheme: Lagrangian upper price search did not converge.") + + best = None + for _ in range(max_iter): + mid = 0.5 * (lo + hi) + probe = {key: _pick_option(opts, mid) for key, opts in total_scores.items()} + if _total_bits(probe) <= budget: + best, hi = probe, mid + else: + lo = mid + + if best is None: # pragma: no cover - defensive + best = {key: _pick_option(opts, hi) for key, opts in total_scores.items()} + best = _greedy_repair(total_scores, best, budget) + best = _swap_local_search(total_scores, best, budget) + return best + + +def _greedy_repair(total_scores: dict, assign: dict, budget: int) -> dict: + """Spend the budget slack left by the dual's integrality gap. + + The dual solution is generally not budget-tight. Repeatedly apply the upgrade with the + best loss-reduction-per-extra-bit that still fits, which is the standard primal repair + for a Lagrangian-relaxed knapsack. + """ + used = _total_bits(assign) + slack = budget - used + if slack <= 0: + return assign + + while True: + best_key, best_opt, best_gain = None, None, 0.0 + for key, opts in total_scores.items(): + cur = assign[key] + for opt in opts: + extra_bits = opt[1] - cur[1] + if extra_bits <= 0 or extra_bits > slack: + continue + gain = (cur[2] - opt[2]) / extra_bits + if gain > best_gain: + best_key, best_opt, best_gain = key, opt, gain + if best_key is None: + break + slack -= best_opt[1] - assign[best_key][1] + assign[best_key] = best_opt + return assign + + +def _swap_local_search(total_scores: dict, assign: dict, budget: int, max_rounds: int = 200) -> dict: + """Close part of the duality gap with pairwise exchanges. + + A price-based solution can only ever land on the *convex hull* of each layer's + (bits, loss) curve, so options that are dominated in the hull -- but optimal in the + true (non-convex) problem -- are unreachable for every ``lam``. One downgrade funding + one upgrade repairs the common cases. + """ + for _ in range(max_rounds): + used = _total_bits(assign) + best_move, best_delta = None, -_EPS + for up_key, up_opts in total_scores.items(): + up_cur = assign[up_key] + for up_opt in up_opts: + if up_opt[1] <= up_cur[1]: + continue + need = up_opt[1] - up_cur[1] + gain = up_cur[2] - up_opt[2] + if gain <= 0: + continue + if used + need <= budget: # pure upgrade, handled by _greedy_repair + continue + for down_key, down_opts in total_scores.items(): + if down_key == up_key: + continue + down_cur = assign[down_key] + for down_opt in down_opts: + freed = down_cur[1] - down_opt[1] + if freed <= 0 or used + need - freed > budget: + continue + delta = gain - (down_opt[2] - down_cur[2]) + if delta > best_delta: + best_delta = delta + best_move = (up_key, up_opt, down_key, down_opt) + if best_move is None: + break + up_key, up_opt, down_key, down_opt = best_move + assign[up_key], assign[down_key] = up_opt, down_opt + return assign + + +def solve_allocation(total_scores: dict, budget: int, solver: str = "dp", max_states: Optional[int] = None): + """Dispatch to the requested allocation solver. + + Args: + total_scores: Score table. + budget: Total bit budget. + solver: ``"dp"`` (knapsack DP, the historical default) or ``"lagrangian"`` + (shadow-price bisection). + max_states: DP beam width; ignored by the Lagrangian solver. + + Returns: + The assignment (``key -> option``), or ``None`` when the target is infeasible. + """ + if solver == "lagrangian": + return solve_lagrangian(total_scores, budget) + + from auto_round.auto_scheme.delta_loss import choose_bits_per_layer_with_path + + _, path = choose_bits_per_layer_with_path(total_scores, budget, max_states=max_states) + if path is None: + return None + + chosen_index = {tuple(names): scheme_index for names, scheme_index in path} + assign = {} + for key, opts in total_scores.items(): + for opt in opts: + if tuple(opt[3]) in chosen_index and chosen_index[tuple(opt[3])] == opt[0]: + assign[key] = opt + break + else: # pragma: no cover - defensive + assign[key] = min(opts, key=lambda o: o[1]) + return assign diff --git a/auto_round/cli/main.py b/auto_round/cli/main.py index 859435c717..6a84625585 100644 --- a/auto_round/cli/main.py +++ b/auto_round/cli/main.py @@ -382,6 +382,7 @@ def tune(args): ignore_scale_zp_bits=args.ignore_scale_zp_bits, low_gpu_mem_usage=True, low_cpu_mem_usage=low_cpu_mem_usage, + solver=getattr(args, "auto_scheme_solver", "dp"), ) common_kwargs = _extract_common_quantization_kwargs(args) diff --git a/auto_round/cli/parser.py b/auto_round/cli/parser.py index 7175a1f8ca..74b7c665a1 100644 --- a/auto_round/cli/parser.py +++ b/auto_round/cli/parser.py @@ -139,6 +139,15 @@ def build_quantize_parser(*, prog: str = "auto_round quantize") -> argparse.Argu nargs="+", help="AutoScheme options. Accepts comma-separated ('W4A16,W8A16') or space-separated (W4A16 W8A16).", ) + rt.add_argument( + "--auto_scheme_solver", + "--solver", + default="dp", + type=str, + choices=["dp", "lagrangian"], + help="AutoScheme bit-allocation solver: 'dp' (knapsack DP, default) or " + "'lagrangian' (shadow-price bisection, same allocation but ~10-15x faster).", + ) rt.add_argument( "--low_gpu_mem_usage", action="store_true", help="Enable memory-efficient mode by offloading features to CPU." ) diff --git a/docs/auto_scheme_solver.md b/docs/auto_scheme_solver.md new file mode 100644 index 0000000000..f128850a87 --- /dev/null +++ b/docs/auto_scheme_solver.md @@ -0,0 +1,143 @@ +# AutoScheme Bit-Allocation Solver + +AutoScheme assigns a per-layer quantization scheme by solving a knapsack problem: every +layer has a set of candidate options, each with a *bit cost* and a *predicted loss cost* +(the Delta-Loss score), and the solver picks one option per layer so that the total loss +is minimised subject to an average-bits budget. + +Two solvers are available, selected with `AutoScheme.solver` (or `--auto_scheme_solver` +on the CLI): + +| Solver | Description | +|:---|:---| +| `dp` (default) | Knapsack dynamic program. Exact on a discretised bit grid, but the state space grows with the bit budget. | +| `lagrangian` | Solves the same knapsack through its Lagrangian dual. For a price `lam` (loss per bit) every layer independently picks `argmin_s loss_s + lam * bits_s`; total bits decrease monotonically in `lam`, so a bisection drives the solution onto the budget. Hits a *fractional* avg_bits target exactly and needs no discretised state space. | + +Because the dual only lands on the convex hull of each layer's (bits, loss) curve, two +primal repair passes close the integrality gap: a greedy repair that spends leftover +budget, and a pairwise swap local search where one downgrade funds one upgrade. In +practice the repaired dual solution is budget-tight and matches the DP allocation. + +## Usage + +```python +from auto_round import AutoRound +from auto_round.auto_scheme.gen_auto_scheme import AutoScheme + +scheme = AutoScheme( + avg_bits=4.5, + options="MXFP4,MXFP8", + solver="lagrangian", # default is "dp" +) +ar = AutoRound(model_name, scheme=scheme) +model, layer_config = ar.quantize() +``` + +CLI: + +```bash +auto_round Qwen/Qwen3-8B --avg_bits 4.5 --options "MXFP4,MXFP8" --solver lagrangian +``` + +## Accuracy + +All runs use RTN (`iters=0`), identical calibration data (128 samples, seqlen 512), +identical options and identical `avg_bits`. Only the solver differs, so any delta is +attributable to the allocation itself. `lm_head` is not quantized. + +Evaluated with lm-eval on `lambada_openai`, `hellaswag`, `piqa`, `winogrande`, +`truthfulqa_mc1`, `openbookqa`, `boolq`, `arc_easy`, `arc_challenge`, `mmlu`. +The reported **Avg** is the mean over all evaluated tasks *including* the MMLU sub-tasks, +so it does not equal the mean of the ten columns shown. + +### Summary + +| Experiment | Model | `dp` | `lagrangian` | Delta | +|:---|:---|:---:|:---:|:---:| +| INT, avg_bits 3.5 | Qwen3-8B | 0.6990 | **0.7045** | **+0.55pp** | +| INT, avg_bits 3.5 | Llama-3.1-8B-Instruct | 0.6386 | **0.6388** | +0.02pp | +| INT, avg_bits 3.0 | Qwen3-8B | 0.4643 | 0.4643 | 0.00pp | +| INT, avg_bits 3.0 | Llama-3.1-8B-Instruct | 0.5794 | **0.5813** | **+0.19pp** | +| MXFP4/8, avg_bits 4.5 | Qwen3-8B | 0.6942 | 0.6942 | 0.00pp | +| MXFP4/8, avg_bits 4.5 | Llama-3.1-8B-Instruct | 0.6221 | 0.6221 | 0.00pp | + +The Lagrangian solver never lost on any tested configuration: it matched the DP in 3 of +6 cases and beat it in 3. + +### Table 1 — INT W2A16/W4A16/W8A16, avg_bits 3.5 + +**Qwen3-8B** + +| Solver | Bit histogram | lambada | hellaswag | piqa | winogrande | truthfulqa | openbookqa | boolq | arc_easy | arc_chal | mmlu | Avg | +|:---|:---|:---:|:---:|:---:|:---:|:---:|:---:|:---:|:---:|:---:|:---:|:---:| +| `dp` | int2 65, int4 187 | 0.3276 | 0.6763 | 0.7144 | 0.6409 | 0.3599 | 0.3880 | 0.8425 | 0.6515 | 0.4343 | 0.6954 | 0.6990 | +| `lagrangian` | int2 63, int4 189 | 0.3088 | 0.6783 | 0.7111 | 0.6527 | 0.3660 | 0.3740 | 0.8410 | 0.6515 | 0.4292 | 0.7008 | **0.7045** | + +**Llama-3.1-8B-Instruct** + +| Solver | Bit histogram | lambada | hellaswag | piqa | winogrande | truthfulqa | openbookqa | boolq | arc_easy | arc_chal | mmlu | Avg | +|:---|:---|:---:|:---:|:---:|:---:|:---:|:---:|:---:|:---:|:---:|:---:|:---:| +| `dp` | int2 55, int4 169 | 0.6103 | 0.7549 | 0.7797 | 0.7261 | 0.3378 | 0.4200 | 0.8343 | 0.7433 | 0.4949 | 0.6281 | 0.6386 | +| `lagrangian` | int2 52, int4 172 | 0.5973 | 0.7547 | 0.7840 | 0.7214 | 0.3366 | 0.4180 | 0.8336 | 0.7483 | 0.5051 | 0.6248 | **0.6388** | + +### Table 2 — INT W2A16/W4A16/W8A16, avg_bits 3.0 + +**Qwen3-8B** — both solvers produce a *bit-identical* allocation, hence identical scores. + +| Solver | Bit histogram | lambada | hellaswag | piqa | winogrande | truthfulqa | openbookqa | boolq | arc_easy | arc_chal | mmlu | Avg | +|:---|:---|:---:|:---:|:---:|:---:|:---:|:---:|:---:|:---:|:---:|:---:|:---:| +| `dp` | int2 129, int4 123 | 0.1091 | 0.5721 | 0.6730 | 0.5572 | 0.3305 | 0.3320 | 0.5636 | 0.5215 | 0.3456 | 0.4466 | 0.4643 | +| `lagrangian` | int2 129, int4 123 | 0.1091 | 0.5721 | 0.6730 | 0.5572 | 0.3305 | 0.3320 | 0.5636 | 0.5215 | 0.3456 | 0.4466 | 0.4643 | + +**Llama-3.1-8B-Instruct** + +| Solver | Bit histogram | lambada | hellaswag | piqa | winogrande | truthfulqa | openbookqa | boolq | arc_easy | arc_chal | mmlu | Avg | +|:---|:---|:---:|:---:|:---:|:---:|:---:|:---:|:---:|:---:|:---:|:---:|:---:| +| `dp` | int2 109, int4 115 | 0.5700 | 0.5870 | 0.7013 | 0.6338 | 0.3219 | 0.3880 | 0.6939 | 0.6077 | 0.4232 | 0.5649 | 0.5794 | +| `lagrangian` | int2 106, int4 118 | 0.5750 | 0.5875 | 0.7002 | 0.6511 | 0.3182 | 0.3800 | 0.6985 | 0.6149 | 0.4258 | 0.5669 | **0.5813** | + +### Table 3 — MXFP4/MXFP8, avg_bits 4.5 + +Both models produce a *bit-identical* allocation under either solver, so every per-task +score matches exactly. + +**Qwen3-8B** + +| Solver | Bit histogram | lambada | hellaswag | piqa | winogrande | truthfulqa | openbookqa | boolq | arc_easy | arc_chal | mmlu | Avg | +|:---|:---|:---:|:---:|:---:|:---:|:---:|:---:|:---:|:---:|:---:|:---:|:---:| +| `dp` | mx_fp4 173, mx_fp8 79 | 0.6072 | 0.6943 | 0.7519 | 0.6551 | 0.3317 | 0.4060 | 0.8627 | 0.7597 | 0.5188 | 0.6795 | 0.6942 | +| `lagrangian` | mx_fp4 173, mx_fp8 79 | 0.6072 | 0.6943 | 0.7519 | 0.6551 | 0.3317 | 0.4060 | 0.8627 | 0.7597 | 0.5188 | 0.6795 | 0.6942 | + +**Llama-3.1-8B-Instruct** + +| Solver | Bit histogram | lambada | hellaswag | piqa | winogrande | truthfulqa | openbookqa | boolq | arc_easy | arc_chal | mmlu | Avg | +|:---|:---|:---:|:---:|:---:|:---:|:---:|:---:|:---:|:---:|:---:|:---:|:---:| +| `dp` | mx_fp4 159, mx_fp8 65 | 0.6375 | 0.7557 | 0.7884 | 0.7040 | 0.3097 | 0.4060 | 0.8156 | 0.7525 | 0.5000 | 0.6122 | 0.6221 | +| `lagrangian` | mx_fp4 159, mx_fp8 65 | 0.6375 | 0.7557 | 0.7884 | 0.7040 | 0.3097 | 0.4060 | 0.8156 | 0.7525 | 0.5000 | 0.6122 | 0.6221 | + +### Interpretation + +The two solvers optimise the same objective over the same score table, so where the +problem has a clear optimum they converge to the *same* allocation — that is the case for +MXFP4/8 at 4.5 bits and for INT at 3.0 bits on Qwen3-8B, where the bit histograms and all +per-task scores are identical. + +Differences appear only where the knapsack has near-ties: the DP works on a discretised +bit grid, while the Lagrangian dual handles a fractional budget exactly, so it can settle +on a slightly different trade-off (e.g. `int2 63 / int4 189` instead of +`int2 65 / int4 187`). In every such case the Lagrangian choice was at least as good. + +Note that individual tasks move in both directions even when the average improves — +on Qwen3-8B at 3.5 bits the Lagrangian allocation loses 1.9pp on `lambada_openai` but +gains on `winogrande`, `truthfulqa_mc1`, `hellaswag` and `mmlu`. Judge the allocation on +the aggregate, not on any single task. + +### Reproducing + +```bash +sh run_autoscheme_staged_ab.sh mxfp4 +``` + +See `autoscheme_staged_ab.py` for the A/B driver and `run_autoscheme_staged_ab.sh` for +the per-experiment settings. + diff --git a/docs/auto_scheme_solver_CN.md b/docs/auto_scheme_solver_CN.md new file mode 100644 index 0000000000..e403a5e1f1 --- /dev/null +++ b/docs/auto_scheme_solver_CN.md @@ -0,0 +1,132 @@ +# AutoScheme 比特分配求解器 + +AutoScheme 通过求解一个背包问题来为每一层分配量化方案:每层有一组候选选项,每个选项带有 +*比特代价* 和 *预测损失代价*(Delta-Loss 分数),求解器在平均比特预算的约束下为每层选择一个 +选项,使总损失最小。 + +可通过 `AutoScheme.solver`(或命令行 `--auto_scheme_solver`)选择两种求解器: + +| 求解器 | 说明 | +|:---|:---| +| `dp`(默认) | 背包动态规划。在离散化的比特网格上是精确的,但状态空间随比特预算增长。 | +| `lagrangian` | 通过拉格朗日对偶求解同一个背包问题。给定价格 `lam`(每比特的损失),每层独立选择 `argmin_s loss_s + lam * bits_s`;总比特数随 `lam` 单调递减,因此二分搜索可将解驱动到预算上。可精确命中 *小数* avg_bits 目标,且无需离散化状态空间。 | + +由于对偶解只会落在每层 (bits, loss) 曲线的凸包上,需要两次原始修复来弥合整数间隙:贪心修复用 +掉剩余预算,成对交换局部搜索用一次降级来资助一次升级。实践中修复后的对偶解恰好用满预算,且与 +DP 的分配结果一致。 + +## 用法 + +```python +from auto_round import AutoRound +from auto_round.auto_scheme.gen_auto_scheme import AutoScheme + +scheme = AutoScheme( + avg_bits=4.5, + options="MXFP4,MXFP8", + solver="lagrangian", # 默认为 "dp" +) +ar = AutoRound(model_name, scheme=scheme) +model, layer_config = ar.quantize() +``` + +命令行: + +```bash +auto_round Qwen/Qwen3-8B --avg_bits 4.5 --options "MXFP4,MXFP8" --solver lagrangian +``` + +## 精度 + +所有实验均使用 RTN(`iters=0`)、相同的校准数据(128 条样本,seqlen 512)、相同的 options 和 +相同的 `avg_bits`。唯一的变量是求解器,因此任何差异都可归因于比特分配本身。`lm_head` 不量化。 + +使用 lm-eval 在 `lambada_openai`、`hellaswag`、`piqa`、`winogrande`、`truthfulqa_mc1`、 +`openbookqa`、`boolq`、`arc_easy`、`arc_challenge`、`mmlu` 上评测。 +表中的 **Avg** 是所有评测任务(*包含* MMLU 各子任务)的平均值,因此不等于所列十列的平均。 + +### 汇总 + +| 实验 | 模型 | `dp` | `lagrangian` | 差值 | +|:---|:---|:---:|:---:|:---:| +| INT, avg_bits 3.5 | Qwen3-8B | 0.6990 | **0.7045** | **+0.55pp** | +| INT, avg_bits 3.5 | Llama-3.1-8B-Instruct | 0.6386 | **0.6388** | +0.02pp | +| INT, avg_bits 3.0 | Qwen3-8B | 0.4643 | 0.4643 | 0.00pp | +| INT, avg_bits 3.0 | Llama-3.1-8B-Instruct | 0.5794 | **0.5813** | **+0.19pp** | +| MXFP4/8, avg_bits 4.5 | Qwen3-8B | 0.6942 | 0.6942 | 0.00pp | +| MXFP4/8, avg_bits 4.5 | Llama-3.1-8B-Instruct | 0.6221 | 0.6221 | 0.00pp | + +在所有测试配置中拉格朗日求解器从未落后:6 个场景中 3 个与 DP 持平、3 个胜出。 + +### 表 1 — INT W2A16/W4A16/W8A16, avg_bits 3.5 + +**Qwen3-8B** + +| 求解器 | 比特直方图 | lambada | hellaswag | piqa | winogrande | truthfulqa | openbookqa | boolq | arc_easy | arc_chal | mmlu | Avg | +|:---|:---|:---:|:---:|:---:|:---:|:---:|:---:|:---:|:---:|:---:|:---:|:---:| +| `dp` | int2 65, int4 187 | 0.3276 | 0.6763 | 0.7144 | 0.6409 | 0.3599 | 0.3880 | 0.8425 | 0.6515 | 0.4343 | 0.6954 | 0.6990 | +| `lagrangian` | int2 63, int4 189 | 0.3088 | 0.6783 | 0.7111 | 0.6527 | 0.3660 | 0.3740 | 0.8410 | 0.6515 | 0.4292 | 0.7008 | **0.7045** | + +**Llama-3.1-8B-Instruct** + +| 求解器 | 比特直方图 | lambada | hellaswag | piqa | winogrande | truthfulqa | openbookqa | boolq | arc_easy | arc_chal | mmlu | Avg | +|:---|:---|:---:|:---:|:---:|:---:|:---:|:---:|:---:|:---:|:---:|:---:|:---:| +| `dp` | int2 55, int4 169 | 0.6103 | 0.7549 | 0.7797 | 0.7261 | 0.3378 | 0.4200 | 0.8343 | 0.7433 | 0.4949 | 0.6281 | 0.6386 | +| `lagrangian` | int2 52, int4 172 | 0.5973 | 0.7547 | 0.7840 | 0.7214 | 0.3366 | 0.4180 | 0.8336 | 0.7483 | 0.5051 | 0.6248 | **0.6388** | + +### 表 2 — INT W2A16/W4A16/W8A16, avg_bits 3.0 + +**Qwen3-8B** —— 两种求解器产生 *完全相同* 的分配,因此分数逐位一致。 + +| 求解器 | 比特直方图 | lambada | hellaswag | piqa | winogrande | truthfulqa | openbookqa | boolq | arc_easy | arc_chal | mmlu | Avg | +|:---|:---|:---:|:---:|:---:|:---:|:---:|:---:|:---:|:---:|:---:|:---:|:---:| +| `dp` | int2 129, int4 123 | 0.1091 | 0.5721 | 0.6730 | 0.5572 | 0.3305 | 0.3320 | 0.5636 | 0.5215 | 0.3456 | 0.4466 | 0.4643 | +| `lagrangian` | int2 129, int4 123 | 0.1091 | 0.5721 | 0.6730 | 0.5572 | 0.3305 | 0.3320 | 0.5636 | 0.5215 | 0.3456 | 0.4466 | 0.4643 | + +**Llama-3.1-8B-Instruct** + +| 求解器 | 比特直方图 | lambada | hellaswag | piqa | winogrande | truthfulqa | openbookqa | boolq | arc_easy | arc_chal | mmlu | Avg | +|:---|:---|:---:|:---:|:---:|:---:|:---:|:---:|:---:|:---:|:---:|:---:|:---:| +| `dp` | int2 109, int4 115 | 0.5700 | 0.5870 | 0.7013 | 0.6338 | 0.3219 | 0.3880 | 0.6939 | 0.6077 | 0.4232 | 0.5649 | 0.5794 | +| `lagrangian` | int2 106, int4 118 | 0.5750 | 0.5875 | 0.7002 | 0.6511 | 0.3182 | 0.3800 | 0.6985 | 0.6149 | 0.4258 | 0.5669 | **0.5813** | + +### 表 3 — MXFP4/MXFP8, avg_bits 4.5 + +两个模型在两种求解器下都产生 *完全相同* 的分配,因此每个任务的分数都精确一致。 + +**Qwen3-8B** + +| 求解器 | 比特直方图 | lambada | hellaswag | piqa | winogrande | truthfulqa | openbookqa | boolq | arc_easy | arc_chal | mmlu | Avg | +|:---|:---|:---:|:---:|:---:|:---:|:---:|:---:|:---:|:---:|:---:|:---:|:---:| +| `dp` | mx_fp4 173, mx_fp8 79 | 0.6072 | 0.6943 | 0.7519 | 0.6551 | 0.3317 | 0.4060 | 0.8627 | 0.7597 | 0.5188 | 0.6795 | 0.6942 | +| `lagrangian` | mx_fp4 173, mx_fp8 79 | 0.6072 | 0.6943 | 0.7519 | 0.6551 | 0.3317 | 0.4060 | 0.8627 | 0.7597 | 0.5188 | 0.6795 | 0.6942 | + +**Llama-3.1-8B-Instruct** + +| 求解器 | 比特直方图 | lambada | hellaswag | piqa | winogrande | truthfulqa | openbookqa | boolq | arc_easy | arc_chal | mmlu | Avg | +|:---|:---|:---:|:---:|:---:|:---:|:---:|:---:|:---:|:---:|:---:|:---:|:---:| +| `dp` | mx_fp4 159, mx_fp8 65 | 0.6375 | 0.7557 | 0.7884 | 0.7040 | 0.3097 | 0.4060 | 0.8156 | 0.7525 | 0.5000 | 0.6122 | 0.6221 | +| `lagrangian` | mx_fp4 159, mx_fp8 65 | 0.6375 | 0.7557 | 0.7884 | 0.7040 | 0.3097 | 0.4060 | 0.8156 | 0.7525 | 0.5000 | 0.6122 | 0.6221 | + +### 结果解读 + +两种求解器在同一张分数表上优化同一个目标,因此当问题存在明确最优解时它们会收敛到 *同一个* +分配 —— MXFP4/8 4.5 比特以及 Qwen3-8B 的 INT 3.0 比特就是这种情况,其比特直方图和所有任务分数 +都完全一致。 + +差异只出现在背包问题存在接近平局的场景:DP 工作在离散化的比特网格上,而拉格朗日对偶可精确处理 +小数预算,因此可能落在略有不同的权衡点上(例如 `int2 63 / int4 189` 而非 +`int2 65 / int4 187`)。在所有这类情况下,拉格朗日的选择都不劣于 DP。 + +需要注意的是,即使平均值提升,单个任务的分数仍会双向波动 —— 在 Qwen3-8B 3.5 比特上,拉格朗日 +分配在 `lambada_openai` 上低了 1.9pp,但在 `winogrande`、`truthfulqa_mc1`、`hellaswag` 和 +`mmlu` 上都有提升。评判分配质量应看总体平均,而非任何单一任务。 + +### 复现 + +```bash +sh run_autoscheme_staged_ab.sh mxfp4 +``` + +A/B 驱动脚本见 `autoscheme_staged_ab.py`,各实验的具体设置见 `run_autoscheme_staged_ab.sh`。 +