diff --git a/CHANGELOG.rst b/CHANGELOG.rst index 2efd87bdda4..8af39a95fc2 100755 --- a/CHANGELOG.rst +++ b/CHANGELOG.rst @@ -13,6 +13,7 @@ Changelog *Quantization* +- Add Q8_0 weight-only quantization with 32-value GGML blocks, packed unified HF and Megatron export, and a built-in ``q8_0`` PTQ recipe. - Add Hugging Face PTQ calibration and export support for Nemotron-H MTP modules, preserving calibrated expert input scales. - Backfill checkpoint aliases for nine more published NVFP4 releases, so each is reachable from its source model's hub path: ``zai-org/GLM-5.1`` and ``GLM-5.2``, ``MiniMaxAI/MiniMax-M2.5`` and ``MiniMax-M3``, ``deepseek-ai/DeepSeek-V3.1`` and ``DeepSeek-V3.2``, ``Qwen/Qwen3-235B-A22B-Instruct-2507`` and ``-Thinking-2507``, and ``Qwen/Qwen3.6-27B``. Each imports an existing general or architecture recipe wholesale rather than copying its body. - Add composed Hugging Face AutoQuantize recipes that run fixed PTQ or weight AutoQuantize before diff --git a/docs/source/deployment/3_unified_hf.rst b/docs/source/deployment/3_unified_hf.rst index 1955506a86f..7791848b0e2 100644 --- a/docs/source/deployment/3_unified_hf.rst +++ b/docs/source/deployment/3_unified_hf.rst @@ -52,6 +52,7 @@ The unified HF export API supports the following quantization formats: 6. W4A8_AWQ - 4-bit weights and 8-bit activations with AWQ optimization 7. IQ1_S - 1-bit codebook quantization using the GGML block layout 8. IQ2_XS - 2-bit codebook quantization using the GGML block layout +9. Q8_0 - 8-bit symmetric integer quantization using the GGML block layout .. note:: GGML has no equivalent for ModelOpt's per-tensor FP8 weight-and-activation format. In particular, @@ -59,32 +60,33 @@ The unified HF export API supports the following quantization formats: activation scale semantics. Converting a ModelOpt FP8 checkpoint to GGUF therefore requires conversion to another GGML-supported tensor type rather than a lossless FP8 encoding. -IQ weight representation -~~~~~~~~~~~~~~~~~~~~~~~~ +GGML weight representation +~~~~~~~~~~~~~~~~~~~~~~~~~~ -For IQ1_S and IQ2_XS, unified export replaces each floating-point ``.weight`` with a -``uint8`` tensor containing byte-exact GGML blocks. Its shape is -``[*logical_shape[:-1], logical_shape[-1] // 256, payload_bytes]``, where ``payload_bytes`` is 50 -for IQ1_S and 74 for IQ2_XS. No separate shape tensor is stored: a loader recovers the logical -shape as ``[*weight.shape[:-2], weight.shape[-2] * 256]``. This is unambiguous because IQ export -requires the logical last dimension to be divisible by 256. +For IQ1_S, IQ2_XS, and Q8_0, unified export replaces each floating-point +``.weight`` with a ``uint8`` tensor containing byte-exact GGML blocks. Its shape is +``[*logical_shape[:-1], logical_shape[-1] // block_size, payload_bytes]``. The block size and +payload size are 256 and 50 for IQ1_S, 256 and 74 for IQ2_XS, and 32 and 34 for Q8_0. No separate +shape tensor is stored: a loader recovers the logical shape as +``[*weight.shape[:-2], weight.shape[-2] * block_size]``. This is unambiguous because export +requires each logical row to contain complete blocks. .. note:: - Megatron IQ export currently requires tensor and pipeline model parallel sizes of 1. Packing + Megatron GGML export currently requires tensor and pipeline model parallel sizes of 1. Packing happens during export, so a tensor-parallel shard would be packed as if it were a whole - weight, and a pipeline stage holding no IQ layer would not reach the same rejection as its + weight, and a pipeline stage holding no GGML layer would not reach the same rejection as its peers. Expert parallelism is supported, assuming every expert uses the same format. .. warning:: - Megatron fused-MoE IQ export is not currently supported. Its packed tensor would require the + Megatron fused-MoE GGML export is not currently supported. Its packed tensor would require the deployment consumer to understand - ``[num_experts, out_features, in_features // 256, payload_bytes]`` rather than the ordinary HF - fused-expert order. The exporter raises ``NotImplementedError`` until a deployment loader owns - this layout and is covered by an integration test. Dense and individually named expert weights - continue to use the representation above. + ``[num_experts, out_features, in_features // block_size, payload_bytes]`` rather than the + ordinary HF fused-expert order. The exporter raises ``NotImplementedError`` until a + deployment loader owns this layout and is covered by an integration test. Dense and + individually named expert weights continue to use the representation above. -The generated configuration records ``quant_method: modelopt``, ``packing: ggml``, the 256-value -block size, and the payload byte count. IQ payloads are not represented as compressed-tensors +The generated configuration records ``quant_method: modelopt``, ``packing: ggml``, the format's +block size, and the payload byte count. GGML payloads are not represented as compressed-tensors integer ``weights`` groups because all scales and indices are embedded in each packed block. Each 74-byte IQ2_XS block represents 256 logical weights: @@ -99,6 +101,10 @@ Each 74-byte IQ2_XS block represents 256 logical weights: The canonical 512-by-8 IQ2_XS codebook is part of the implementation rather than the checkpoint. The complete block therefore costs ``74 * 8 / 256 = 2.3125`` bits per logical weight. +Each 34-byte Q8_0 block represents 32 logical weights: bytes 0--1 hold the little-endian FP16 +scale, and bytes 2--33 hold 32 signed int8 quants. The block costs +``34 * 8 / 32 = 8.5`` bits per logical weight. + Minimum Framework Versions -------------------------- diff --git a/modelopt/torch/export/convert_hf_config.py b/modelopt/torch/export/convert_hf_config.py index 06ce2a116ed..d78410c661e 100644 --- a/modelopt/torch/export/convert_hf_config.py +++ b/modelopt/torch/export/convert_hf_config.py @@ -19,9 +19,9 @@ from collections import defaultdict from typing import Any -from modelopt.torch.quantization.ggml import IQ_FORMAT_REGISTRY +from modelopt.torch.quantization.ggml import GGML_FORMAT_REGISTRY -from .quant_format import IQ_FORMATS +from .quant_format import GGML_FORMATS def _quant_algo_to_group_config(quant_algo: str, group_size: int | None = None) -> dict[str, Any]: @@ -34,7 +34,7 @@ def _quant_algo_to_group_config(quant_algo: str, group_size: int | None = None) Returns: Dictionary with ``input_activations`` and ``weights`` entries suitable for a compressed-tensors ``config_groups`` entry, or ModelOpt-owned metadata for - self-contained IQ payloads. + self-contained GGML payloads. """ if quant_algo == "FP8": return { @@ -122,13 +122,13 @@ def _quant_algo_to_group_config(quant_algo: str, group_size: int | None = None) }, "weights": {"dynamic": False, "num_bits": 8, "type": "float", "group_size": gs}, } - elif quant_algo.lower() in IQ_FORMATS: - iq_format = IQ_FORMAT_REGISTRY[quant_algo.lower()] - block_size, payload_bytes = iq_format.block_size, iq_format.block_bytes - effective_bits = iq_format.effective_bits + elif quant_algo.lower() in GGML_FORMATS: + ggml_format = GGML_FORMAT_REGISTRY[quant_algo.lower()] + block_size, payload_bytes = ggml_format.block_size, ggml_format.block_bytes + effective_bits = ggml_format.effective_bits if group_size not in (None, block_size): raise ValueError(f"{quant_algo} requires group size {block_size}, got {group_size}") - # IQ payloads are self-contained blocks, not compressed-tensors integer groups. + # GGML payloads are self-contained blocks, not compressed-tensors integer groups. # Keep their format marker outside a ``weights`` quantization scheme. return { "quant_algo": quant_algo, @@ -229,13 +229,13 @@ def convert_hf_quant_config_format(input_config: dict[str, Any]) -> dict[str, An "targets": ["Linear"], } new_config["config_groups"] = {"group_0": config_group_details} - elif str(quant_algo_value).lower() in IQ_FORMATS: + elif str(quant_algo_value).lower() in GGML_FORMATS: # Forward the caller's group size so a mismatched one is rejected rather than rewritten # to the format's block size. - iq_metadata = _quant_algo_to_group_config( + ggml_metadata = _quant_algo_to_group_config( quant_algo_value, original_quantization_details.get("group_size") ) - new_config.update(iq_metadata) + new_config.update(ggml_metadata) elif quant_algo_value == "NVFP4_SVD": # NVFP4 + SVDQuant: NVFP4 weights/activations plus an AWQ-style # pre_quant_scale and a low-rank residual (svdquant_lora_a/b) stored as diff --git a/modelopt/torch/export/quant_format.py b/modelopt/torch/export/quant_format.py index 55e7d25f398..8a520f972b4 100644 --- a/modelopt/torch/export/quant_format.py +++ b/modelopt/torch/export/quant_format.py @@ -19,7 +19,7 @@ constants, for example, are in :mod:`modelopt.torch.export.trtllm.model_config`. """ -from modelopt.torch.quantization.ggml import IQ_FORMAT_REGISTRY +from modelopt.torch.quantization.ggml import GGML_FORMAT_REGISTRY, IQ_FORMAT_REGISTRY QUANTIZATION_NONE = None QUANTIZATION_FP8 = "fp8" @@ -43,22 +43,24 @@ QUANTIZATION_IQ2_XXS = "iq2_xxs" QUANTIZATION_IQ2_XS = "iq2_xs" QUANTIZATION_IQ2_S = "iq2_s" +QUANTIZATION_Q8_0 = "q8_0" -# Every GGML IQ format, derived from the registry the quantization backend dispatches through, so -# export and dispatch cannot disagree about which formats exist. They share the weight-only, -# 256-value-block, per-module-scale shape, so export treats them as one family. A format's block -# geometry and packer are read from IQ_FORMAT_REGISTRY directly. +# Every GGML format is derived from the registry the quantization backend dispatches through, so +# export and dispatch cannot disagree about which formats exist. A format's block geometry and +# packer are read from GGML_FORMAT_REGISTRY directly. IQ_FORMATS remains as the compatibility +# subset used by IQ-specific conformance tests. # # Registering a format therefore declares it exportable, and that is intended rather than a side # effect: fake quant is dequantize(quantize(w)), so a format cannot be dispatched without the # packer and block geometry that are all export reads. IQ_FORMATS = frozenset(IQ_FORMAT_REGISTRY) +GGML_FORMATS = frozenset(GGML_FORMAT_REGISTRY) # Formats whose scales are purely per-module, so export never merges them across the q/k/v # and gate/up groups that share an input. Every other format unifies input_amax (and, for # NVFP4, weight_scale_2) across such a group, which only a whole-model forward can discover. -FUSION_FREE_FORMATS = IQ_FORMATS | frozenset( +FUSION_FREE_FORMATS = GGML_FORMATS | frozenset( { QUANTIZATION_FP8, QUANTIZATION_NONE, diff --git a/modelopt/torch/export/quant_utils.py b/modelopt/torch/export/quant_utils.py index 7b779a0ac59..6450c1afc74 100755 --- a/modelopt/torch/export/quant_utils.py +++ b/modelopt/torch/export/quant_utils.py @@ -27,7 +27,7 @@ from modelopt import __version__ from modelopt.torch.models import get_spec, list_all_possible -from modelopt.torch.quantization.ggml import IQ_FORMAT_REGISTRY +from modelopt.torch.quantization.ggml import GGML_FORMAT_REGISTRY from modelopt.torch.quantization.model_calib import ( enable_stats_collection, finish_stats_collection, @@ -52,7 +52,7 @@ from ..quantization.nn import NVFP4StaticQuantizer, SequentialQuantizer, TensorQuantizer from .model_utils import TiedWeightMap, get_language_model_from_vl from .quant_format import ( - IQ_FORMATS, + GGML_FORMATS, KV_CACHE_FP8, KV_CACHE_FP8_K_NVFP4_V, KV_CACHE_INT8, @@ -444,23 +444,23 @@ def get_weight_block_size(module: nn.Module, weight_name: str = "weight") -> int def uses_iq_quantization(module) -> bool: - """Whether any weight quantizer in ``module`` or its children targets an IQ format. + """Whether any weight quantizer in ``module`` or its children targets a GGML format. ``get_quantization_format`` returns the *first* non-``NONE`` format it finds, so in a - mixed-format model IQ layers sitting behind, say, an FP8 layer are invisible to it. Callers - that must reject IQ specifically need to see every layer. + mixed-format model GGML layers sitting behind, say, an FP8 layer are invisible to it. Callers + that must reject GGML specifically need to see every layer. This reads ``num_bits`` directly rather than resolving each layer's full format, so an unrelated unsupported quantizer elsewhere in the model cannot turn the check into an error. """ for weight_name in weight_attr_names(module): weight_quantizer = representative_weight_quantizer(module, weight_name) - # getattr: a SequentialQuantizer has is_enabled but no num_bits, and is never IQ -- - # IQ is a single quantizer with backend="ggml". + # getattr: a SequentialQuantizer has is_enabled but no num_bits, and is never GGML -- + # GGML is a single quantizer with backend="ggml". if ( weight_quantizer is not None and weight_quantizer.is_enabled - and getattr(weight_quantizer, "num_bits", None) in IQ_FORMATS + and getattr(weight_quantizer, "num_bits", None) in GGML_FORMATS ): return True return any(uses_iq_quantization(child) for _, child in module.named_children()) @@ -500,21 +500,21 @@ def _get_quantization_from_layer(layer, quantizer_attr_names: QuantizerAttrNames return QUANTIZATION_W4A8_AWQ # Handle individual num_bits cases - if weight_quantizer.num_bits in IQ_FORMATS: + if weight_quantizer.num_bits in GGML_FORMATS: if weight_quantizer.backend != "ggml": - raise ValueError("IQ formats require the built-in 'ggml' quantization backend") + raise ValueError("GGML formats require the built-in 'ggml' quantization backend") # Both exporters return before collecting input_scale and before the pre_quant_scale # handling below, so an enabled activation quantizer would be dropped without a trace # and the checkpoint would load as weight-only. Refuse instead. if input_quantizer is not None and input_quantizer.is_enabled: raise NotImplementedError( - "IQ1_S/IQ2_XS export is weight-only, but this layer has an enabled input " + "GGML export is weight-only, but this layer has an enabled input " "quantizer. The GGML block payload carries no activation scale, so the " "activation quantization would be silently lost." ) if input_quantizer is not None and hasattr(input_quantizer, "_pre_quant_scale"): raise NotImplementedError( - "IQ1_S/IQ2_XS export does not support an AWQ-style pre_quant_scale." + "GGML export does not support an AWQ-style pre_quant_scale." ) return weight_quantizer.num_bits @@ -766,10 +766,10 @@ def process_layer_quant_config(layer_config_dict): "quant_algo": "MXFP8", "group_size": block_size_value, } - elif v in IQ_FORMATS: - iq_format = IQ_FORMAT_REGISTRY[v] - block_size, payload_bytes = iq_format.block_size, iq_format.block_bytes - effective_bits = iq_format.effective_bits + elif v in GGML_FORMATS: + ggml_format = GGML_FORMAT_REGISTRY[v] + block_size, payload_bytes = ggml_format.block_size, ggml_format.block_bytes + effective_bits = ggml_format.effective_bits if block_size_value != block_size: raise ValueError( f"{v.upper()} requires block size {block_size}, got {block_size_value}" diff --git a/modelopt/torch/export/unified_export_hf.py b/modelopt/torch/export/unified_export_hf.py index 92f2c0771be..4c39674f585 100644 --- a/modelopt/torch/export/unified_export_hf.py +++ b/modelopt/torch/export/unified_export_hf.py @@ -68,7 +68,7 @@ from modelopt.torch.opt.conversion import ModeloptStateManager, modelopt_state from modelopt.torch.opt.plugins.huggingface import _MODELOPT_STATE_SAVE_NAME from modelopt.torch.quantization import set_quantizer_by_cfg_context -from modelopt.torch.quantization.ggml import IQ_FORMAT_REGISTRY +from modelopt.torch.quantization.ggml import GGML_FORMAT_REGISTRY from modelopt.torch.quantization.nn import SequentialQuantizer, TensorQuantizer from modelopt.torch.quantization.qtensor import MXFP8QTensor, NVFP4QTensor from modelopt.torch.quantization.qtensor.base_qtensor import QTensorWrapper @@ -102,7 +102,7 @@ ) from .quant_format import ( FUSION_FREE_FORMATS, - IQ_FORMATS, + GGML_FORMATS, QUANTIZATION_FP8, QUANTIZATION_FP8_PB_REAL, QUANTIZATION_FP8_PC_PT, @@ -634,13 +634,13 @@ def _export_quantized_weight( "which dispatches to the streaming writer that materialises weights layer-by-layer." ) - if quantization_format in IQ_FORMATS: + if quantization_format in GGML_FORMATS: if weight_name != "weight": raise NotImplementedError( - "IQ unified export currently supports modules with a standard 'weight' " + "GGML unified export currently supports modules with a standard 'weight' " f"attribute, got {weight_name!r} on {type(sub_module).__name__}" ) - packed_weight = IQ_FORMAT_REGISTRY[quantization_format].pack( + packed_weight = GGML_FORMAT_REGISTRY[quantization_format].pack( weight.to(dtype), getattr(sub_module, quantizer_attrs.weight_quantizer, None) ) setattr(sub_module, weight_name, nn.Parameter(packed_weight, requires_grad=False)) diff --git a/modelopt/torch/export/unified_export_megatron.py b/modelopt/torch/export/unified_export_megatron.py index caa0057019e..1d6a0fc99d7 100644 --- a/modelopt/torch/export/unified_export_megatron.py +++ b/modelopt/torch/export/unified_export_megatron.py @@ -35,7 +35,7 @@ from safetensors.torch import save_file from modelopt import __version__ -from modelopt.torch.quantization.ggml import IQ_FORMAT_REGISTRY +from modelopt.torch.quantization.ggml import GGML_FORMAT_REGISTRY from modelopt.torch.quantization.nn.modules.tensor_quantizer import GroupedQuantizer from modelopt.torch.utils import import_plugin, warn_rank_0 from modelopt.torch.utils.plugins.hf_checkpoint_utils import ( @@ -57,7 +57,7 @@ ) from .plugins.megatron_importer import GPTModelImporter, _get_mamba_conv1d from .quant_format import ( - IQ_FORMATS, + GGML_FORMATS, KV_CACHE_FP8, KV_CACHE_NVFP4, QUANTIZATION_FP8, @@ -331,21 +331,19 @@ def save_pretrained( quantization_format = self._get_quantization_format(self.model) if self._any_rank_uses_iq_quantization(): - # Both sizes below are identical on every rank, and the IQ flag is agreed across + # Both sizes below are identical on every rank, and the GGML flag is agreed across # ranks, so these raise everywhere or nowhere. Raising on only a subset would strand # the rest in the collectives further down. if get_tensor_model_parallel_world_size() != 1: raise NotImplementedError( - "Megatron IQ1_S/IQ2_XS unified export currently requires tensor model " - "parallel size 1" + "Megatron GGML unified export currently requires tensor model parallel size 1" ) # Requiring PP=1 is also what makes the per-expert fused-MoE rejection safe: with # every rank holding the same layers, that check runs on all of them rather than # only the stages that happen to own an MoE block. if pp_size != 1: raise NotImplementedError( - "Megatron IQ1_S/IQ2_XS unified export currently requires pipeline model " - "parallel size 1" + "Megatron GGML unified export currently requires pipeline model parallel size 1" ) # SequentialMLP rules index experts by local position, so EP>1 would collide. Agreed across # ranks so that stages without such experts don't block in the collectives below. @@ -381,7 +379,7 @@ def save_pretrained( quantization = "NVFP4" elif quantization_format == QUANTIZATION_W4A16_NVFP4: quantization = "W4A16_NVFP4" - elif quantization_format in IQ_FORMATS: + elif quantization_format in GGML_FORMATS: quantization = quantization_format.upper() if is_last_stage_main_rank: @@ -1224,7 +1222,7 @@ def _get_quantized_state( self._record_excluded_module(prefix) block_size = get_weight_block_size(module) - is_iq = qformat in IQ_FORMATS + is_iq = qformat in GGML_FORMATS name_to_value = self._get_weight_bias( module, dtype, name_to_value, keep_weight_device=is_iq ) @@ -1234,7 +1232,7 @@ def _get_quantized_state( if qformat == QUANTIZATION_NONE: return name_to_value, qformat, block_size - # IQ formats derive all block metadata directly from the weight and do not use amax or + # GGML formats derive all block metadata directly from the weight and do not use amax or # separately exported scaling tensors. Keep the weight on-device until it can be packed # along its contraction axis, so the CUDA packer can be used. if is_iq: @@ -1259,13 +1257,13 @@ def _get_quantized_state( return name_to_value, qformat, block_size def _any_rank_uses_iq_quantization(self) -> bool: - """Whether any rank's local stage holds an IQ layer. + """Whether any rank's local stage holds a GGML layer. Two reasons this is not ``self._get_quantization_format(self.model) in (...)``. That - returns only the first non-NONE format in the tree, so a mixed-format model whose IQ + returns only the first non-NONE format in the tree, so a mixed-format model whose GGML layers follow, say, an FP8 one would slip past the caller's guard and pack TP-sharded weights as whole ones. And the scan is rank-local: under pipeline parallelism a stage - holding no IQ layer would skip the raise and then block in the next collective while its + holding no GGML layer would skip the raise and then block in the next collective while its peers exit. Agree across ranks first, mirroring ``_gather_exclude_modules``. """ return self._any_rank(uses_iq_quantization(self.model)) @@ -1302,21 +1300,21 @@ def _pack_iq_weight(weight: torch.Tensor, qformat: str, quantizer=None) -> torch Packing through ``quantizer`` reuses a payload GPTQ pinned to it, also for the row slices the fused projections are split into, rather than encoding the GPTQ'd weight again. """ - return IQ_FORMAT_REGISTRY[qformat].pack(weight, quantizer).detach().cpu() + return GGML_FORMAT_REGISTRY[qformat].pack(weight, quantizer).detach().cpu() @classmethod def _get_iq_weight_state( cls, weight_key: str, weight: torch.Tensor, qformat: str, quantizer=None ) -> dict[str, torch.Tensor]: - """Pack one ``[out, in]`` weight into the IQ checkpoint representation.""" + """Pack one ``[out, in]`` weight into the GGML checkpoint representation.""" return {weight_key: cls._pack_iq_weight(weight, qformat, quantizer)} @staticmethod def _reject_unsupported_fused_iq_export(qformat: str) -> None: - """Reject fused-expert IQ payloads until a deployment loader owns their layout. + """Reject fused-expert GGML payloads until a deployment loader owns their layout. Raised from inside the per-expert loops, so it only runs on ranks that own an expert. - The guards in ``save_pretrained`` are what make that safe: IQ export requires PP=1 and + The guards in ``save_pretrained`` are what make that safe: GGML export requires PP=1 and TP=1, so every rank holds the same layers and reaches the same loops, and expert parallelism shards a set of experts quantized alike -- so every rank arrives here with the same ``qformat`` and they raise together rather than stranding each other in a @@ -1325,10 +1323,10 @@ def _reject_unsupported_fused_iq_export(qformat: str) -> None: The one gap left is a rank holding no local expert at all, which needs expert-parallel size to exceed the expert count. Worth revisiting if that becomes a supported topology. """ - if qformat in IQ_FORMATS: + if qformat in GGML_FORMATS: raise NotImplementedError( - "Fused-MoE IQ export requires a deployment loader that supports " - "[num_experts, out_features, in_features // 256, payload_bytes]" + "Fused-MoE GGML export requires a deployment loader that supports " + "[num_experts, out_features, in_features // block_size, payload_bytes]" ) def _record_layer_quant_config(self, prefix: str, qformat: str | None, block_size: int | None): @@ -1395,7 +1393,7 @@ def _name_remapping( weight = weight + 1.0 weight_scale, weight_scale_2 = self._get_weight_scales(name_to_value, qformat) - if qformat in IQ_FORMATS: + if qformat in GGML_FORMATS: self._state_dict.update( self._get_iq_weight_state( prefix + "weight", weight, qformat, getattr(module, "weight_quantizer", None) @@ -1446,7 +1444,7 @@ def _gated_mlp_slicing( gate_proj_weight = weight[:ffn_hidden_size, :] up_proj_weight = weight[ffn_hidden_size:, :] - if qformat in IQ_FORMATS: + if qformat in GGML_FORMATS: self._state_dict.update( self._get_iq_weight_state( gate_proj_prefix + "weight", @@ -1630,7 +1628,7 @@ def _grouped_mlp_slicing( seen_qformat, seen_block_size = qformat, block_size weight = state_dict[weight_key].to(self.dtype) - if qformat not in IQ_FORMATS: + if qformat not in GGML_FORMATS: weight = weight.cpu() weight_scale_cpu = ( weight_scale.detach().cpu().clone() if weight_scale is not None else None @@ -1662,7 +1660,7 @@ def _grouped_mlp_slicing( ] for shard_prefix, shard_weight, shard_scale in shards: - if qformat in IQ_FORMATS: + if qformat in GGML_FORMATS: local_expert_state.update( self._get_iq_weight_state( shard_prefix + "weight", @@ -1840,7 +1838,7 @@ def _take(tensor, index, last_dim, with_gate=False): proj_weights = [_take(weight, s, hidden_size, g) for s, g in zip(slices, gated)] proj_keys = [p + "weight" for p in prefixes] - if qformat in IQ_FORMATS: + if qformat in GGML_FORMATS: for key, weight in zip(proj_keys, proj_weights): self._state_dict.update( self._get_iq_weight_state( @@ -1962,7 +1960,7 @@ def _gated_delta_net_slicing(self, module, prefix, is_mtp=False): proj_keys = [p + "weight" for p in proj_prefixes] weight_scale, weight_scale_2 = self._get_weight_scales(name_to_value, qformat) - if qformat in IQ_FORMATS: + if qformat in GGML_FORMATS: for proj_prefix, proj_weight in zip(proj_prefixes, proj_weights): if proj_prefix in keep_bf16: self._state_dict[proj_prefix + "weight"] = proj_weight.cpu() diff --git a/modelopt/torch/quantization/ggml/__init__.py b/modelopt/torch/quantization/ggml/__init__.py index eb6af7e2e6f..c4c5da3cba2 100644 --- a/modelopt/torch/quantization/ggml/__init__.py +++ b/modelopt/torch/quantization/ggml/__init__.py @@ -30,7 +30,7 @@ from .iq2_xxs import __all__ as _iq2_xxs_all from .q8_0 import * from .q8_0 import __all__ as _q8_0_all -from .registry import GGML_FORMAT_REGISTRY, IQ_FORMAT_REGISTRY, GGMLFormat, IQFormat +from .registry import GGML_FORMAT_REGISTRY, IQ_FORMAT_REGISTRY, GGMLFormat __all__ = [ # noqa: PLE0604 *_iq1_m_all, @@ -42,5 +42,4 @@ "GGML_FORMAT_REGISTRY", "GGMLFormat", "IQ_FORMAT_REGISTRY", - "IQFormat", ] diff --git a/modelopt/torch/quantization/ggml/common.py b/modelopt/torch/quantization/ggml/common.py index ead85c96e2e..61c9e501cfc 100644 --- a/modelopt/torch/quantization/ggml/common.py +++ b/modelopt/torch/quantization/ggml/common.py @@ -328,11 +328,6 @@ def _pinned_rows(self, quantizer, weight: torch.Tensor) -> torch.Tensor | None: return pinned[index].reshape(*weight.shape[:-1], width, self.block_bytes) -# Compatibility alias for callers that imported the record type before the registry was -# generalized from the IQ family to every supported GGML block format. -IQFormat = GGMLFormat - - def narrow_to_float32(blocks: torch.Tensor) -> torch.Tensor: """Narrow ``blocks`` to float32 the way the CUDA ``load_float`` helper does. diff --git a/modelopt/torch/quantization/ggml/iq1_m.py b/modelopt/torch/quantization/ggml/iq1_m.py index 0799a7aaeba..68d34cef4f0 100644 --- a/modelopt/torch/quantization/ggml/iq1_m.py +++ b/modelopt/torch/quantization/ggml/iq1_m.py @@ -42,7 +42,7 @@ from ..extensions import get_cuda_ext_ggml from .common import ( GGML_BLOCK_SIZE, - IQFormat, + GGMLFormat, narrow_to_float32, validate_block_chunk_size, validate_packed_weights, @@ -238,7 +238,7 @@ def dequantize_iq1_m( return decoded.reshape(shape) -IQ1_M_FORMAT = IQFormat( +IQ1_M_FORMAT = GGMLFormat( name="iq1_m", block_size=IQ1_M_BLOCK_SIZE, block_bytes=IQ1_M_BLOCK_BYTES, diff --git a/modelopt/torch/quantization/ggml/iq1_s.py b/modelopt/torch/quantization/ggml/iq1_s.py index 83aff654b8a..b98587d44a8 100644 --- a/modelopt/torch/quantization/ggml/iq1_s.py +++ b/modelopt/torch/quantization/ggml/iq1_s.py @@ -38,7 +38,7 @@ from .codebooks import iq1_s_grid_bytes from .common import ( GGML_BLOCK_SIZE, - IQFormat, + GGMLFormat, narrow_to_float32, validate_block_chunk_size, validate_packed_weights, @@ -262,7 +262,7 @@ def dequantize_iq1_s( return decoded.reshape(shape) -IQ1_S_FORMAT = IQFormat( +IQ1_S_FORMAT = GGMLFormat( name="iq1_s", block_size=IQ1_S_BLOCK_SIZE, block_bytes=IQ1_S_BLOCK_BYTES, diff --git a/modelopt/torch/quantization/ggml/iq2_s.py b/modelopt/torch/quantization/ggml/iq2_s.py index 9667301acf5..f28952d14e7 100644 --- a/modelopt/torch/quantization/ggml/iq2_s.py +++ b/modelopt/torch/quantization/ggml/iq2_s.py @@ -41,7 +41,7 @@ from .codebooks import iq2_s_grid_bytes from .common import ( GGML_BLOCK_SIZE, - IQFormat, + GGMLFormat, narrow_to_float32, validate_block_chunk_size, validate_packed_weights, @@ -236,7 +236,7 @@ def dequantize_iq2_s( return decoded.reshape(shape) -IQ2_S_FORMAT = IQFormat( +IQ2_S_FORMAT = GGMLFormat( name="iq2_s", block_size=IQ2_S_BLOCK_SIZE, block_bytes=IQ2_S_BLOCK_BYTES, diff --git a/modelopt/torch/quantization/ggml/iq2_xs.py b/modelopt/torch/quantization/ggml/iq2_xs.py index 9045d95de91..ac20c5cfee2 100644 --- a/modelopt/torch/quantization/ggml/iq2_xs.py +++ b/modelopt/torch/quantization/ggml/iq2_xs.py @@ -36,7 +36,7 @@ from .codebooks import iq2_xs_grid_bytes from .common import ( GGML_BLOCK_SIZE, - IQFormat, + GGMLFormat, narrow_to_float32, validate_block_chunk_size, validate_packed_weights, @@ -260,7 +260,7 @@ def dequantize_iq2_xs( return decoded.reshape(shape) -IQ2_XS_FORMAT = IQFormat( +IQ2_XS_FORMAT = GGMLFormat( name="iq2_xs", block_size=IQ2_XS_BLOCK_SIZE, block_bytes=IQ2_XS_BLOCK_BYTES, diff --git a/modelopt/torch/quantization/ggml/iq2_xxs.py b/modelopt/torch/quantization/ggml/iq2_xxs.py index 3ac4d8a079b..3a6b4bf1432 100644 --- a/modelopt/torch/quantization/ggml/iq2_xxs.py +++ b/modelopt/torch/quantization/ggml/iq2_xxs.py @@ -41,7 +41,7 @@ from .codebooks import iq2_xxs_grid_bytes from .common import ( GGML_BLOCK_SIZE, - IQFormat, + GGMLFormat, narrow_to_float32, validate_block_chunk_size, validate_packed_weights, @@ -270,7 +270,7 @@ def dequantize_iq2_xxs( return decoded.reshape(shape) -IQ2_XXS_FORMAT = IQFormat( +IQ2_XXS_FORMAT = GGMLFormat( name="iq2_xxs", block_size=IQ2_XXS_BLOCK_SIZE, block_bytes=IQ2_XXS_BLOCK_BYTES, diff --git a/modelopt/torch/quantization/ggml/registry.py b/modelopt/torch/quantization/ggml/registry.py index c95e9569fe7..33d26a8abd9 100644 --- a/modelopt/torch/quantization/ggml/registry.py +++ b/modelopt/torch/quantization/ggml/registry.py @@ -15,7 +15,7 @@ """GGML block formats, listed once for backend dispatch and export.""" -from .common import GGMLFormat, IQFormat +from .common import GGMLFormat from .iq1_m import IQ1_M_FORMAT from .iq1_s import IQ1_S_FORMAT from .iq2_s import IQ2_S_FORMAT @@ -23,7 +23,7 @@ from .iq2_xxs import IQ2_XXS_FORMAT from .q8_0 import Q8_0_FORMAT -__all__ = ["GGML_FORMAT_REGISTRY", "IQ_FORMAT_REGISTRY", "GGMLFormat", "IQFormat"] +__all__ = ["GGML_FORMAT_REGISTRY", "IQ_FORMAT_REGISTRY", "GGMLFormat"] # Every GGML block format, keyed by the name a quantizer's num_bits carries, in increasing bits # per weight. Backend dispatch reads this mapping, and export will derive its supported formats diff --git a/modelopt_recipes/configs/numerics/q8_0.yaml b/modelopt_recipes/configs/numerics/q8_0.yaml new file mode 100644 index 00000000000..5bbf3f3f6d5 --- /dev/null +++ b/modelopt_recipes/configs/numerics/q8_0.yaml @@ -0,0 +1,26 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +# Q8_0 weight quantizer using one symmetric int8 scale per 32 weights. + +# modelopt-schema: modelopt.torch.quantization.config.QuantizerAttributeConfig +num_bits: q8_0 +# Cost metadata for AutoQuantize's compression estimate only; it drives no packing or +# numerics. num_bits is the string "q8_0", so the generic estimator cannot derive the +# storage cost: 34 packed bytes * 8 / 32 weights. Keep in sync with Q8_0_BLOCK_BYTES. +effective_bits: 8.5 +block_sizes: + -1: 32 +backend: ggml diff --git a/modelopt_recipes/configs/ptq/presets/model/q8_0.yaml b/modelopt_recipes/configs/ptq/presets/model/q8_0.yaml new file mode 100644 index 00000000000..005d71b5c25 --- /dev/null +++ b/modelopt_recipes/configs/ptq/presets/model/q8_0.yaml @@ -0,0 +1,32 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +# QuantizeConfig preset for Q8_0 weight-only quantization. + +# modelopt-schema: modelopt.torch.quantization.config.QuantizeConfig +imports: + base_disable_all: configs/ptq/units/base_disable_all + default_disabled_quantizers: configs/ptq/units/default_disabled_quantizers + q8_0: configs/numerics/q8_0 + +algorithm: +quant_cfg: + - $import: base_disable_all + - quantizer_name: '*weight_quantizer' + cfg: + $import: q8_0 + - quantizer_name: '*input_quantizer' + enable: false + - $import: default_disabled_quantizers diff --git a/modelopt_recipes/general/ptq/q8_0.yaml b/modelopt_recipes/general/ptq/q8_0.yaml new file mode 100644 index 00000000000..96491df8b07 --- /dev/null +++ b/modelopt_recipes/general/ptq/q8_0.yaml @@ -0,0 +1,27 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +# Q8_0 weight-only PTQ. + +# modelopt-schema: modelopt.recipe.config.ModelOptPTQRecipe +imports: + preset: configs/ptq/presets/model/q8_0 + +metadata: + description: >- + Applies uniform GGML-compatible Q8_0 weight-only quantization to eligible linear layers. + This is not a mixed per-tensor precision preset. No calibration data is required. +quantize: + $import: preset diff --git a/modelopt_recipes/ptq.md b/modelopt_recipes/ptq.md index ea314f111c0..33c43e42359 100644 --- a/modelopt_recipes/ptq.md +++ b/modelopt_recipes/ptq.md @@ -28,7 +28,7 @@ supported combinations. ### The shipped recipes
-All 31 general/ptq/ recipes (click to expand) +All 32 general/ptq/ recipes (click to expand) | Recipe | Model body | KV cache | Calibration | |--------|-----------|----------|-------------| @@ -63,6 +63,7 @@ supported combinations. | `iq2_xxs` | IQ2_XXS W2A16 (2.06 bpw), MLP + MoE weights only | none | GPTQ (layerwise) | | `iq2_xs` | IQ2_XS W2A16 (2.31 bpw), MLP + MoE weights only | none | GPTQ (layerwise) | | `iq2_s` | IQ2_S W2A16 (2.56 bpw), MLP + MoE weights only | none | GPTQ (layerwise) | +| `q8_0` | Q8_0 W8A16 (8.5 bpw), eligible linears | none | none (no calibration) |
@@ -152,6 +153,12 @@ activations and tensor-core math are what deliver the throughput. Unified HF export writes the packed GGML blocks; Megatron export additionally requires tensor and pipeline parallel sizes of 1, and does not support fused-MoE experts. +- **`q8_0`** — GGML-compatible weight-only quantization on eligible linear layers, + with BF16 activations; `lm_head`, MoE routers, `conv1d` and the vision branch stay + in BF16. Q8_0 uses 8.5 bits per weight and requires no calibration data. Quantized + weights must have a final dimension divisible by 32. Unified HF and Megatron + export write packed GGML blocks; Megatron requires tensor and pipeline parallel + sizes of 1 and does not support fused-MoE experts. --- diff --git a/tests/examples/hf_ptq/test_llm_ptq.py b/tests/examples/hf_ptq/test_llm_ptq.py index 3639a7b1867..e619f88a855 100644 --- a/tests/examples/hf_ptq/test_llm_ptq.py +++ b/tests/examples/hf_ptq/test_llm_ptq.py @@ -84,6 +84,8 @@ def test_ptq_whisper(command): PTQCommand(recipe="general/ptq/iq2_xxs", kv_cache_quant="none"), PTQCommand(recipe="general/ptq/iq2_xs", kv_cache_quant="none"), PTQCommand(recipe="general/ptq/iq2_s", kv_cache_quant="none"), + # Q8_0 has 32-value blocks and needs no calibration forward pass. + PTQCommand(recipe="general/ptq/q8_0", kv_cache_quant="none"), PTQCommand(quant="nvfp4"), PTQCommand(quant="nvfp4_awq_lite"), # autoquant (recipe-driven) diff --git a/tests/gpu_megatron/torch/export/test_unified_export_megatron.py b/tests/gpu_megatron/torch/export/test_unified_export_megatron.py index 1060e1d6ea1..286700a9338 100644 --- a/tests/gpu_megatron/torch/export/test_unified_export_megatron.py +++ b/tests/gpu_megatron/torch/export/test_unified_export_megatron.py @@ -48,7 +48,7 @@ import modelopt.torch.speculative as mtsp from modelopt.torch.export import KV_CACHE_FP8, export_mcore_gpt_to_hf, import_mcore_gpt_from_hf from modelopt.torch.export.plugins.mcore_common import all_mcore_hf_export_mapping -from modelopt.torch.export.quant_format import IQ_FORMATS +from modelopt.torch.export.quant_format import GGML_FORMATS from modelopt.torch.export.unified_export_megatron import GPTModelExporter from modelopt.torch.quantization.config import QuantizerAttributeConfig from modelopt.torch.quantization.nn import TensorQuantizer @@ -92,22 +92,23 @@ def _verify_model_quant_config( assert quant_config_dict["kv_cache_quant_algo"] == KV_CACHE_FP8 -# Every IQ format the exporter accepts. Only the list of formats comes from the export -# tables; each test resolves what it expects from the codec module itself, so a wrong entry -# in IQ_FORMAT_REGISTRY cannot make both sides of an assertion agree. -IQ_FORMAT_NAMES = sorted(IQ_FORMATS) +# Every GGML format the exporter accepts. Only the list of formats comes from the export +# tables; each test resolves what it expects from the codec module itself, so a wrong registry +# entry cannot make both sides of an assertion agree. +GGML_FORMAT_NAMES = sorted(GGML_FORMATS) -@pytest.mark.parametrize("qformat", IQ_FORMAT_NAMES) -def test_megatron_name_remapping_exports_iq_payload(qformat): - """Megatron export writes the same scale-free IQ representation as HF export.""" +@pytest.mark.parametrize("qformat", GGML_FORMAT_NAMES) +def test_megatron_name_remapping_exports_ggml_payload(qformat): + """Megatron export writes the same self-contained GGML representation as HF export.""" + block_size = getattr(ggml, f"{qformat.upper()}_BLOCK_SIZE") payload_bytes = getattr(ggml, f"{qformat.upper()}_BLOCK_BYTES") dequantize = getattr(ggml, f"dequantize_{qformat}") linear = torch.nn.Linear(256, 2, bias=False, dtype=torch.bfloat16) linear.weight_quantizer = TensorQuantizer( QuantizerAttributeConfig( num_bits=qformat, - block_sizes={-1: 256}, + block_sizes={-1: block_size}, backend="ggml", ) ) @@ -121,15 +122,15 @@ def test_megatron_name_remapping_exports_iq_payload(qformat): packed_key = "model.layers.0.mlp.down_proj.weight" packed = exporter._state_dict[packed_key] - assert packed.shape == (2, 1, payload_bytes) + assert packed.shape == (2, linear.in_features // block_size, payload_bytes) assert packed.dtype == torch.uint8 # Exact bytes against the format's own packer, as the slicing tests below check. - _assert_iq_payload_matches(qformat, packed, linear.weight) + _assert_ggml_payload_matches(qformat, packed, linear.weight) # And the payload decodes to exactly what the fake quantizer reconstructs. Compare with the # decoded reference, not the fake-quant forward: that returns the straight-through form # a + (r - a), which in bf16 differs from r by up to one ULP of a -- enough to fail a # relative tolerance wherever r is small next to a, as IQ1_S's grid near zero often is. - logical_shape = torch.tensor([*packed.shape[:-2], packed.shape[-2] * 256]) + logical_shape = torch.tensor([*packed.shape[:-2], packed.shape[-2] * block_size]) reference, _ = getattr(ggml, f"quantize_{qformat}")(linear.weight) torch.testing.assert_close( dequantize(packed, logical_shape, dtype=torch.bfloat16), @@ -139,11 +140,12 @@ def test_megatron_name_remapping_exports_iq_payload(qformat): ) assert exporter.layer_config_dict == { "model.layers.0.mlp.down_proj.quantization": qformat, - "model.layers.0.mlp.down_proj.awq_block_size": 256, + "model.layers.0.mlp.down_proj.awq_block_size": block_size, } -def _make_iq_experts(qformat, layer_type, *, bias=False): +def _make_ggml_experts(qformat, layer_type, *, bias=False): + block_size = getattr(ggml, f"{qformat.upper()}_BLOCK_SIZE") experts = torch.nn.ModuleList() generator = torch.Generator().manual_seed(1234) for _ in range(2): @@ -156,7 +158,7 @@ def _make_iq_experts(qformat, layer_type, *, bias=False): linear.weight_quantizer = TensorQuantizer( QuantizerAttributeConfig( num_bits=qformat, - block_sizes={-1: 256}, + block_sizes={-1: block_size}, backend="ggml", ) ) @@ -165,7 +167,7 @@ def _make_iq_experts(qformat, layer_type, *, bias=False): return experts -def _make_iq_exporter(): +def _make_ggml_exporter(): exporter = object.__new__(GPTModelExporter) exporter.dtype = torch.bfloat16 exporter._state_dict = {} @@ -174,46 +176,48 @@ def _make_iq_exporter(): return exporter -def _make_iq_weight(rows): +def _make_ggml_weight(rows): return torch.linspace(-1, 1, rows * 256, dtype=torch.float32).reshape(rows, 256).bfloat16() -def _assert_iq_payload_matches(qformat, packed, logical_weight): +def _assert_ggml_payload_matches(qformat, packed, logical_weight): expected, _ = getattr(ggml, f"quantize_{qformat}")(logical_weight) torch.testing.assert_close(packed, expected.cpu(), rtol=0, atol=0) -@pytest.mark.parametrize("qformat", IQ_FORMAT_NAMES) -def test_megatron_gated_mlp_slicing_exports_iq_payloads(qformat): - weight = _make_iq_weight(8) +@pytest.mark.parametrize("qformat", GGML_FORMAT_NAMES) +def test_megatron_gated_mlp_slicing_exports_ggml_payloads(qformat): + block_size = getattr(ggml, f"{qformat.upper()}_BLOCK_SIZE") + weight = _make_ggml_weight(8) module = SimpleNamespace(config=SimpleNamespace(ffn_hidden_size=4)) - exporter = _make_iq_exporter() - exporter._get_quantized_state = lambda *a, **k: ({"weight": weight}, qformat, 256) + exporter = _make_ggml_exporter() + exporter._get_quantized_state = lambda *a, **k: ({"weight": weight}, qformat, block_size) exporter._gated_mlp_slicing(module, "model.layers.0.mlp.") - _assert_iq_payload_matches( + _assert_ggml_payload_matches( qformat, exporter._state_dict["model.layers.0.mlp.gate_proj.weight"], weight[:4] ) - _assert_iq_payload_matches( + _assert_ggml_payload_matches( qformat, exporter._state_dict["model.layers.0.mlp.up_proj.weight"], weight[4:] ) -@pytest.mark.parametrize("qformat", IQ_FORMAT_NAMES) -def test_megatron_grouped_mlp_slicing_exports_iq_payloads(qformat): - weight = _make_iq_weight(8) +@pytest.mark.parametrize("qformat", GGML_FORMAT_NAMES) +def test_megatron_grouped_mlp_slicing_exports_ggml_payloads(qformat): + block_size = getattr(ggml, f"{qformat.upper()}_BLOCK_SIZE") + weight = _make_ggml_weight(8) module = SimpleNamespace( num_gemms=1, weight0=weight, local_expert_indices=[0], state_dict=lambda: {"weight0": weight}, ) - exporter = _make_iq_exporter() + exporter = _make_ggml_exporter() exporter._get_quantized_state = lambda *a, **k: ( {"weight": module.weight}, qformat, - 256, + block_size, ) exporter._grouped_mlp_slicing( @@ -223,17 +227,18 @@ def test_megatron_grouped_mlp_slicing_exports_iq_payloads(qformat): up_proj_name="up_proj", ) - _assert_iq_payload_matches( + _assert_ggml_payload_matches( qformat, exporter._state_dict["model.layers.0.mlp.experts.0.gate_proj.weight"], weight[:4] ) - _assert_iq_payload_matches( + _assert_ggml_payload_matches( qformat, exporter._state_dict["model.layers.0.mlp.experts.0.up_proj.weight"], weight[4:] ) -@pytest.mark.parametrize("qformat", IQ_FORMAT_NAMES) -def test_megatron_qkv_slicing_exports_iq_payloads(qformat): - weight = _make_iq_weight(8) +@pytest.mark.parametrize("qformat", GGML_FORMAT_NAMES) +def test_megatron_qkv_slicing_exports_ggml_payloads(qformat): + block_size = getattr(ggml, f"{qformat.upper()}_BLOCK_SIZE") + weight = _make_ggml_weight(8) module = SimpleNamespace( config=SimpleNamespace( hidden_size=256, @@ -243,8 +248,8 @@ def test_megatron_qkv_slicing_exports_iq_payloads(qformat): attention_output_gate=False, ) ) - exporter = _make_iq_exporter() - exporter._get_quantized_state = lambda *a, **k: ({"weight": weight}, qformat, 256) + exporter = _make_ggml_exporter() + exporter._get_quantized_state = lambda *a, **k: ({"weight": weight}, qformat, block_size) exporter._qkv_slicing(module, "model.layers.0.self_attn.") @@ -255,30 +260,31 @@ def test_megatron_qkv_slicing_exports_iq_payloads(qformat): "v_proj": reshaped[3].reshape(2, 256), } for projection, logical_weight in expected.items(): - _assert_iq_payload_matches( + _assert_ggml_payload_matches( qformat, exporter._state_dict[f"model.layers.0.self_attn.{projection}.weight"], logical_weight, ) -@pytest.mark.parametrize("qformat", IQ_FORMAT_NAMES) -def test_megatron_gated_delta_net_slicing_exports_iq_payloads(qformat): - weight = _make_iq_weight(12) +@pytest.mark.parametrize("qformat", GGML_FORMAT_NAMES) +def test_megatron_gated_delta_net_slicing_exports_ggml_payloads(qformat): + block_size = getattr(ggml, f"{qformat.upper()}_BLOCK_SIZE") + weight = _make_ggml_weight(12) module = SimpleNamespace( in_proj=object(), in_proj_split_names=("query", "key", "value", "z", "beta", "alpha"), in_proj_split_sections=(2, 2, 2, 2, 2, 2), ) - exporter = _make_iq_exporter() - exporter._get_quantized_state = lambda *a, **k: ({"weight": weight}, qformat, 256) + exporter = _make_ggml_exporter() + exporter._get_quantized_state = lambda *a, **k: ({"weight": weight}, qformat, block_size) exporter._gated_delta_net_slicing(module, "model.layers.0.mixer.") - _assert_iq_payload_matches( + _assert_ggml_payload_matches( qformat, exporter._state_dict["model.layers.0.mixer.in_proj_qkv.weight"], weight[:6] ) - _assert_iq_payload_matches( + _assert_ggml_payload_matches( qformat, exporter._state_dict["model.layers.0.mixer.in_proj_z.weight"], weight[6:8] ) torch.testing.assert_close( @@ -289,12 +295,12 @@ def test_megatron_gated_delta_net_slicing_exports_iq_payloads(qformat): ) -@pytest.mark.parametrize("qformat", IQ_FORMAT_NAMES) -def test_megatron_packed_experts_reject_iq_without_deployment_loader(qformat): - experts = _make_iq_experts(qformat, "linear_fc2") - exporter = _make_iq_exporter() +@pytest.mark.parametrize("qformat", GGML_FORMAT_NAMES) +def test_megatron_packed_experts_reject_ggml_without_deployment_loader(qformat): + experts = _make_ggml_experts(qformat, "linear_fc2") + exporter = _make_ggml_exporter() - with pytest.raises(NotImplementedError, match="Fused-MoE IQ export requires"): + with pytest.raises(NotImplementedError, match="Fused-MoE GGML export requires"): exporter._pack_name_remapping( experts, "model.layers.0.mlp.experts.down_proj", @@ -303,12 +309,12 @@ def test_megatron_packed_experts_reject_iq_without_deployment_loader(qformat): assert exporter._state_dict == {} -@pytest.mark.parametrize("qformat", IQ_FORMAT_NAMES) -def test_megatron_gpt_oss_packed_experts_reject_iq_without_deployment_loader(qformat): - experts = _make_iq_experts(qformat, "linear_fc1", bias=True) - exporter = _make_iq_exporter() +@pytest.mark.parametrize("qformat", GGML_FORMAT_NAMES) +def test_megatron_gpt_oss_packed_experts_reject_ggml_without_deployment_loader(qformat): + experts = _make_ggml_experts(qformat, "linear_fc1", bias=True) + exporter = _make_ggml_exporter() - with pytest.raises(NotImplementedError, match="Fused-MoE IQ export requires"): + with pytest.raises(NotImplementedError, match="Fused-MoE GGML export requires"): exporter._pack_name_remapping_gpt_oss( experts, "model.layers.0.mlp.experts.gate_up_proj", @@ -317,14 +323,15 @@ def test_megatron_gpt_oss_packed_experts_reject_iq_without_deployment_loader(qfo assert exporter._state_dict == {} -@pytest.mark.parametrize("qformat", IQ_FORMAT_NAMES) -def test_megatron_iq_export_rejects_tensor_parallelism(qformat): - """IQ packing is intentionally limited to complete TP=1 weights.""" +@pytest.mark.parametrize("qformat", GGML_FORMAT_NAMES) +def test_megatron_ggml_export_rejects_tensor_parallelism(qformat): + """GGML packing is intentionally limited to complete TP=1 weights.""" + block_size = getattr(ggml, f"{qformat.upper()}_BLOCK_SIZE") linear = torch.nn.Linear(256, 2, bias=False, dtype=torch.bfloat16) linear.weight_quantizer = TensorQuantizer( QuantizerAttributeConfig( num_bits=qformat, - block_sizes={-1: 256}, + block_sizes={-1: block_size}, backend="ggml", ) ) @@ -387,19 +394,20 @@ def test_mla_export_keeps_hf_head_dim(dist_workers_size_1, tmp_path): dist_workers_size_1.run(partial(_test_mla_export_keeps_hf_head_dim, model_dir)) -@pytest.mark.parametrize("qformat", IQ_FORMAT_NAMES) -def test_megatron_iq_export_rejects_pipeline_parallelism(qformat): - """IQ packing requires PP=1 so the fused-MoE rejection reaches every rank. +@pytest.mark.parametrize("qformat", GGML_FORMAT_NAMES) +def test_megatron_ggml_export_rejects_pipeline_parallelism(qformat): + """GGML packing requires PP=1 so the fused-MoE rejection reaches every rank. The rejection raises from inside the per-expert loops, so a stage owning no expert would skip it and block in save_pretrained's collectives while its peers exit. PP=1 removes the divergence rather than trying to detect it. """ + block_size = getattr(ggml, f"{qformat.upper()}_BLOCK_SIZE") linear = torch.nn.Linear(256, 2, bias=False, dtype=torch.bfloat16) linear.weight_quantizer = TensorQuantizer( QuantizerAttributeConfig( num_bits=qformat, - block_sizes={-1: 256}, + block_sizes={-1: block_size}, backend="ggml", ) ) diff --git a/tests/unit/recipe/test_presets.py b/tests/unit/recipe/test_presets.py index c45182b0ddb..f97b2d59bd4 100644 --- a/tests/unit/recipe/test_presets.py +++ b/tests/unit/recipe/test_presets.py @@ -44,6 +44,8 @@ IQ2_XS_EFFECTIVE_BITS, IQ2_XXS_BLOCK_SIZE, IQ2_XXS_EFFECTIVE_BITS, + Q8_0_BLOCK_SIZE, + Q8_0_EFFECTIVE_BITS, ) @@ -145,6 +147,7 @@ def test_mlp_weight_only_recipe_matches_its_mtq_cfg(recipe_name, cfg_name): ("iq2_xxs", IQ2_XXS_BLOCK_SIZE, IQ2_XXS_EFFECTIVE_BITS), ("iq2_xs", IQ2_XS_BLOCK_SIZE, IQ2_XS_EFFECTIVE_BITS), ("iq2_s", IQ2_S_BLOCK_SIZE, IQ2_S_EFFECTIVE_BITS), + ("q8_0", Q8_0_BLOCK_SIZE, Q8_0_EFFECTIVE_BITS), ], ) def test_iq_recipe_matches_packing_contract(qformat, block_size, effective_bits): @@ -157,13 +160,21 @@ def test_iq_recipe_matches_packing_contract(qformat, block_size, effective_bits) } assert qformat in presets.QUANT_CFG_CHOICES - assert set(weight_cfgs) == {"*mlp*weight_quantizer", "*block_sparse_moe*weight_quantizer"} + expected_weight_quantizers = ( + {"*weight_quantizer"} + if qformat == "q8_0" + else {"*mlp*weight_quantizer", "*block_sparse_moe*weight_quantizer"} + ) + assert set(weight_cfgs) == expected_weight_quantizers for weight_cfg in weight_cfgs.values(): assert weight_cfg["backend"] == "ggml" assert weight_cfg["block_sizes"][-1] == block_size assert weight_cfg["effective_bits"] == effective_bits - assert quantize["algorithm"]["method"] == "gptq" - assert quantize["algorithm"]["block_size"] % block_size == 0 + if qformat == "q8_0": + assert quantize["algorithm"] is None + else: + assert quantize["algorithm"]["method"] == "gptq" + assert quantize["algorithm"]["block_size"] % block_size == 0 # --- RecipeSupersededAction: the flags --recipe replaces ---------------------------------------- diff --git a/tests/unit/torch/export/test_export_weight.py b/tests/unit/torch/export/test_export_weight.py index d6ae98483c7..8815788e9d6 100644 --- a/tests/unit/torch/export/test_export_weight.py +++ b/tests/unit/torch/export/test_export_weight.py @@ -28,7 +28,7 @@ _process_quantized_modules, ) from modelopt.torch.quantization.config import QuantizerAttributeConfig -from modelopt.torch.quantization.ggml import IQ_FORMAT_REGISTRY +from modelopt.torch.quantization.ggml import GGML_FORMAT_REGISTRY from modelopt.torch.quantization.nn import TensorQuantizer from modelopt.torch.quantization.utils import quantizer_attr_names @@ -113,7 +113,7 @@ def _iq_linear(num_bits, in_features=256): linear.weight_quantizer = TensorQuantizer( QuantizerAttributeConfig( num_bits=num_bits, - block_sizes={-1: 256}, + block_sizes={-1: GGML_FORMAT_REGISTRY[num_bits].block_size}, backend="ggml", ) ) @@ -121,17 +121,24 @@ def _iq_linear(num_bits, in_features=256): @pytest.mark.parametrize( - ("num_bits", "payload_bytes"), - [("iq1_s", 50), ("iq1_m", 56), ("iq2_xxs", 66), ("iq2_xs", 74), ("iq2_s", 82)], + ("num_bits", "block_size", "payload_bytes"), + [ + ("iq1_s", 256, 50), + ("iq1_m", 256, 56), + ("iq2_xxs", 256, 66), + ("iq2_xs", 256, 74), + ("iq2_s", 256, 82), + ("q8_0", 32, 34), + ], ) -def test_export_iq_payload_as_weight(num_bits, payload_bytes): +def test_export_iq_payload_as_weight(num_bits, block_size, payload_bytes): linear = _iq_linear(num_bits) _export_quantized_weight(linear, torch.bfloat16) state_dict = postprocess_state_dict(linear.state_dict(), maxbound=448, quantization=None) assert isinstance(linear.weight, nn.Parameter) - assert state_dict["weight"].shape == (4, 1, payload_bytes) + assert state_dict["weight"].shape == (4, 256 // block_size, payload_bytes) assert state_dict["weight"].dtype == torch.uint8 assert "packed_weights" not in state_dict assert "weight_shape" not in state_dict @@ -143,15 +150,15 @@ def _without_search(monkeypatch, num_bits): def search(*args, **kwargs): raise AssertionError(f"export re-ran the {num_bits} search") - record = dataclasses.replace(IQ_FORMAT_REGISTRY[num_bits], quantize=search) - monkeypatch.setitem(IQ_FORMAT_REGISTRY, num_bits, record) + record = dataclasses.replace(GGML_FORMAT_REGISTRY[num_bits], quantize=search) + monkeypatch.setitem(GGML_FORMAT_REGISTRY, num_bits, record) -@pytest.mark.parametrize("num_bits", sorted(IQ_FORMAT_REGISTRY)) +@pytest.mark.parametrize("num_bits", sorted(GGML_FORMAT_REGISTRY)) def test_export_reuses_the_payload_fake_quant_packed(monkeypatch, num_bits): """A weight fake quant already packed is exported from those bytes, not packed again. - Fake quant sees the weight reshaped into 256-value blocks, so the cached payload is keyed on a + Fake quant sees the weight reshaped into format-sized blocks, so the cached payload is keyed on a different shape than the weight export holds; a wider weight keeps that difference real. """ linear = _iq_linear(num_bits, in_features=512) @@ -163,17 +170,17 @@ def test_export_reuses_the_payload_fake_quant_packed(monkeypatch, num_bits): _export_quantized_weight(linear, torch.bfloat16) assert torch.equal(linear.weight, cache.packed_weights.reshape(linear.weight.shape)) - assert linear.weight.shape[:2] == (4, 2) + assert linear.weight.shape[:2] == (4, 512 // GGML_FORMAT_REGISTRY[num_bits].block_size) -@pytest.mark.parametrize("num_bits", sorted(IQ_FORMAT_REGISTRY)) +@pytest.mark.parametrize("num_bits", sorted(GGML_FORMAT_REGISTRY)) def test_export_repacks_a_weight_changed_since_fake_quant(num_bits): """An in-place update after the forward invalidates the cached payload.""" linear = _iq_linear(num_bits, in_features=512) linear.weight_quantizer(linear.weight) with torch.no_grad(): linear.weight.mul_(-1) - expected, _ = IQ_FORMAT_REGISTRY[num_bits].quantize(linear.weight.detach().clone()) + expected, _ = GGML_FORMAT_REGISTRY[num_bits].quantize(linear.weight.detach().clone()) _export_quantized_weight(linear, torch.bfloat16) diff --git a/tests/unit/torch/export/test_get_quantization.py b/tests/unit/torch/export/test_get_quantization.py index 147cec3b605..edf55bf4bea 100644 --- a/tests/unit/torch/export/test_get_quantization.py +++ b/tests/unit/torch/export/test_get_quantization.py @@ -36,6 +36,7 @@ QUANTIZATION_IQ1_S, QUANTIZATION_IQ2_XS, QUANTIZATION_NVFP4, + QUANTIZATION_Q8_0, QUANTIZATION_W4A8_AWQ, ) from modelopt.torch.export.quant_utils import ( @@ -65,13 +66,16 @@ def __init__(self): @pytest.mark.parametrize( - ("num_bits", "quantization_format", "payload_bytes", "effective_bits"), + ("num_bits", "quantization_format", "block_size", "payload_bytes", "effective_bits"), [ - ("iq1_s", QUANTIZATION_IQ1_S, 50, 1.5625), - ("iq2_xs", QUANTIZATION_IQ2_XS, 74, 2.3125), + ("iq1_s", QUANTIZATION_IQ1_S, 256, 50, 1.5625), + ("iq2_xs", QUANTIZATION_IQ2_XS, 256, 74, 2.3125), + ("q8_0", QUANTIZATION_Q8_0, 32, 34, 8.5), ], ) -def test_iq_quantization_config(num_bits, quantization_format, payload_bytes, effective_bits): +def test_iq_quantization_config( + num_bits, quantization_format, block_size, payload_bytes, effective_bits +): model = torch.nn.Sequential(torch.nn.Linear(256, 256, bias=False)) mtq.quantize( model, @@ -82,7 +86,7 @@ def test_iq_quantization_config(num_bits, quantization_format, payload_bytes, ef "quantizer_name": "*weight_quantizer", "cfg": { "num_bits": num_bits, - "block_sizes": {-1: 256}, + "block_sizes": {-1: block_size}, "backend": "ggml", }, }, @@ -98,7 +102,7 @@ def test_iq_quantization_config(num_bits, quantization_format, payload_bytes, ef assert config["quantization"]["effective_bits"] == effective_bits hf_config = convert_hf_quant_config_format(config) assert "config_groups" not in hf_config - assert hf_config["group_size"] == 256 + assert hf_config["group_size"] == block_size assert hf_config["effective_bits"] == effective_bits assert hf_config["packing"] == "ggml" assert hf_config["block_payload_bytes"] == payload_bytes @@ -184,6 +188,19 @@ def test_uses_iq_quantization_tolerates_sequential_quantizer(): assert not uses_iq_quantization(torch.nn.Sequential(layer)) +def test_grouped_q8_0_quantizer_is_detected_for_export(): + module = torch.nn.Module() + module.weight0 = torch.nn.Parameter(torch.empty(256, 256)) + quantizer = TensorQuantizer() + quantizer.set_from_attribute_config( + {"num_bits": "q8_0", "block_sizes": {-1: 32}, "backend": "ggml"} + ) + module.weight_quantizer = GroupedQuantizer(quantizer) + + assert uses_iq_quantization(module) + assert get_quantization_format(module) == QUANTIZATION_Q8_0 + + def test_iq_export_rejects_enabled_input_quantizer(): """IQ payloads carry no activation scale, so W-IQ + A-FP8 must not export as weight-only.""" model = _quantize_sequential( diff --git a/tests/unit/torch/quantization/test_ggml_backend.py b/tests/unit/torch/quantization/test_ggml_backend.py index 29c122c7107..136ce92f6ed 100644 --- a/tests/unit/torch/quantization/test_ggml_backend.py +++ b/tests/unit/torch/quantization/test_ggml_backend.py @@ -237,7 +237,6 @@ def test_registry_lists_every_exported_encoder(): def test_iq_registry_remains_a_compatible_subset_of_ggml_registry(): - assert ggml.IQFormat is ggml.GGMLFormat assert set(ggml.IQ_FORMAT_REGISTRY) == { name for name in GGML_FORMAT_REGISTRY if name.startswith("iq") } @@ -252,6 +251,7 @@ def test_registry_record_is_wired_to_its_own_codec(num_bits, module): record = GGML_FORMAT_REGISTRY[num_bits] upper = num_bits.upper() + assert isinstance(record, ggml.GGMLFormat) assert record.name == num_bits assert record.quantize is getattr(module, f"quantize_{num_bits}") assert record.dequantize is getattr(module, f"dequantize_{num_bits}")