From 681b83edf69c6e19008f31b24d4add899ba501df Mon Sep 17 00:00:00 2001 From: Hung-Yueh Chiang Date: Tue, 22 Sep 2026 15:13:32 -0700 Subject: [PATCH 1/6] [OMNIML-5899] Export Q8_0 checkpoints and add recipes Signed-off-by: Hung-Yueh Chiang --- CHANGELOG.rst | 1 + docs/source/deployment/3_unified_hf.rst | 40 ++++++++------ modelopt/torch/export/convert_hf_config.py | 22 ++++---- modelopt/torch/export/quant_format.py | 14 ++--- modelopt/torch/export/quant_utils.py | 34 ++++++------ modelopt/torch/export/unified_export_hf.py | 12 ++--- .../torch/export/unified_export_megatron.py | 52 +++++++++---------- modelopt_recipes/configs/numerics/q8_0.yaml | 26 ++++++++++ .../configs/ptq/presets/model/q8_0.yaml | 32 ++++++++++++ modelopt_recipes/general/ptq/q8_0.yaml | 27 ++++++++++ modelopt_recipes/ptq.md | 12 +++-- tests/examples/hf_ptq/test_llm_ptq.py | 11 ++-- .../export/test_unified_export_megatron.py | 22 ++++---- tests/unit/recipe/test_presets.py | 3 ++ tests/unit/torch/export/test_export_weight.py | 11 ++-- .../torch/export/test_get_quantization.py | 16 +++--- 16 files changed, 220 insertions(+), 115 deletions(-) create mode 100644 modelopt_recipes/configs/numerics/q8_0.yaml create mode 100644 modelopt_recipes/configs/ptq/presets/model/q8_0.yaml create mode 100644 modelopt_recipes/general/ptq/q8_0.yaml diff --git a/CHANGELOG.rst b/CHANGELOG.rst index f7360953c7a..b3c7cfec5cc 100755 --- a/CHANGELOG.rst +++ b/CHANGELOG.rst @@ -13,6 +13,7 @@ Changelog *Quantization* +- Add Q8_0 weight-only quantization with 32-value GGML blocks, packed unified HF and Megatron export, and a built-in ``q8_0`` PTQ recipe. - Add IQ1_S and IQ2_XS weight-only quantization with GGML-compatible 256-value block encoders, built-in ``iq1_s`` / ``iq2_xs`` PTQ recipes, and unified HF and Megatron export of the packed blocks. Quantized weights must have a final dimension divisible by 256, and Megatron export requires tensor and pipeline parallel sizes of 1. - Add ``iq2_xxs`` weight-only quantization with a CUDA encoder and a ``general/ptq`` recipe, at 2.0625 bits per weight between ``iq1_s`` and ``iq2_xs``. The same 256-value block constraint applies. - A recipe can now **delegate its whole body to another recipe** with a top-level ``$import``; any top-level key given alongside it overrides the imported one. ``metadata.recipe_type`` became optional along with it: a recipe states its kind with a ``# modelopt-schema:`` comment, with ``metadata.recipe_type``, or by delegating to a recipe that does, and only a recipe that another file imports has to carry the schema comment. Whatever a recipe does state must be true: a schema comment and a ``recipe_type`` must agree, and so must a recipe and the recipe it delegates to. ``modelopt_recipes/models/`` uses this for checkpoint entries that a portable recipe already reproduces: the entry aliases that recipe instead of copying it. diff --git a/docs/source/deployment/3_unified_hf.rst b/docs/source/deployment/3_unified_hf.rst index 1955506a86f..7791848b0e2 100644 --- a/docs/source/deployment/3_unified_hf.rst +++ b/docs/source/deployment/3_unified_hf.rst @@ -52,6 +52,7 @@ The unified HF export API supports the following quantization formats: 6. W4A8_AWQ - 4-bit weights and 8-bit activations with AWQ optimization 7. IQ1_S - 1-bit codebook quantization using the GGML block layout 8. IQ2_XS - 2-bit codebook quantization using the GGML block layout +9. Q8_0 - 8-bit symmetric integer quantization using the GGML block layout .. note:: GGML has no equivalent for ModelOpt's per-tensor FP8 weight-and-activation format. In particular, @@ -59,32 +60,33 @@ The unified HF export API supports the following quantization formats: activation scale semantics. Converting a ModelOpt FP8 checkpoint to GGUF therefore requires conversion to another GGML-supported tensor type rather than a lossless FP8 encoding. -IQ weight representation -~~~~~~~~~~~~~~~~~~~~~~~~ +GGML weight representation +~~~~~~~~~~~~~~~~~~~~~~~~~~ -For IQ1_S and IQ2_XS, unified export replaces each floating-point ``.weight`` with a -``uint8`` tensor containing byte-exact GGML blocks. Its shape is -``[*logical_shape[:-1], logical_shape[-1] // 256, payload_bytes]``, where ``payload_bytes`` is 50 -for IQ1_S and 74 for IQ2_XS. No separate shape tensor is stored: a loader recovers the logical -shape as ``[*weight.shape[:-2], weight.shape[-2] * 256]``. This is unambiguous because IQ export -requires the logical last dimension to be divisible by 256. +For IQ1_S, IQ2_XS, and Q8_0, unified export replaces each floating-point +``.weight`` with a ``uint8`` tensor containing byte-exact GGML blocks. Its shape is +``[*logical_shape[:-1], logical_shape[-1] // block_size, payload_bytes]``. The block size and +payload size are 256 and 50 for IQ1_S, 256 and 74 for IQ2_XS, and 32 and 34 for Q8_0. No separate +shape tensor is stored: a loader recovers the logical shape as +``[*weight.shape[:-2], weight.shape[-2] * block_size]``. This is unambiguous because export +requires each logical row to contain complete blocks. .. note:: - Megatron IQ export currently requires tensor and pipeline model parallel sizes of 1. Packing + Megatron GGML export currently requires tensor and pipeline model parallel sizes of 1. Packing happens during export, so a tensor-parallel shard would be packed as if it were a whole - weight, and a pipeline stage holding no IQ layer would not reach the same rejection as its + weight, and a pipeline stage holding no GGML layer would not reach the same rejection as its peers. Expert parallelism is supported, assuming every expert uses the same format. .. warning:: - Megatron fused-MoE IQ export is not currently supported. Its packed tensor would require the + Megatron fused-MoE GGML export is not currently supported. Its packed tensor would require the deployment consumer to understand - ``[num_experts, out_features, in_features // 256, payload_bytes]`` rather than the ordinary HF - fused-expert order. The exporter raises ``NotImplementedError`` until a deployment loader owns - this layout and is covered by an integration test. Dense and individually named expert weights - continue to use the representation above. + ``[num_experts, out_features, in_features // block_size, payload_bytes]`` rather than the + ordinary HF fused-expert order. The exporter raises ``NotImplementedError`` until a + deployment loader owns this layout and is covered by an integration test. Dense and + individually named expert weights continue to use the representation above. -The generated configuration records ``quant_method: modelopt``, ``packing: ggml``, the 256-value -block size, and the payload byte count. IQ payloads are not represented as compressed-tensors +The generated configuration records ``quant_method: modelopt``, ``packing: ggml``, the format's +block size, and the payload byte count. GGML payloads are not represented as compressed-tensors integer ``weights`` groups because all scales and indices are embedded in each packed block. Each 74-byte IQ2_XS block represents 256 logical weights: @@ -99,6 +101,10 @@ Each 74-byte IQ2_XS block represents 256 logical weights: The canonical 512-by-8 IQ2_XS codebook is part of the implementation rather than the checkpoint. The complete block therefore costs ``74 * 8 / 256 = 2.3125`` bits per logical weight. +Each 34-byte Q8_0 block represents 32 logical weights: bytes 0--1 hold the little-endian FP16 +scale, and bytes 2--33 hold 32 signed int8 quants. The block costs +``34 * 8 / 32 = 8.5`` bits per logical weight. + Minimum Framework Versions -------------------------- diff --git a/modelopt/torch/export/convert_hf_config.py b/modelopt/torch/export/convert_hf_config.py index c50e25cdf5a..5799cb81711 100644 --- a/modelopt/torch/export/convert_hf_config.py +++ b/modelopt/torch/export/convert_hf_config.py @@ -19,9 +19,9 @@ from collections import defaultdict from typing import Any -from modelopt.torch.quantization.ggml import IQ_FORMAT_REGISTRY +from modelopt.torch.quantization.ggml import GGML_FORMAT_REGISTRY -from .quant_format import IQ_FORMATS +from .quant_format import GGML_FORMATS def _quant_algo_to_group_config(quant_algo: str, group_size: int | None = None) -> dict[str, Any]: @@ -34,7 +34,7 @@ def _quant_algo_to_group_config(quant_algo: str, group_size: int | None = None) Returns: Dictionary with ``input_activations`` and ``weights`` entries suitable for a compressed-tensors ``config_groups`` entry, or ModelOpt-owned metadata for - self-contained IQ payloads. + self-contained GGML payloads. """ if quant_algo == "FP8": return { @@ -122,13 +122,13 @@ def _quant_algo_to_group_config(quant_algo: str, group_size: int | None = None) }, "weights": {"dynamic": False, "num_bits": 8, "type": "float", "group_size": gs}, } - elif quant_algo.lower() in IQ_FORMATS: - iq_format = IQ_FORMAT_REGISTRY[quant_algo.lower()] - block_size, payload_bytes = iq_format.block_size, iq_format.block_bytes - effective_bits = iq_format.effective_bits + elif quant_algo.lower() in GGML_FORMATS: + ggml_format = GGML_FORMAT_REGISTRY[quant_algo.lower()] + block_size, payload_bytes = ggml_format.block_size, ggml_format.block_bytes + effective_bits = ggml_format.effective_bits if group_size not in (None, block_size): raise ValueError(f"{quant_algo} requires group size {block_size}, got {group_size}") - # IQ payloads are self-contained blocks, not compressed-tensors integer groups. + # GGML payloads are self-contained blocks, not compressed-tensors integer groups. # Keep their format marker outside a ``weights`` quantization scheme. return { "quant_algo": quant_algo, @@ -229,13 +229,13 @@ def convert_hf_quant_config_format(input_config: dict[str, Any]) -> dict[str, An "targets": ["Linear"], } new_config["config_groups"] = {"group_0": config_group_details} - elif str(quant_algo_value).lower() in IQ_FORMATS: + elif str(quant_algo_value).lower() in GGML_FORMATS: # Forward the caller's group size so a mismatched one is rejected rather than rewritten # to the format's block size. - iq_metadata = _quant_algo_to_group_config( + ggml_metadata = _quant_algo_to_group_config( quant_algo_value, original_quantization_details.get("group_size") ) - new_config.update(iq_metadata) + new_config.update(ggml_metadata) elif quant_algo_value == "NVFP4_SVD": # NVFP4 + SVDQuant: NVFP4 weights/activations plus an AWQ-style # pre_quant_scale and a low-rank residual (svdquant_lora_a/b) stored as diff --git a/modelopt/torch/export/quant_format.py b/modelopt/torch/export/quant_format.py index ffe655c2d71..dd47c59b3fa 100644 --- a/modelopt/torch/export/quant_format.py +++ b/modelopt/torch/export/quant_format.py @@ -19,7 +19,7 @@ constants, for example, are in :mod:`modelopt.torch.export.trtllm.model_config`. """ -from modelopt.torch.quantization.ggml import IQ_FORMAT_REGISTRY +from modelopt.torch.quantization.ggml import GGML_FORMAT_REGISTRY, IQ_FORMAT_REGISTRY QUANTIZATION_NONE = None QUANTIZATION_FP8 = "fp8" @@ -41,22 +41,24 @@ QUANTIZATION_IQ1_S = "iq1_s" QUANTIZATION_IQ2_XXS = "iq2_xxs" QUANTIZATION_IQ2_XS = "iq2_xs" +QUANTIZATION_Q8_0 = "q8_0" -# Every GGML IQ format, derived from the registry the quantization backend dispatches through, so -# export and dispatch cannot disagree about which formats exist. They share the weight-only, -# 256-value-block, per-module-scale shape, so export treats them as one family. A format's block -# geometry and packer are read from IQ_FORMAT_REGISTRY directly. +# Every GGML format is derived from the registry the quantization backend dispatches through, so +# export and dispatch cannot disagree about which formats exist. A format's block geometry and +# packer are read from GGML_FORMAT_REGISTRY directly. IQ_FORMATS remains the vector-codebook +# subset for callers that specifically need it. # # Registering a format therefore declares it exportable, and that is intended rather than a side # effect: fake quant is dequantize(quantize(w)), so a format cannot be dispatched without the # packer and block geometry that are all export reads. IQ_FORMATS = frozenset(IQ_FORMAT_REGISTRY) +GGML_FORMATS = frozenset(GGML_FORMAT_REGISTRY) # Formats whose scales are purely per-module, so export never merges them across the q/k/v # and gate/up groups that share an input. Every other format unifies input_amax (and, for # NVFP4, weight_scale_2) across such a group, which only a whole-model forward can discover. -FUSION_FREE_FORMATS = IQ_FORMATS | frozenset( +FUSION_FREE_FORMATS = GGML_FORMATS | frozenset( { QUANTIZATION_FP8, QUANTIZATION_NONE, diff --git a/modelopt/torch/export/quant_utils.py b/modelopt/torch/export/quant_utils.py index 5226ec94c3e..bea78ee7546 100755 --- a/modelopt/torch/export/quant_utils.py +++ b/modelopt/torch/export/quant_utils.py @@ -27,7 +27,7 @@ from modelopt import __version__ from modelopt.torch.models import get_spec, list_all_possible -from modelopt.torch.quantization.ggml import IQ_FORMAT_REGISTRY +from modelopt.torch.quantization.ggml import GGML_FORMAT_REGISTRY from modelopt.torch.quantization.model_calib import ( enable_stats_collection, finish_stats_collection, @@ -52,7 +52,7 @@ from ..quantization.nn import NVFP4StaticQuantizer, SequentialQuantizer, TensorQuantizer from .model_utils import TiedWeightMap, get_language_model_from_vl from .quant_format import ( - IQ_FORMATS, + GGML_FORMATS, KV_CACHE_FP8, KV_CACHE_FP8_K_NVFP4_V, KV_CACHE_INT8, @@ -444,11 +444,11 @@ def get_weight_block_size(module: nn.Module, weight_name: str = "weight") -> int def uses_iq_quantization(module) -> bool: - """Whether any weight quantizer in ``module`` or its children targets an IQ format. + """Whether any weight quantizer in ``module`` or its children targets a GGML format. ``get_quantization_format`` returns the *first* non-``NONE`` format it finds, so in a - mixed-format model IQ layers sitting behind, say, an FP8 layer are invisible to it. Callers - that must reject IQ specifically need to see every layer. + mixed-format model GGML layers sitting behind, say, an FP8 layer are invisible to it. Callers + that must reject GGML specifically need to see every layer. This reads ``num_bits`` directly rather than resolving each layer's full format, so an unrelated unsupported quantizer elsewhere in the model cannot turn the check into an error. @@ -456,17 +456,17 @@ def uses_iq_quantization(module) -> bool: Known gap, shared with ``get_quantization_format``: ``weight_attr_names`` yields nothing for a TEGroupedLinear, whose parameters are ``weight0..N`` while its quantizer is a single ``GroupedQuantizer`` under ``weight_quantizer``. Neither function sees such a module, so an - experts-only IQ model reports no format at all -- not just here. Closing it belongs in + experts-only GGML model reports no format at all -- not just here. Closing it belongs in ``weight_attr_names``, where it affects every format, rather than in this helper. """ for weight_name in weight_attr_names(module): weight_quantizer = representative_weight_quantizer(module, weight_name) - # getattr: a SequentialQuantizer has is_enabled but no num_bits, and is never IQ -- - # IQ is a single quantizer with backend="ggml". + # getattr: a SequentialQuantizer has is_enabled but no num_bits, and is never GGML -- + # GGML is a single quantizer with backend="ggml". if ( weight_quantizer is not None and weight_quantizer.is_enabled - and getattr(weight_quantizer, "num_bits", None) in IQ_FORMATS + and getattr(weight_quantizer, "num_bits", None) in GGML_FORMATS ): return True return any(uses_iq_quantization(child) for _, child in module.named_children()) @@ -506,21 +506,21 @@ def _get_quantization_from_layer(layer, quantizer_attr_names: QuantizerAttrNames return QUANTIZATION_W4A8_AWQ # Handle individual num_bits cases - if weight_quantizer.num_bits in IQ_FORMATS: + if weight_quantizer.num_bits in GGML_FORMATS: if weight_quantizer.backend != "ggml": - raise ValueError("IQ formats require the built-in 'ggml' quantization backend") + raise ValueError("GGML formats require the built-in 'ggml' quantization backend") # Both exporters return before collecting input_scale and before the pre_quant_scale # handling below, so an enabled activation quantizer would be dropped without a trace # and the checkpoint would load as weight-only. Refuse instead. if input_quantizer is not None and input_quantizer.is_enabled: raise NotImplementedError( - "IQ1_S/IQ2_XS export is weight-only, but this layer has an enabled input " + "GGML export is weight-only, but this layer has an enabled input " "quantizer. The GGML block payload carries no activation scale, so the " "activation quantization would be silently lost." ) if input_quantizer is not None and hasattr(input_quantizer, "_pre_quant_scale"): raise NotImplementedError( - "IQ1_S/IQ2_XS export does not support an AWQ-style pre_quant_scale." + "GGML export does not support an AWQ-style pre_quant_scale." ) return weight_quantizer.num_bits @@ -772,10 +772,10 @@ def process_layer_quant_config(layer_config_dict): "quant_algo": "MXFP8", "group_size": block_size_value, } - elif v in IQ_FORMATS: - iq_format = IQ_FORMAT_REGISTRY[v] - block_size, payload_bytes = iq_format.block_size, iq_format.block_bytes - effective_bits = iq_format.effective_bits + elif v in GGML_FORMATS: + ggml_format = GGML_FORMAT_REGISTRY[v] + block_size, payload_bytes = ggml_format.block_size, ggml_format.block_bytes + effective_bits = ggml_format.effective_bits if block_size_value != block_size: raise ValueError( f"{v.upper()} requires block size {block_size}, got {block_size_value}" diff --git a/modelopt/torch/export/unified_export_hf.py b/modelopt/torch/export/unified_export_hf.py index 64e158c05bb..93e62b66b56 100644 --- a/modelopt/torch/export/unified_export_hf.py +++ b/modelopt/torch/export/unified_export_hf.py @@ -67,7 +67,7 @@ from modelopt.torch.opt.conversion import ModeloptStateManager, modelopt_state from modelopt.torch.opt.plugins.huggingface import _MODELOPT_STATE_SAVE_NAME from modelopt.torch.quantization import set_quantizer_by_cfg_context -from modelopt.torch.quantization.ggml import IQ_FORMAT_REGISTRY +from modelopt.torch.quantization.ggml import GGML_FORMAT_REGISTRY from modelopt.torch.quantization.nn import SequentialQuantizer, TensorQuantizer from modelopt.torch.quantization.qtensor import MXFP8QTensor, NVFP4QTensor from modelopt.torch.quantization.qtensor.base_qtensor import QTensorWrapper @@ -101,7 +101,7 @@ ) from .quant_format import ( FUSION_FREE_FORMATS, - IQ_FORMATS, + GGML_FORMATS, QUANTIZATION_FP8, QUANTIZATION_FP8_PB_REAL, QUANTIZATION_FP8_PC_PT, @@ -629,14 +629,14 @@ def _export_quantized_weight( "which dispatches to the streaming writer that materialises weights layer-by-layer." ) - if quantization_format in IQ_FORMATS: + if quantization_format in GGML_FORMATS: if weight_name != "weight": raise NotImplementedError( - "IQ unified export currently supports modules with a standard 'weight' " + "GGML unified export currently supports modules with a standard 'weight' " f"attribute, got {weight_name!r} on {type(sub_module).__name__}" ) - quantize_iq = IQ_FORMAT_REGISTRY[quantization_format].quantize - packed_weight, _ = quantize_iq(weight.to(dtype)) + quantize_ggml = GGML_FORMAT_REGISTRY[quantization_format].quantize + packed_weight, _ = quantize_ggml(weight.to(dtype)) setattr(sub_module, weight_name, nn.Parameter(packed_weight, requires_grad=False)) maybe_clear_cuda_cache() return diff --git a/modelopt/torch/export/unified_export_megatron.py b/modelopt/torch/export/unified_export_megatron.py index 9a87632d417..9ce88ca0c3f 100644 --- a/modelopt/torch/export/unified_export_megatron.py +++ b/modelopt/torch/export/unified_export_megatron.py @@ -35,7 +35,7 @@ from safetensors.torch import save_file from modelopt import __version__ -from modelopt.torch.quantization.ggml import IQ_FORMAT_REGISTRY +from modelopt.torch.quantization.ggml import GGML_FORMAT_REGISTRY from modelopt.torch.quantization.nn.modules.tensor_quantizer import GroupedQuantizer from modelopt.torch.utils import import_plugin, warn_rank_0 from modelopt.torch.utils.plugins.hf_checkpoint_utils import ( @@ -57,7 +57,7 @@ ) from .plugins.megatron_importer import GPTModelImporter, _get_mamba_conv1d from .quant_format import ( - IQ_FORMATS, + GGML_FORMATS, KV_CACHE_FP8, KV_CACHE_NVFP4, QUANTIZATION_FP8, @@ -319,21 +319,19 @@ def save_pretrained( quantization_format = self._get_quantization_format(self.model) if self._any_rank_uses_iq_quantization(): - # Both sizes below are identical on every rank, and the IQ flag is agreed across + # Both sizes below are identical on every rank, and the GGML flag is agreed across # ranks, so these raise everywhere or nowhere. Raising on only a subset would strand # the rest in the collectives further down. if get_tensor_model_parallel_world_size() != 1: raise NotImplementedError( - "Megatron IQ1_S/IQ2_XS unified export currently requires tensor model " - "parallel size 1" + "Megatron GGML unified export currently requires tensor model parallel size 1" ) # Requiring PP=1 is also what makes the per-expert fused-MoE rejection safe: with # every rank holding the same layers, that check runs on all of them rather than # only the stages that happen to own an MoE block. if pp_size != 1: raise NotImplementedError( - "Megatron IQ1_S/IQ2_XS unified export currently requires pipeline model " - "parallel size 1" + "Megatron GGML unified export currently requires pipeline model parallel size 1" ) # Main export process @@ -351,7 +349,7 @@ def save_pretrained( quantization = "NVFP4" elif quantization_format == QUANTIZATION_W4A16_NVFP4: quantization = "W4A16_NVFP4" - elif quantization_format in IQ_FORMATS: + elif quantization_format in GGML_FORMATS: quantization = quantization_format.upper() if is_last_stage_main_rank: @@ -1115,7 +1113,7 @@ def _get_quantized_state( self._record_excluded_module(prefix) block_size = get_weight_block_size(module) - is_iq = qformat in IQ_FORMATS + is_iq = qformat in GGML_FORMATS name_to_value = self._get_weight_bias( module, dtype, name_to_value, keep_weight_device=is_iq ) @@ -1125,7 +1123,7 @@ def _get_quantized_state( if qformat == QUANTIZATION_NONE: return name_to_value, qformat, block_size - # IQ formats derive all block metadata directly from the weight and do not use amax or + # GGML formats derive all block metadata directly from the weight and do not use amax or # separately exported scaling tensors. Keep the weight on-device until it can be packed # along its contraction axis, so the CUDA packer can be used. if is_iq: @@ -1150,13 +1148,13 @@ def _get_quantized_state( return name_to_value, qformat, block_size def _any_rank_uses_iq_quantization(self) -> bool: - """Whether any rank's local stage holds an IQ layer. + """Whether any rank's local stage holds a GGML layer. Two reasons this is not ``self._get_quantization_format(self.model) in (...)``. That - returns only the first non-NONE format in the tree, so a mixed-format model whose IQ + returns only the first non-NONE format in the tree, so a mixed-format model whose GGML layers follow, say, an FP8 one would slip past the caller's guard and pack TP-sharded weights as whole ones. And the scan is rank-local: under pipeline parallelism a stage - holding no IQ layer would skip the raise and then block in the next collective while its + holding no GGML layer would skip the raise and then block in the next collective while its peers exit. Agree across ranks first, mirroring ``_gather_exclude_modules``. """ local_uses_iq = uses_iq_quantization(self.model) @@ -1185,23 +1183,23 @@ def _get_weight_scales(self, quantized_state: dict[str, Any], qformat: str): @staticmethod def _pack_iq_weight(weight: torch.Tensor, qformat: str) -> torch.Tensor: """Pack one ``[out, in]`` weight and return its CPU payload.""" - quantize_iq = IQ_FORMAT_REGISTRY[qformat].quantize - packed_weight, _ = quantize_iq(weight) + quantize_ggml = GGML_FORMAT_REGISTRY[qformat].quantize + packed_weight, _ = quantize_ggml(weight) return packed_weight.detach().cpu() @classmethod def _get_iq_weight_state( cls, weight_key: str, weight: torch.Tensor, qformat: str ) -> dict[str, torch.Tensor]: - """Pack one ``[out, in]`` weight into the IQ checkpoint representation.""" + """Pack one ``[out, in]`` weight into the GGML checkpoint representation.""" return {weight_key: cls._pack_iq_weight(weight, qformat)} @staticmethod def _reject_unsupported_fused_iq_export(qformat: str) -> None: - """Reject fused-expert IQ payloads until a deployment loader owns their layout. + """Reject fused-expert GGML payloads until a deployment loader owns their layout. Raised from inside the per-expert loops, so it only runs on ranks that own an expert. - The guards in ``save_pretrained`` are what make that safe: IQ export requires PP=1 and + The guards in ``save_pretrained`` are what make that safe: GGML export requires PP=1 and TP=1, so every rank holds the same layers and reaches the same loops, and expert parallelism shards a set of experts quantized alike -- so every rank arrives here with the same ``qformat`` and they raise together rather than stranding each other in a @@ -1210,10 +1208,10 @@ def _reject_unsupported_fused_iq_export(qformat: str) -> None: The one gap left is a rank holding no local expert at all, which needs expert-parallel size to exceed the expert count. Worth revisiting if that becomes a supported topology. """ - if qformat in IQ_FORMATS: + if qformat in GGML_FORMATS: raise NotImplementedError( - "Fused-MoE IQ export requires a deployment loader that supports " - "[num_experts, out_features, in_features // 256, payload_bytes]" + "Fused-MoE GGML export requires a deployment loader that supports " + "[num_experts, out_features, in_features // block_size, payload_bytes]" ) def _record_layer_quant_config(self, prefix: str, qformat: str | None, block_size: int | None): @@ -1280,7 +1278,7 @@ def _name_remapping( weight = weight + 1.0 weight_scale, weight_scale_2 = self._get_weight_scales(name_to_value, qformat) - if qformat in IQ_FORMATS: + if qformat in GGML_FORMATS: self._state_dict.update(self._get_iq_weight_state(prefix + "weight", weight, qformat)) elif weight_scale is None: self._state_dict[prefix + "weight"] = weight @@ -1327,7 +1325,7 @@ def _gated_mlp_slicing( gate_proj_weight = weight[:ffn_hidden_size, :] up_proj_weight = weight[ffn_hidden_size:, :] - if qformat in IQ_FORMATS: + if qformat in GGML_FORMATS: self._state_dict.update( self._get_iq_weight_state(gate_proj_prefix + "weight", gate_proj_weight, qformat) ) @@ -1501,7 +1499,7 @@ def _grouped_mlp_slicing( seen_qformat, seen_block_size = qformat, block_size weight = state_dict[weight_key].to(self.dtype) - if qformat not in IQ_FORMATS: + if qformat not in GGML_FORMATS: weight = weight.cpu() weight_scale_cpu = ( weight_scale.detach().cpu().clone() if weight_scale is not None else None @@ -1533,7 +1531,7 @@ def _grouped_mlp_slicing( ] for shard_prefix, shard_weight, shard_scale in shards: - if qformat in IQ_FORMATS: + if qformat in GGML_FORMATS: local_expert_state.update( self._get_iq_weight_state( shard_prefix + "weight", shard_weight, qformat @@ -1702,7 +1700,7 @@ def _take(tensor, index, last_dim, with_gate=False): proj_weights = [_take(weight, s, hidden_size, g) for s, g in zip(slices, gated)] proj_keys = [p + "weight" for p in prefixes] - if qformat in IQ_FORMATS: + if qformat in GGML_FORMATS: for key, weight in zip(proj_keys, proj_weights): self._state_dict.update(self._get_iq_weight_state(key, weight, qformat)) elif weight_scale is None: @@ -1820,7 +1818,7 @@ def _gated_delta_net_slicing(self, module, prefix, is_mtp=False): proj_keys = [p + "weight" for p in proj_prefixes] weight_scale, weight_scale_2 = self._get_weight_scales(name_to_value, qformat) - if qformat in IQ_FORMATS: + if qformat in GGML_FORMATS: for proj_prefix, proj_weight in zip(proj_prefixes, proj_weights): if proj_prefix in keep_bf16: self._state_dict[proj_prefix + "weight"] = proj_weight.cpu() diff --git a/modelopt_recipes/configs/numerics/q8_0.yaml b/modelopt_recipes/configs/numerics/q8_0.yaml new file mode 100644 index 00000000000..5bbf3f3f6d5 --- /dev/null +++ b/modelopt_recipes/configs/numerics/q8_0.yaml @@ -0,0 +1,26 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +# Q8_0 weight quantizer using one symmetric int8 scale per 32 weights. + +# modelopt-schema: modelopt.torch.quantization.config.QuantizerAttributeConfig +num_bits: q8_0 +# Cost metadata for AutoQuantize's compression estimate only; it drives no packing or +# numerics. num_bits is the string "q8_0", so the generic estimator cannot derive the +# storage cost: 34 packed bytes * 8 / 32 weights. Keep in sync with Q8_0_BLOCK_BYTES. +effective_bits: 8.5 +block_sizes: + -1: 32 +backend: ggml diff --git a/modelopt_recipes/configs/ptq/presets/model/q8_0.yaml b/modelopt_recipes/configs/ptq/presets/model/q8_0.yaml new file mode 100644 index 00000000000..005d71b5c25 --- /dev/null +++ b/modelopt_recipes/configs/ptq/presets/model/q8_0.yaml @@ -0,0 +1,32 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +# QuantizeConfig preset for Q8_0 weight-only quantization. + +# modelopt-schema: modelopt.torch.quantization.config.QuantizeConfig +imports: + base_disable_all: configs/ptq/units/base_disable_all + default_disabled_quantizers: configs/ptq/units/default_disabled_quantizers + q8_0: configs/numerics/q8_0 + +algorithm: +quant_cfg: + - $import: base_disable_all + - quantizer_name: '*weight_quantizer' + cfg: + $import: q8_0 + - quantizer_name: '*input_quantizer' + enable: false + - $import: default_disabled_quantizers diff --git a/modelopt_recipes/general/ptq/q8_0.yaml b/modelopt_recipes/general/ptq/q8_0.yaml new file mode 100644 index 00000000000..96491df8b07 --- /dev/null +++ b/modelopt_recipes/general/ptq/q8_0.yaml @@ -0,0 +1,27 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +# Q8_0 weight-only PTQ. + +# modelopt-schema: modelopt.recipe.config.ModelOptPTQRecipe +imports: + preset: configs/ptq/presets/model/q8_0 + +metadata: + description: >- + Applies uniform GGML-compatible Q8_0 weight-only quantization to eligible linear layers. + This is not a mixed per-tensor precision preset. No calibration data is required. +quantize: + $import: preset diff --git a/modelopt_recipes/ptq.md b/modelopt_recipes/ptq.md index ed1d1e003cb..5ac560684a5 100644 --- a/modelopt_recipes/ptq.md +++ b/modelopt_recipes/ptq.md @@ -61,6 +61,7 @@ supported combinations. | `iq1_s` | IQ1_S W1A16, eligible linears | none | none (no calibration) | | `iq2_xxs` | IQ2_XXS W2A16 (2.06 bpw), eligible linears | none | none (no calibration) | | `iq2_xs` | IQ2_XS W2A16 (2.31 bpw), eligible linears | none | none (no calibration) | +| `q8_0` | Q8_0 W8A16 (8.5 bpw), eligible linears | none | none (no calibration) | @@ -139,11 +140,12 @@ activations and tensor-core math are what deliver the throughput. - **`mxfp4_mlp_weight_only`** — MXFP4 weights on MLP/MoE layers only, BF16 activations. Needs no calibration forward pass; the QAT starting point for the GPT-OSS family (see `examples/gpt-oss`). -- **`iq1_s` / `iq2_xxs` / `iq2_xs`** — GGML-compatible IQ weights - on the eligible linear layers, with BF16 activations; `lm_head`, MoE routers, - `conv1d` and the vision branch stay in BF16 like every other preset. The formats - trade size against accuracy in order: 1.56, 2.06 and 2.31 bits per weight. No calibration data is - required. Quantized weights must have a final dimension divisible by 256. +- **`iq1_s` / `iq2_xxs` / `iq2_xs` / `q8_0`** — GGML-compatible weight-only + quantization on the eligible linear layers, with BF16 activations; `lm_head`, MoE routers, + `conv1d` and the vision branch stay in BF16 like every other preset. The IQ formats + trade size against accuracy at 1.56, 2.06 and 2.31 bits per weight; Q8_0 uses 8.5 bits per + weight. No calibration data is required. IQ weights must have a final dimension divisible by + 256; Q8_0 requires divisibility by 32. Unified HF export writes the packed GGML blocks; Megatron export additionally requires tensor and pipeline parallel sizes of 1, and does not support fused-MoE experts. diff --git a/tests/examples/hf_ptq/test_llm_ptq.py b/tests/examples/hf_ptq/test_llm_ptq.py index 98906428372..5031f1022d2 100644 --- a/tests/examples/hf_ptq/test_llm_ptq.py +++ b/tests/examples/hf_ptq/test_llm_ptq.py @@ -76,15 +76,14 @@ def test_ptq_whisper(command): PTQCommand(quant="int8_weight_only", kv_cache_quant="none"), PTQCommand(quant="int4_awq", kv_cache_quant="none"), PTQCommand(quant="w4a8_awq_beta", kv_cache_quant="none"), - # GGML IQ weight-only, recipe-driven: three formats between 1.56 and 2.31 bits - # per weight. These encoders require every weight's input dimension to be a - # multiple of 256; TinyLlama's 2048 and 5632 both are. None of them calibrates -- - # every recipe sets algorithm: null -- so the only IQ-specific cost is packing each - # weight once on a CUDA encoder and decoding it on each forward, which fits the - # 300s tests/examples default. + # GGML weight-only, recipe-driven. The IQ encoders require every weight's input + # dimension to be a multiple of 256; Q8_0 requires 32. TinyLlama's 2048 and 5632 + # satisfy both. These recipes set algorithm: null, so the format-specific cost is + # packing each weight once and decoding it on each forward. PTQCommand(recipe="general/ptq/iq1_s", kv_cache_quant="none"), PTQCommand(recipe="general/ptq/iq2_xxs", kv_cache_quant="none"), PTQCommand(recipe="general/ptq/iq2_xs", kv_cache_quant="none"), + PTQCommand(recipe="general/ptq/q8_0", kv_cache_quant="none"), PTQCommand(quant="nvfp4"), PTQCommand(quant="nvfp4_awq_lite"), # autoquant (recipe-driven) diff --git a/tests/gpu_megatron/torch/export/test_unified_export_megatron.py b/tests/gpu_megatron/torch/export/test_unified_export_megatron.py index 561cd6ae050..906e4e728d6 100644 --- a/tests/gpu_megatron/torch/export/test_unified_export_megatron.py +++ b/tests/gpu_megatron/torch/export/test_unified_export_megatron.py @@ -46,7 +46,7 @@ import modelopt.torch.quantization.ggml as ggml import modelopt.torch.speculative as mtsp from modelopt.torch.export import KV_CACHE_FP8, export_mcore_gpt_to_hf, import_mcore_gpt_from_hf -from modelopt.torch.export.quant_format import IQ_FORMATS +from modelopt.torch.export.quant_format import GGML_FORMATS, IQ_FORMATS from modelopt.torch.export.unified_export_megatron import GPTModelExporter from modelopt.torch.quantization.config import QuantizerAttributeConfig from modelopt.torch.quantization.nn import TensorQuantizer @@ -90,22 +90,24 @@ def _verify_model_quant_config( assert quant_config_dict["kv_cache_quant_algo"] == KV_CACHE_FP8 -# Every IQ format the exporter accepts. Only the list of formats comes from the export -# tables; each test resolves what it expects from the codec module itself, so a wrong entry -# in IQ_FORMAT_REGISTRY cannot make both sides of an assertion agree. +# Every GGML format the exporter accepts. Only the list of formats comes from the export +# tables; each test resolves what it expects from the codec module itself, so a wrong registry +# entry cannot make both sides of an assertion agree. IQ_FORMAT_NAMES = sorted(IQ_FORMATS) +GGML_FORMAT_NAMES = sorted(GGML_FORMATS) -@pytest.mark.parametrize("qformat", IQ_FORMAT_NAMES) -def test_megatron_name_remapping_exports_iq_payload(qformat): - """Megatron export writes the same scale-free IQ representation as HF export.""" +@pytest.mark.parametrize("qformat", GGML_FORMAT_NAMES) +def test_megatron_name_remapping_exports_ggml_payload(qformat): + """Megatron export writes the same self-contained GGML representation as HF export.""" + block_size = getattr(ggml, f"{qformat.upper()}_BLOCK_SIZE") payload_bytes = getattr(ggml, f"{qformat.upper()}_BLOCK_BYTES") dequantize = getattr(ggml, f"dequantize_{qformat}") linear = torch.nn.Linear(256, 2, bias=False, dtype=torch.bfloat16) linear.weight_quantizer = TensorQuantizer( QuantizerAttributeConfig( num_bits=qformat, - block_sizes={-1: 256}, + block_sizes={-1: block_size}, backend="ggml", ) ) @@ -127,7 +129,7 @@ def test_megatron_name_remapping_exports_iq_payload(qformat): # decoded reference, not the fake-quant forward: that returns the straight-through form # a + (r - a), which in bf16 differs from r by up to one ULP of a -- enough to fail a # relative tolerance wherever r is small next to a, as IQ1_S's grid near zero often is. - logical_shape = torch.tensor([*packed.shape[:-2], packed.shape[-2] * 256]) + logical_shape = torch.tensor([*packed.shape[:-2], packed.shape[-2] * block_size]) reference, _ = getattr(ggml, f"quantize_{qformat}")(linear.weight) torch.testing.assert_close( dequantize(packed, logical_shape, dtype=torch.bfloat16), @@ -137,7 +139,7 @@ def test_megatron_name_remapping_exports_iq_payload(qformat): ) assert exporter.layer_config_dict == { "model.layers.0.mlp.down_proj.quantization": qformat, - "model.layers.0.mlp.down_proj.awq_block_size": 256, + "model.layers.0.mlp.down_proj.awq_block_size": block_size, } diff --git a/tests/unit/recipe/test_presets.py b/tests/unit/recipe/test_presets.py index bf9ca840b10..172cd09e7bf 100644 --- a/tests/unit/recipe/test_presets.py +++ b/tests/unit/recipe/test_presets.py @@ -40,6 +40,8 @@ IQ2_XS_EFFECTIVE_BITS, IQ2_XXS_BLOCK_SIZE, IQ2_XXS_EFFECTIVE_BITS, + Q8_0_BLOCK_SIZE, + Q8_0_EFFECTIVE_BITS, ) @@ -139,6 +141,7 @@ def test_mlp_weight_only_recipe_matches_its_mtq_cfg(recipe_name, cfg_name): ("iq1_s", IQ1_S_BLOCK_SIZE, IQ1_S_EFFECTIVE_BITS), ("iq2_xxs", IQ2_XXS_BLOCK_SIZE, IQ2_XXS_EFFECTIVE_BITS), ("iq2_xs", IQ2_XS_BLOCK_SIZE, IQ2_XS_EFFECTIVE_BITS), + ("q8_0", Q8_0_BLOCK_SIZE, Q8_0_EFFECTIVE_BITS), ], ) def test_iq_recipe_matches_packing_contract(qformat, block_size, effective_bits): diff --git a/tests/unit/torch/export/test_export_weight.py b/tests/unit/torch/export/test_export_weight.py index 94790846330..45d59d5475e 100644 --- a/tests/unit/torch/export/test_export_weight.py +++ b/tests/unit/torch/export/test_export_weight.py @@ -105,13 +105,16 @@ def test_export_per_block_quantized_weight(): assert not hasattr(model.linears[2], quantizer_attrs.output_scale) -@pytest.mark.parametrize(("num_bits", "payload_bytes"), [("iq1_s", 50), ("iq2_xs", 74)]) -def test_export_iq_payload_as_weight(num_bits, payload_bytes): +@pytest.mark.parametrize( + ("num_bits", "block_size", "payload_bytes"), + [("iq1_s", 256, 50), ("iq2_xs", 256, 74), ("q8_0", 32, 34)], +) +def test_export_iq_payload_as_weight(num_bits, block_size, payload_bytes): linear = nn.Linear(256, 4, bias=False, dtype=torch.bfloat16) linear.weight_quantizer = TensorQuantizer( QuantizerAttributeConfig( num_bits=num_bits, - block_sizes={-1: 256}, + block_sizes={-1: block_size}, backend="ggml", ) ) @@ -120,7 +123,7 @@ def test_export_iq_payload_as_weight(num_bits, payload_bytes): state_dict = postprocess_state_dict(linear.state_dict(), maxbound=448, quantization=None) assert isinstance(linear.weight, nn.Parameter) - assert state_dict["weight"].shape == (4, 1, payload_bytes) + assert state_dict["weight"].shape == (4, 256 // block_size, payload_bytes) assert state_dict["weight"].dtype == torch.uint8 assert "packed_weights" not in state_dict assert "weight_shape" not in state_dict diff --git a/tests/unit/torch/export/test_get_quantization.py b/tests/unit/torch/export/test_get_quantization.py index 8a8a7ed93f2..4d4d21a7a96 100644 --- a/tests/unit/torch/export/test_get_quantization.py +++ b/tests/unit/torch/export/test_get_quantization.py @@ -36,6 +36,7 @@ QUANTIZATION_IQ1_S, QUANTIZATION_IQ2_XS, QUANTIZATION_NVFP4, + QUANTIZATION_Q8_0, QUANTIZATION_W4A8_AWQ, ) from modelopt.torch.export.quant_utils import ( @@ -64,13 +65,16 @@ def __init__(self): @pytest.mark.parametrize( - ("num_bits", "quantization_format", "payload_bytes", "effective_bits"), + ("num_bits", "quantization_format", "block_size", "payload_bytes", "effective_bits"), [ - ("iq1_s", QUANTIZATION_IQ1_S, 50, 1.5625), - ("iq2_xs", QUANTIZATION_IQ2_XS, 74, 2.3125), + ("iq1_s", QUANTIZATION_IQ1_S, 256, 50, 1.5625), + ("iq2_xs", QUANTIZATION_IQ2_XS, 256, 74, 2.3125), + ("q8_0", QUANTIZATION_Q8_0, 32, 34, 8.5), ], ) -def test_iq_quantization_config(num_bits, quantization_format, payload_bytes, effective_bits): +def test_iq_quantization_config( + num_bits, quantization_format, block_size, payload_bytes, effective_bits +): model = torch.nn.Sequential(torch.nn.Linear(256, 256, bias=False)) mtq.quantize( model, @@ -81,7 +85,7 @@ def test_iq_quantization_config(num_bits, quantization_format, payload_bytes, ef "quantizer_name": "*weight_quantizer", "cfg": { "num_bits": num_bits, - "block_sizes": {-1: 256}, + "block_sizes": {-1: block_size}, "backend": "ggml", }, }, @@ -97,7 +101,7 @@ def test_iq_quantization_config(num_bits, quantization_format, payload_bytes, ef assert config["quantization"]["effective_bits"] == effective_bits hf_config = convert_hf_quant_config_format(config) assert "config_groups" not in hf_config - assert hf_config["group_size"] == 256 + assert hf_config["group_size"] == block_size assert hf_config["effective_bits"] == effective_bits assert hf_config["packing"] == "ggml" assert hf_config["block_payload_bytes"] == payload_bytes From f543e10c079429324b9e573a847014c4b5318a76 Mon Sep 17 00:00:00 2001 From: Hung-Yueh Chiang Date: Mon, 28 Sep 2026 09:30:44 -0700 Subject: [PATCH 2/6] [OMNIML-5899] Address Q8_0 review feedback Signed-off-by: Hung-Yueh Chiang --- modelopt/torch/export/quant_utils.py | 21 +++++++++++++------ modelopt_recipes/ptq.md | 2 +- .../torch/export/test_get_quantization.py | 13 ++++++++++++ 3 files changed, 29 insertions(+), 7 deletions(-) diff --git a/modelopt/torch/export/quant_utils.py b/modelopt/torch/export/quant_utils.py index bea78ee7546..d9beebb1684 100755 --- a/modelopt/torch/export/quant_utils.py +++ b/modelopt/torch/export/quant_utils.py @@ -49,7 +49,12 @@ ) from modelopt.torch.utils import clear_cuda_cache -from ..quantization.nn import NVFP4StaticQuantizer, SequentialQuantizer, TensorQuantizer +from ..quantization.nn import ( + GroupedQuantizer, + NVFP4StaticQuantizer, + SequentialQuantizer, + TensorQuantizer, +) from .model_utils import TiedWeightMap, get_language_model_from_vl from .quant_format import ( GGML_FORMATS, @@ -453,12 +458,16 @@ def uses_iq_quantization(module) -> bool: This reads ``num_bits`` directly rather than resolving each layer's full format, so an unrelated unsupported quantizer elsewhere in the model cannot turn the check into an error. - Known gap, shared with ``get_quantization_format``: ``weight_attr_names`` yields nothing for - a TEGroupedLinear, whose parameters are ``weight0..N`` while its quantizer is a single - ``GroupedQuantizer`` under ``weight_quantizer``. Neither function sees such a module, so an - experts-only GGML model reports no format at all -- not just here. Closing it belongs in - ``weight_attr_names``, where it affects every format, rather than in this helper. + A TEGroupedLinear stores its per-expert quantizers in one ``GroupedQuantizer`` rather than + beside its ``weight0..N`` parameters, so inspect that container directly. """ + grouped_quantizer = getattr(module, "weight_quantizer", None) + if isinstance(grouped_quantizer, GroupedQuantizer) and any( + quantizer.is_enabled and getattr(quantizer, "num_bits", None) in GGML_FORMATS + for quantizer in grouped_quantizer + ): + return True + for weight_name in weight_attr_names(module): weight_quantizer = representative_weight_quantizer(module, weight_name) # getattr: a SequentialQuantizer has is_enabled but no num_bits, and is never GGML -- diff --git a/modelopt_recipes/ptq.md b/modelopt_recipes/ptq.md index 5ac560684a5..c9926169934 100644 --- a/modelopt_recipes/ptq.md +++ b/modelopt_recipes/ptq.md @@ -28,7 +28,7 @@ supported combinations. ### The shipped recipes
-All 29 general/ptq/ recipes (click to expand) +All 30 general/ptq/ recipes (click to expand) | Recipe | Model body | KV cache | Calibration | |--------|-----------|----------|-------------| diff --git a/tests/unit/torch/export/test_get_quantization.py b/tests/unit/torch/export/test_get_quantization.py index 4d4d21a7a96..bf62636b995 100644 --- a/tests/unit/torch/export/test_get_quantization.py +++ b/tests/unit/torch/export/test_get_quantization.py @@ -51,6 +51,7 @@ uses_iq_quantization, ) from modelopt.torch.quantization.nn import ( + GroupedQuantizer, NVFP4StaticQuantizer, SequentialQuantizer, TensorQuantizer, @@ -159,6 +160,18 @@ def test_uses_iq_quantization_tolerates_sequential_quantizer(): assert not uses_iq_quantization(torch.nn.Sequential(layer)) +def test_uses_iq_quantization_sees_grouped_q8_0_quantizer(): + module = torch.nn.Module() + module.weight0 = torch.nn.Parameter(torch.empty(256, 256)) + quantizer = TensorQuantizer() + quantizer.set_from_attribute_config( + {"num_bits": "q8_0", "block_sizes": {-1: 32}, "backend": "ggml"} + ) + module.weight_quantizer = GroupedQuantizer(quantizer) + + assert uses_iq_quantization(module) + + def test_iq_export_rejects_enabled_input_quantizer(): """IQ payloads carry no activation scale, so W-IQ + A-FP8 must not export as weight-only.""" model = _quantize_sequential( From 0c50fa2ccb896b670ce83747947b221a19448666 Mon Sep 17 00:00:00 2001 From: Hung-Yueh Chiang Date: Mon, 28 Sep 2026 09:53:29 -0700 Subject: [PATCH 3/6] [OMNIML-5899] Fix Q8_0 export review findings Signed-off-by: Hung-Yueh Chiang --- modelopt/torch/export/quant_format.py | 4 ++-- modelopt/torch/quantization/utils/core_utils.py | 13 ++++++++++++- .../torch/export/test_unified_export_megatron.py | 2 +- tests/unit/torch/export/test_get_quantization.py | 3 ++- 4 files changed, 17 insertions(+), 5 deletions(-) diff --git a/modelopt/torch/export/quant_format.py b/modelopt/torch/export/quant_format.py index dd47c59b3fa..a5b48753978 100644 --- a/modelopt/torch/export/quant_format.py +++ b/modelopt/torch/export/quant_format.py @@ -45,8 +45,8 @@ # Every GGML format is derived from the registry the quantization backend dispatches through, so # export and dispatch cannot disagree about which formats exist. A format's block geometry and -# packer are read from GGML_FORMAT_REGISTRY directly. IQ_FORMATS remains the vector-codebook -# subset for callers that specifically need it. +# packer are read from GGML_FORMAT_REGISTRY directly. IQ_FORMATS remains as the compatibility +# subset used by IQ-specific conformance tests. # # Registering a format therefore declares it exportable, and that is intended rather than a side # effect: fake quant is dequantize(quantize(w)), so a format cannot be dispatched without the diff --git a/modelopt/torch/quantization/utils/core_utils.py b/modelopt/torch/quantization/utils/core_utils.py index eca3302a425..9b074430f84 100644 --- a/modelopt/torch/quantization/utils/core_utils.py +++ b/modelopt/torch/quantization/utils/core_utils.py @@ -247,9 +247,11 @@ def representative_weight_quantizer(module: nn.Module, weight_name: str = "weigh def weight_attr_names(module: nn.Module) -> "Generator[str, None, None]": """Get the weight param attribute names in a converted module, non-recursive. - Covers three layouts: + Covers four layouts: - standard ``nn.Linear``: ``weight`` + ``weight_quantizer``. + - ``TEGroupedLinear``: ``weight0..N`` + one ``GroupedQuantizer`` exposed through the + logical ``weight`` name. - custom per-weight quantizer (e.g. ``Llama4TextExperts`` with ``gate_up_proj`` + ``gate_up_proj_weight_quantizer``). - fused-experts ``nn.ModuleList`` quantizers (``_QuantFusedExperts`` with @@ -259,6 +261,15 @@ def weight_attr_names(module: nn.Module) -> "Generator[str, None, None]": if getattr(module, "weight", None) is not None: if representative_weight_quantizer(module, "weight") is not None: yield "weight" + elif getattr(module, "weight0", None) is not None: + # TEGroupedLinear has no physical ``weight`` parameter after setup, but its + # GroupedQuantizer is deliberately stored as ``weight_quantizer``. Yield the logical + # name so format and block-size discovery see the same representative quantizer used by + # its export path. + from ..nn import GroupedQuantizer + + if isinstance(getattr(module, "weight_quantizer", None), GroupedQuantizer): + yield "weight" # per-parameter custom attr names for name, _ in module.named_parameters(recurse=False): diff --git a/tests/gpu_megatron/torch/export/test_unified_export_megatron.py b/tests/gpu_megatron/torch/export/test_unified_export_megatron.py index 906e4e728d6..fb70b4a43e2 100644 --- a/tests/gpu_megatron/torch/export/test_unified_export_megatron.py +++ b/tests/gpu_megatron/torch/export/test_unified_export_megatron.py @@ -121,7 +121,7 @@ def test_megatron_name_remapping_exports_ggml_payload(qformat): packed_key = "model.layers.0.mlp.down_proj.weight" packed = exporter._state_dict[packed_key] - assert packed.shape == (2, 1, payload_bytes) + assert packed.shape == (2, linear.in_features // block_size, payload_bytes) assert packed.dtype == torch.uint8 # Exact bytes against the format's own packer, as the slicing tests below check. _assert_iq_payload_matches(qformat, packed, linear.weight) diff --git a/tests/unit/torch/export/test_get_quantization.py b/tests/unit/torch/export/test_get_quantization.py index bf62636b995..860c75c62b2 100644 --- a/tests/unit/torch/export/test_get_quantization.py +++ b/tests/unit/torch/export/test_get_quantization.py @@ -160,7 +160,7 @@ def test_uses_iq_quantization_tolerates_sequential_quantizer(): assert not uses_iq_quantization(torch.nn.Sequential(layer)) -def test_uses_iq_quantization_sees_grouped_q8_0_quantizer(): +def test_grouped_q8_0_quantizer_is_detected_for_export(): module = torch.nn.Module() module.weight0 = torch.nn.Parameter(torch.empty(256, 256)) quantizer = TensorQuantizer() @@ -170,6 +170,7 @@ def test_uses_iq_quantization_sees_grouped_q8_0_quantizer(): module.weight_quantizer = GroupedQuantizer(quantizer) assert uses_iq_quantization(module) + assert get_quantization_format(module) == QUANTIZATION_Q8_0 def test_iq_export_rejects_enabled_input_quantizer(): From 02f8a171648ff47155ae61457d5cd43bbbdf2a33 Mon Sep 17 00:00:00 2001 From: Hung-Yueh Chiang Date: Fri, 2 Oct 2026 14:56:51 -0700 Subject: [PATCH 4/6] Align Megatron GGML rejection assertions Signed-off-by: Hung-Yueh Chiang --- .../gpu_megatron/torch/export/test_unified_export_megatron.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/tests/gpu_megatron/torch/export/test_unified_export_megatron.py b/tests/gpu_megatron/torch/export/test_unified_export_megatron.py index 56af618f85f..31020d6f305 100644 --- a/tests/gpu_megatron/torch/export/test_unified_export_megatron.py +++ b/tests/gpu_megatron/torch/export/test_unified_export_megatron.py @@ -295,7 +295,7 @@ def test_megatron_packed_experts_reject_iq_without_deployment_loader(qformat): experts = _make_iq_experts(qformat, "linear_fc2") exporter = _make_iq_exporter() - with pytest.raises(NotImplementedError, match="Fused-MoE IQ export requires"): + with pytest.raises(NotImplementedError, match="Fused-MoE GGML export requires"): exporter._pack_name_remapping( experts, "model.layers.0.mlp.experts.down_proj", @@ -309,7 +309,7 @@ def test_megatron_gpt_oss_packed_experts_reject_iq_without_deployment_loader(qfo experts = _make_iq_experts(qformat, "linear_fc1", bias=True) exporter = _make_iq_exporter() - with pytest.raises(NotImplementedError, match="Fused-MoE IQ export requires"): + with pytest.raises(NotImplementedError, match="Fused-MoE GGML export requires"): exporter._pack_name_remapping_gpt_oss( experts, "model.layers.0.mlp.experts.gate_up_proj", From 81b99f9092994d3366258495ee14fcbf1ce925c4 Mon Sep 17 00:00:00 2001 From: Hung-Yueh Chiang Date: Mon, 5 Oct 2026 10:07:02 -0700 Subject: [PATCH 5/6] Use GGMLFormat for every GGML codec Signed-off-by: Hung-Yueh Chiang --- modelopt/torch/quantization/ggml/__init__.py | 3 +-- modelopt/torch/quantization/ggml/common.py | 5 ----- modelopt/torch/quantization/ggml/iq1_m.py | 4 ++-- modelopt/torch/quantization/ggml/iq1_s.py | 4 ++-- modelopt/torch/quantization/ggml/iq2_s.py | 4 ++-- modelopt/torch/quantization/ggml/iq2_xs.py | 4 ++-- modelopt/torch/quantization/ggml/iq2_xxs.py | 4 ++-- modelopt/torch/quantization/ggml/registry.py | 4 ++-- tests/unit/torch/quantization/test_ggml_backend.py | 2 +- 9 files changed, 14 insertions(+), 20 deletions(-) diff --git a/modelopt/torch/quantization/ggml/__init__.py b/modelopt/torch/quantization/ggml/__init__.py index 5950fd7b087..8dbd379dbf2 100644 --- a/modelopt/torch/quantization/ggml/__init__.py +++ b/modelopt/torch/quantization/ggml/__init__.py @@ -29,7 +29,7 @@ from .iq2_xxs import __all__ as _iq2_xxs_all from .q8_0 import * from .q8_0 import __all__ as _q8_0_all -from .registry import GGML_FORMAT_REGISTRY, IQ_FORMAT_REGISTRY, GGMLFormat, IQFormat +from .registry import GGML_FORMAT_REGISTRY, IQ_FORMAT_REGISTRY, GGMLFormat __all__ = [ # noqa: PLE0604 *_iq1_m_all, @@ -41,5 +41,4 @@ "GGML_FORMAT_REGISTRY", "GGMLFormat", "IQ_FORMAT_REGISTRY", - "IQFormat", ] diff --git a/modelopt/torch/quantization/ggml/common.py b/modelopt/torch/quantization/ggml/common.py index 96a490716e0..ed24023cc6f 100644 --- a/modelopt/torch/quantization/ggml/common.py +++ b/modelopt/torch/quantization/ggml/common.py @@ -243,11 +243,6 @@ def pack(self, weight: torch.Tensor, quantizer=None) -> torch.Tensor: ) -# Compatibility alias for callers that imported the record type before the registry was -# generalized from the IQ family to every supported GGML block format. -IQFormat = GGMLFormat - - def narrow_to_float32(blocks: torch.Tensor) -> torch.Tensor: """Narrow ``blocks`` to float32 the way the CUDA ``load_float`` helper does. diff --git a/modelopt/torch/quantization/ggml/iq1_m.py b/modelopt/torch/quantization/ggml/iq1_m.py index 0799a7aaeba..68d34cef4f0 100644 --- a/modelopt/torch/quantization/ggml/iq1_m.py +++ b/modelopt/torch/quantization/ggml/iq1_m.py @@ -42,7 +42,7 @@ from ..extensions import get_cuda_ext_ggml from .common import ( GGML_BLOCK_SIZE, - IQFormat, + GGMLFormat, narrow_to_float32, validate_block_chunk_size, validate_packed_weights, @@ -238,7 +238,7 @@ def dequantize_iq1_m( return decoded.reshape(shape) -IQ1_M_FORMAT = IQFormat( +IQ1_M_FORMAT = GGMLFormat( name="iq1_m", block_size=IQ1_M_BLOCK_SIZE, block_bytes=IQ1_M_BLOCK_BYTES, diff --git a/modelopt/torch/quantization/ggml/iq1_s.py b/modelopt/torch/quantization/ggml/iq1_s.py index 83aff654b8a..b98587d44a8 100644 --- a/modelopt/torch/quantization/ggml/iq1_s.py +++ b/modelopt/torch/quantization/ggml/iq1_s.py @@ -38,7 +38,7 @@ from .codebooks import iq1_s_grid_bytes from .common import ( GGML_BLOCK_SIZE, - IQFormat, + GGMLFormat, narrow_to_float32, validate_block_chunk_size, validate_packed_weights, @@ -262,7 +262,7 @@ def dequantize_iq1_s( return decoded.reshape(shape) -IQ1_S_FORMAT = IQFormat( +IQ1_S_FORMAT = GGMLFormat( name="iq1_s", block_size=IQ1_S_BLOCK_SIZE, block_bytes=IQ1_S_BLOCK_BYTES, diff --git a/modelopt/torch/quantization/ggml/iq2_s.py b/modelopt/torch/quantization/ggml/iq2_s.py index 9667301acf5..f28952d14e7 100644 --- a/modelopt/torch/quantization/ggml/iq2_s.py +++ b/modelopt/torch/quantization/ggml/iq2_s.py @@ -41,7 +41,7 @@ from .codebooks import iq2_s_grid_bytes from .common import ( GGML_BLOCK_SIZE, - IQFormat, + GGMLFormat, narrow_to_float32, validate_block_chunk_size, validate_packed_weights, @@ -236,7 +236,7 @@ def dequantize_iq2_s( return decoded.reshape(shape) -IQ2_S_FORMAT = IQFormat( +IQ2_S_FORMAT = GGMLFormat( name="iq2_s", block_size=IQ2_S_BLOCK_SIZE, block_bytes=IQ2_S_BLOCK_BYTES, diff --git a/modelopt/torch/quantization/ggml/iq2_xs.py b/modelopt/torch/quantization/ggml/iq2_xs.py index 9045d95de91..ac20c5cfee2 100644 --- a/modelopt/torch/quantization/ggml/iq2_xs.py +++ b/modelopt/torch/quantization/ggml/iq2_xs.py @@ -36,7 +36,7 @@ from .codebooks import iq2_xs_grid_bytes from .common import ( GGML_BLOCK_SIZE, - IQFormat, + GGMLFormat, narrow_to_float32, validate_block_chunk_size, validate_packed_weights, @@ -260,7 +260,7 @@ def dequantize_iq2_xs( return decoded.reshape(shape) -IQ2_XS_FORMAT = IQFormat( +IQ2_XS_FORMAT = GGMLFormat( name="iq2_xs", block_size=IQ2_XS_BLOCK_SIZE, block_bytes=IQ2_XS_BLOCK_BYTES, diff --git a/modelopt/torch/quantization/ggml/iq2_xxs.py b/modelopt/torch/quantization/ggml/iq2_xxs.py index 3ac4d8a079b..3a6b4bf1432 100644 --- a/modelopt/torch/quantization/ggml/iq2_xxs.py +++ b/modelopt/torch/quantization/ggml/iq2_xxs.py @@ -41,7 +41,7 @@ from .codebooks import iq2_xxs_grid_bytes from .common import ( GGML_BLOCK_SIZE, - IQFormat, + GGMLFormat, narrow_to_float32, validate_block_chunk_size, validate_packed_weights, @@ -270,7 +270,7 @@ def dequantize_iq2_xxs( return decoded.reshape(shape) -IQ2_XXS_FORMAT = IQFormat( +IQ2_XXS_FORMAT = GGMLFormat( name="iq2_xxs", block_size=IQ2_XXS_BLOCK_SIZE, block_bytes=IQ2_XXS_BLOCK_BYTES, diff --git a/modelopt/torch/quantization/ggml/registry.py b/modelopt/torch/quantization/ggml/registry.py index c95e9569fe7..33d26a8abd9 100644 --- a/modelopt/torch/quantization/ggml/registry.py +++ b/modelopt/torch/quantization/ggml/registry.py @@ -15,7 +15,7 @@ """GGML block formats, listed once for backend dispatch and export.""" -from .common import GGMLFormat, IQFormat +from .common import GGMLFormat from .iq1_m import IQ1_M_FORMAT from .iq1_s import IQ1_S_FORMAT from .iq2_s import IQ2_S_FORMAT @@ -23,7 +23,7 @@ from .iq2_xxs import IQ2_XXS_FORMAT from .q8_0 import Q8_0_FORMAT -__all__ = ["GGML_FORMAT_REGISTRY", "IQ_FORMAT_REGISTRY", "GGMLFormat", "IQFormat"] +__all__ = ["GGML_FORMAT_REGISTRY", "IQ_FORMAT_REGISTRY", "GGMLFormat"] # Every GGML block format, keyed by the name a quantizer's num_bits carries, in increasing bits # per weight. Backend dispatch reads this mapping, and export will derive its supported formats diff --git a/tests/unit/torch/quantization/test_ggml_backend.py b/tests/unit/torch/quantization/test_ggml_backend.py index 29c122c7107..136ce92f6ed 100644 --- a/tests/unit/torch/quantization/test_ggml_backend.py +++ b/tests/unit/torch/quantization/test_ggml_backend.py @@ -237,7 +237,6 @@ def test_registry_lists_every_exported_encoder(): def test_iq_registry_remains_a_compatible_subset_of_ggml_registry(): - assert ggml.IQFormat is ggml.GGMLFormat assert set(ggml.IQ_FORMAT_REGISTRY) == { name for name in GGML_FORMAT_REGISTRY if name.startswith("iq") } @@ -252,6 +251,7 @@ def test_registry_record_is_wired_to_its_own_codec(num_bits, module): record = GGML_FORMAT_REGISTRY[num_bits] upper = num_bits.upper() + assert isinstance(record, ggml.GGMLFormat) assert record.name == num_bits assert record.quantize is getattr(module, f"quantize_{num_bits}") assert record.dequantize is getattr(module, f"dequantize_{num_bits}") From 8bc6d155b652f72d4738683e3097559f7a7eb000 Mon Sep 17 00:00:00 2001 From: Hung-Yueh Chiang Date: Mon, 5 Oct 2026 11:34:36 -0700 Subject: [PATCH 6/6] Cover all GGML formats in Megatron export tests Signed-off-by: Hung-Yueh Chiang --- .../export/test_unified_export_megatron.py | 108 +++++++++--------- 1 file changed, 57 insertions(+), 51 deletions(-) diff --git a/tests/gpu_megatron/torch/export/test_unified_export_megatron.py b/tests/gpu_megatron/torch/export/test_unified_export_megatron.py index 5fa659e35e5..286700a9338 100644 --- a/tests/gpu_megatron/torch/export/test_unified_export_megatron.py +++ b/tests/gpu_megatron/torch/export/test_unified_export_megatron.py @@ -48,7 +48,7 @@ import modelopt.torch.speculative as mtsp from modelopt.torch.export import KV_CACHE_FP8, export_mcore_gpt_to_hf, import_mcore_gpt_from_hf from modelopt.torch.export.plugins.mcore_common import all_mcore_hf_export_mapping -from modelopt.torch.export.quant_format import GGML_FORMATS, IQ_FORMATS +from modelopt.torch.export.quant_format import GGML_FORMATS from modelopt.torch.export.unified_export_megatron import GPTModelExporter from modelopt.torch.quantization.config import QuantizerAttributeConfig from modelopt.torch.quantization.nn import TensorQuantizer @@ -95,7 +95,6 @@ def _verify_model_quant_config( # Every GGML format the exporter accepts. Only the list of formats comes from the export # tables; each test resolves what it expects from the codec module itself, so a wrong registry # entry cannot make both sides of an assertion agree. -IQ_FORMAT_NAMES = sorted(IQ_FORMATS) GGML_FORMAT_NAMES = sorted(GGML_FORMATS) @@ -126,7 +125,7 @@ def test_megatron_name_remapping_exports_ggml_payload(qformat): assert packed.shape == (2, linear.in_features // block_size, payload_bytes) assert packed.dtype == torch.uint8 # Exact bytes against the format's own packer, as the slicing tests below check. - _assert_iq_payload_matches(qformat, packed, linear.weight) + _assert_ggml_payload_matches(qformat, packed, linear.weight) # And the payload decodes to exactly what the fake quantizer reconstructs. Compare with the # decoded reference, not the fake-quant forward: that returns the straight-through form # a + (r - a), which in bf16 differs from r by up to one ULP of a -- enough to fail a @@ -145,7 +144,8 @@ def test_megatron_name_remapping_exports_ggml_payload(qformat): } -def _make_iq_experts(qformat, layer_type, *, bias=False): +def _make_ggml_experts(qformat, layer_type, *, bias=False): + block_size = getattr(ggml, f"{qformat.upper()}_BLOCK_SIZE") experts = torch.nn.ModuleList() generator = torch.Generator().manual_seed(1234) for _ in range(2): @@ -158,7 +158,7 @@ def _make_iq_experts(qformat, layer_type, *, bias=False): linear.weight_quantizer = TensorQuantizer( QuantizerAttributeConfig( num_bits=qformat, - block_sizes={-1: 256}, + block_sizes={-1: block_size}, backend="ggml", ) ) @@ -167,7 +167,7 @@ def _make_iq_experts(qformat, layer_type, *, bias=False): return experts -def _make_iq_exporter(): +def _make_ggml_exporter(): exporter = object.__new__(GPTModelExporter) exporter.dtype = torch.bfloat16 exporter._state_dict = {} @@ -176,46 +176,48 @@ def _make_iq_exporter(): return exporter -def _make_iq_weight(rows): +def _make_ggml_weight(rows): return torch.linspace(-1, 1, rows * 256, dtype=torch.float32).reshape(rows, 256).bfloat16() -def _assert_iq_payload_matches(qformat, packed, logical_weight): +def _assert_ggml_payload_matches(qformat, packed, logical_weight): expected, _ = getattr(ggml, f"quantize_{qformat}")(logical_weight) torch.testing.assert_close(packed, expected.cpu(), rtol=0, atol=0) -@pytest.mark.parametrize("qformat", IQ_FORMAT_NAMES) -def test_megatron_gated_mlp_slicing_exports_iq_payloads(qformat): - weight = _make_iq_weight(8) +@pytest.mark.parametrize("qformat", GGML_FORMAT_NAMES) +def test_megatron_gated_mlp_slicing_exports_ggml_payloads(qformat): + block_size = getattr(ggml, f"{qformat.upper()}_BLOCK_SIZE") + weight = _make_ggml_weight(8) module = SimpleNamespace(config=SimpleNamespace(ffn_hidden_size=4)) - exporter = _make_iq_exporter() - exporter._get_quantized_state = lambda *a, **k: ({"weight": weight}, qformat, 256) + exporter = _make_ggml_exporter() + exporter._get_quantized_state = lambda *a, **k: ({"weight": weight}, qformat, block_size) exporter._gated_mlp_slicing(module, "model.layers.0.mlp.") - _assert_iq_payload_matches( + _assert_ggml_payload_matches( qformat, exporter._state_dict["model.layers.0.mlp.gate_proj.weight"], weight[:4] ) - _assert_iq_payload_matches( + _assert_ggml_payload_matches( qformat, exporter._state_dict["model.layers.0.mlp.up_proj.weight"], weight[4:] ) -@pytest.mark.parametrize("qformat", IQ_FORMAT_NAMES) -def test_megatron_grouped_mlp_slicing_exports_iq_payloads(qformat): - weight = _make_iq_weight(8) +@pytest.mark.parametrize("qformat", GGML_FORMAT_NAMES) +def test_megatron_grouped_mlp_slicing_exports_ggml_payloads(qformat): + block_size = getattr(ggml, f"{qformat.upper()}_BLOCK_SIZE") + weight = _make_ggml_weight(8) module = SimpleNamespace( num_gemms=1, weight0=weight, local_expert_indices=[0], state_dict=lambda: {"weight0": weight}, ) - exporter = _make_iq_exporter() + exporter = _make_ggml_exporter() exporter._get_quantized_state = lambda *a, **k: ( {"weight": module.weight}, qformat, - 256, + block_size, ) exporter._grouped_mlp_slicing( @@ -225,17 +227,18 @@ def test_megatron_grouped_mlp_slicing_exports_iq_payloads(qformat): up_proj_name="up_proj", ) - _assert_iq_payload_matches( + _assert_ggml_payload_matches( qformat, exporter._state_dict["model.layers.0.mlp.experts.0.gate_proj.weight"], weight[:4] ) - _assert_iq_payload_matches( + _assert_ggml_payload_matches( qformat, exporter._state_dict["model.layers.0.mlp.experts.0.up_proj.weight"], weight[4:] ) -@pytest.mark.parametrize("qformat", IQ_FORMAT_NAMES) -def test_megatron_qkv_slicing_exports_iq_payloads(qformat): - weight = _make_iq_weight(8) +@pytest.mark.parametrize("qformat", GGML_FORMAT_NAMES) +def test_megatron_qkv_slicing_exports_ggml_payloads(qformat): + block_size = getattr(ggml, f"{qformat.upper()}_BLOCK_SIZE") + weight = _make_ggml_weight(8) module = SimpleNamespace( config=SimpleNamespace( hidden_size=256, @@ -245,8 +248,8 @@ def test_megatron_qkv_slicing_exports_iq_payloads(qformat): attention_output_gate=False, ) ) - exporter = _make_iq_exporter() - exporter._get_quantized_state = lambda *a, **k: ({"weight": weight}, qformat, 256) + exporter = _make_ggml_exporter() + exporter._get_quantized_state = lambda *a, **k: ({"weight": weight}, qformat, block_size) exporter._qkv_slicing(module, "model.layers.0.self_attn.") @@ -257,30 +260,31 @@ def test_megatron_qkv_slicing_exports_iq_payloads(qformat): "v_proj": reshaped[3].reshape(2, 256), } for projection, logical_weight in expected.items(): - _assert_iq_payload_matches( + _assert_ggml_payload_matches( qformat, exporter._state_dict[f"model.layers.0.self_attn.{projection}.weight"], logical_weight, ) -@pytest.mark.parametrize("qformat", IQ_FORMAT_NAMES) -def test_megatron_gated_delta_net_slicing_exports_iq_payloads(qformat): - weight = _make_iq_weight(12) +@pytest.mark.parametrize("qformat", GGML_FORMAT_NAMES) +def test_megatron_gated_delta_net_slicing_exports_ggml_payloads(qformat): + block_size = getattr(ggml, f"{qformat.upper()}_BLOCK_SIZE") + weight = _make_ggml_weight(12) module = SimpleNamespace( in_proj=object(), in_proj_split_names=("query", "key", "value", "z", "beta", "alpha"), in_proj_split_sections=(2, 2, 2, 2, 2, 2), ) - exporter = _make_iq_exporter() - exporter._get_quantized_state = lambda *a, **k: ({"weight": weight}, qformat, 256) + exporter = _make_ggml_exporter() + exporter._get_quantized_state = lambda *a, **k: ({"weight": weight}, qformat, block_size) exporter._gated_delta_net_slicing(module, "model.layers.0.mixer.") - _assert_iq_payload_matches( + _assert_ggml_payload_matches( qformat, exporter._state_dict["model.layers.0.mixer.in_proj_qkv.weight"], weight[:6] ) - _assert_iq_payload_matches( + _assert_ggml_payload_matches( qformat, exporter._state_dict["model.layers.0.mixer.in_proj_z.weight"], weight[6:8] ) torch.testing.assert_close( @@ -291,10 +295,10 @@ def test_megatron_gated_delta_net_slicing_exports_iq_payloads(qformat): ) -@pytest.mark.parametrize("qformat", IQ_FORMAT_NAMES) -def test_megatron_packed_experts_reject_iq_without_deployment_loader(qformat): - experts = _make_iq_experts(qformat, "linear_fc2") - exporter = _make_iq_exporter() +@pytest.mark.parametrize("qformat", GGML_FORMAT_NAMES) +def test_megatron_packed_experts_reject_ggml_without_deployment_loader(qformat): + experts = _make_ggml_experts(qformat, "linear_fc2") + exporter = _make_ggml_exporter() with pytest.raises(NotImplementedError, match="Fused-MoE GGML export requires"): exporter._pack_name_remapping( @@ -305,10 +309,10 @@ def test_megatron_packed_experts_reject_iq_without_deployment_loader(qformat): assert exporter._state_dict == {} -@pytest.mark.parametrize("qformat", IQ_FORMAT_NAMES) -def test_megatron_gpt_oss_packed_experts_reject_iq_without_deployment_loader(qformat): - experts = _make_iq_experts(qformat, "linear_fc1", bias=True) - exporter = _make_iq_exporter() +@pytest.mark.parametrize("qformat", GGML_FORMAT_NAMES) +def test_megatron_gpt_oss_packed_experts_reject_ggml_without_deployment_loader(qformat): + experts = _make_ggml_experts(qformat, "linear_fc1", bias=True) + exporter = _make_ggml_exporter() with pytest.raises(NotImplementedError, match="Fused-MoE GGML export requires"): exporter._pack_name_remapping_gpt_oss( @@ -319,14 +323,15 @@ def test_megatron_gpt_oss_packed_experts_reject_iq_without_deployment_loader(qfo assert exporter._state_dict == {} -@pytest.mark.parametrize("qformat", IQ_FORMAT_NAMES) -def test_megatron_iq_export_rejects_tensor_parallelism(qformat): - """IQ packing is intentionally limited to complete TP=1 weights.""" +@pytest.mark.parametrize("qformat", GGML_FORMAT_NAMES) +def test_megatron_ggml_export_rejects_tensor_parallelism(qformat): + """GGML packing is intentionally limited to complete TP=1 weights.""" + block_size = getattr(ggml, f"{qformat.upper()}_BLOCK_SIZE") linear = torch.nn.Linear(256, 2, bias=False, dtype=torch.bfloat16) linear.weight_quantizer = TensorQuantizer( QuantizerAttributeConfig( num_bits=qformat, - block_sizes={-1: 256}, + block_sizes={-1: block_size}, backend="ggml", ) ) @@ -389,19 +394,20 @@ def test_mla_export_keeps_hf_head_dim(dist_workers_size_1, tmp_path): dist_workers_size_1.run(partial(_test_mla_export_keeps_hf_head_dim, model_dir)) -@pytest.mark.parametrize("qformat", IQ_FORMAT_NAMES) -def test_megatron_iq_export_rejects_pipeline_parallelism(qformat): - """IQ packing requires PP=1 so the fused-MoE rejection reaches every rank. +@pytest.mark.parametrize("qformat", GGML_FORMAT_NAMES) +def test_megatron_ggml_export_rejects_pipeline_parallelism(qformat): + """GGML packing requires PP=1 so the fused-MoE rejection reaches every rank. The rejection raises from inside the per-expert loops, so a stage owning no expert would skip it and block in save_pretrained's collectives while its peers exit. PP=1 removes the divergence rather than trying to detect it. """ + block_size = getattr(ggml, f"{qformat.upper()}_BLOCK_SIZE") linear = torch.nn.Linear(256, 2, bias=False, dtype=torch.bfloat16) linear.weight_quantizer = TensorQuantizer( QuantizerAttributeConfig( num_bits=qformat, - block_sizes={-1: 256}, + block_sizes={-1: block_size}, backend="ggml", ) )