From 74c4858eb44def110d483a27611808a85d4e1866 Mon Sep 17 00:00:00 2001 From: Sylvester Kaczmarek <16242628+sylvesterkaczmarek@users.noreply.github.com> Date: Sun, 4 Oct 2026 21:28:00 +0100 Subject: [PATCH] Preserve histogram counts during entropy calibration Signed-off-by: Sylvester Kaczmarek <16242628+sylvesterkaczmarek@users.noreply.github.com> --- CHANGELOG.rst | 2 ++ .../torch/quantization/calib/histogram.py | 2 +- .../torch/quantization/test_calibrator.py | 26 +++++++++++++++++++ 3 files changed, 29 insertions(+), 1 deletion(-) diff --git a/CHANGELOG.rst b/CHANGELOG.rst index adcaec96729..6d8b47f6d2b 100755 --- a/CHANGELOG.rst +++ b/CHANGELOG.rst @@ -101,6 +101,8 @@ Changelog **Bug Fixes** +- Preserve accumulated histogram counts when computing an entropy calibration threshold with NumPy histograms. + - Fix Megatron unified HF export of MoE models with grouped-GEMM experts when only the experts are quantized (e.g. ``nvfp4_experts_only-*`` recipes): ``hf_quant_config.json`` and the ``quantization_config`` in ``config.json`` were not written, so the quantized experts were served as unquantized weights. Re-export such checkpoints. - Fix Megatron-Core checkpoint saving for quantized grouped MoE experts when tensor and expert parallelism are both enabled. - Fix unified HuggingFace export of RADIO-based VLMs retaining post-conversion vision and diff --git a/modelopt/torch/quantization/calib/histogram.py b/modelopt/torch/quantization/calib/histogram.py index e27a5471e30..73ebc327892 100644 --- a/modelopt/torch/quantization/calib/histogram.py +++ b/modelopt/torch/quantization/calib/histogram.py @@ -218,7 +218,7 @@ def _normalize_distr(distr): if summ != 0: distr = distr / summ - bins = calib_hist[:] + bins = calib_hist.copy() bins[0] = bins[1] total_data = np.sum(bins) diff --git a/tests/unit/torch/quantization/test_calibrator.py b/tests/unit/torch/quantization/test_calibrator.py index 9f7d77ce6f7..2d489b14543 100644 --- a/tests/unit/torch/quantization/test_calibrator.py +++ b/tests/unit/torch/quantization/test_calibrator.py @@ -89,6 +89,32 @@ def test_track_amax_raises(self): class TestHistogramCalibrator: + @pytest.mark.parametrize("torch_hist", [False, True]) + def test_entropy_preserves_collected_histogram(self, torch_hist): + calibrator = calib.HistogramCalibrator(4, None, False, num_bins=32, torch_hist=torch_hist) + reference = calib.HistogramCalibrator(4, None, False, num_bins=32, torch_hist=torch_hist) + first_batch = torch.cat((torch.zeros(100), torch.arange(1, 33).float())) + calibrator.collect(first_batch) + reference.collect(first_batch) + expected_hist = np.asarray(calibrator._calib_hist).copy() + expected_edges = np.asarray(calibrator._calib_bin_edges).copy() + percentile = calibrator.compute_amax("percentile", percentile=50) + + first_entropy = calibrator.compute_amax("entropy", start_bin=16) + second_entropy = calibrator.compute_amax("entropy", start_bin=16) + + np.testing.assert_array_equal(calibrator._calib_hist, expected_hist) + np.testing.assert_array_equal(calibrator._calib_bin_edges, expected_edges) + assert calibrator.compute_amax("percentile", percentile=50) == percentile + assert first_entropy == second_entropy + next_batch = torch.tensor([0.0, 2.0, 4.0, 64.0]) + calibrator.collect(next_batch) + reference.collect(next_batch) + np.testing.assert_array_equal(calibrator._calib_hist, reference._calib_hist) + assert calibrator.compute_amax("percentile", percentile=50) == reference.compute_amax( + "percentile", percentile=50 + ) + @pytest.mark.skip(reason="TODO: Fix assertions in test_grow") def test_grow(self, verbose): x_1 = torch.tensor([0, 255, 255, 255, 255, 255])