diff --git a/CHANGELOG.rst b/CHANGELOG.rst index adcaec96729..6d8b47f6d2b 100755 --- a/CHANGELOG.rst +++ b/CHANGELOG.rst @@ -101,6 +101,8 @@ Changelog **Bug Fixes** +- Preserve accumulated histogram counts when computing an entropy calibration threshold with NumPy histograms. + - Fix Megatron unified HF export of MoE models with grouped-GEMM experts when only the experts are quantized (e.g. ``nvfp4_experts_only-*`` recipes): ``hf_quant_config.json`` and the ``quantization_config`` in ``config.json`` were not written, so the quantized experts were served as unquantized weights. Re-export such checkpoints. - Fix Megatron-Core checkpoint saving for quantized grouped MoE experts when tensor and expert parallelism are both enabled. - Fix unified HuggingFace export of RADIO-based VLMs retaining post-conversion vision and diff --git a/modelopt/torch/quantization/calib/histogram.py b/modelopt/torch/quantization/calib/histogram.py index e27a5471e30..73ebc327892 100644 --- a/modelopt/torch/quantization/calib/histogram.py +++ b/modelopt/torch/quantization/calib/histogram.py @@ -218,7 +218,7 @@ def _normalize_distr(distr): if summ != 0: distr = distr / summ - bins = calib_hist[:] + bins = calib_hist.copy() bins[0] = bins[1] total_data = np.sum(bins) diff --git a/tests/unit/torch/quantization/test_calibrator.py b/tests/unit/torch/quantization/test_calibrator.py index 9f7d77ce6f7..2d489b14543 100644 --- a/tests/unit/torch/quantization/test_calibrator.py +++ b/tests/unit/torch/quantization/test_calibrator.py @@ -89,6 +89,32 @@ def test_track_amax_raises(self): class TestHistogramCalibrator: + @pytest.mark.parametrize("torch_hist", [False, True]) + def test_entropy_preserves_collected_histogram(self, torch_hist): + calibrator = calib.HistogramCalibrator(4, None, False, num_bins=32, torch_hist=torch_hist) + reference = calib.HistogramCalibrator(4, None, False, num_bins=32, torch_hist=torch_hist) + first_batch = torch.cat((torch.zeros(100), torch.arange(1, 33).float())) + calibrator.collect(first_batch) + reference.collect(first_batch) + expected_hist = np.asarray(calibrator._calib_hist).copy() + expected_edges = np.asarray(calibrator._calib_bin_edges).copy() + percentile = calibrator.compute_amax("percentile", percentile=50) + + first_entropy = calibrator.compute_amax("entropy", start_bin=16) + second_entropy = calibrator.compute_amax("entropy", start_bin=16) + + np.testing.assert_array_equal(calibrator._calib_hist, expected_hist) + np.testing.assert_array_equal(calibrator._calib_bin_edges, expected_edges) + assert calibrator.compute_amax("percentile", percentile=50) == percentile + assert first_entropy == second_entropy + next_batch = torch.tensor([0.0, 2.0, 4.0, 64.0]) + calibrator.collect(next_batch) + reference.collect(next_batch) + np.testing.assert_array_equal(calibrator._calib_hist, reference._calib_hist) + assert calibrator.compute_amax("percentile", percentile=50) == reference.compute_amax( + "percentile", percentile=50 + ) + @pytest.mark.skip(reason="TODO: Fix assertions in test_grow") def test_grow(self, verbose): x_1 = torch.tensor([0, 255, 255, 255, 255, 255])