Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 2 additions & 0 deletions CHANGELOG.rst
Original file line number Diff line number Diff line change
Expand Up @@ -101,6 +101,8 @@ Changelog

**Bug Fixes**

- Preserve accumulated histogram counts when computing an entropy calibration threshold with NumPy histograms.

- Fix Megatron unified HF export of MoE models with grouped-GEMM experts when only the experts are quantized (e.g. ``nvfp4_experts_only-*`` recipes): ``hf_quant_config.json`` and the ``quantization_config`` in ``config.json`` were not written, so the quantized experts were served as unquantized weights. Re-export such checkpoints.
- Fix Megatron-Core checkpoint saving for quantized grouped MoE experts when tensor and expert parallelism are both enabled.
- Fix unified HuggingFace export of RADIO-based VLMs retaining post-conversion vision and
Expand Down
2 changes: 1 addition & 1 deletion modelopt/torch/quantization/calib/histogram.py
Original file line number Diff line number Diff line change
Expand Up @@ -218,7 +218,7 @@ def _normalize_distr(distr):
if summ != 0:
distr = distr / summ

bins = calib_hist[:]
bins = calib_hist.copy()
bins[0] = bins[1]

total_data = np.sum(bins)
Expand Down
26 changes: 26 additions & 0 deletions tests/unit/torch/quantization/test_calibrator.py
Original file line number Diff line number Diff line change
Expand Up @@ -89,6 +89,32 @@ def test_track_amax_raises(self):


class TestHistogramCalibrator:
@pytest.mark.parametrize("torch_hist", [False, True])
def test_entropy_preserves_collected_histogram(self, torch_hist):
calibrator = calib.HistogramCalibrator(4, None, False, num_bins=32, torch_hist=torch_hist)
reference = calib.HistogramCalibrator(4, None, False, num_bins=32, torch_hist=torch_hist)
first_batch = torch.cat((torch.zeros(100), torch.arange(1, 33).float()))
calibrator.collect(first_batch)
reference.collect(first_batch)
expected_hist = np.asarray(calibrator._calib_hist).copy()
expected_edges = np.asarray(calibrator._calib_bin_edges).copy()
percentile = calibrator.compute_amax("percentile", percentile=50)

first_entropy = calibrator.compute_amax("entropy", start_bin=16)
second_entropy = calibrator.compute_amax("entropy", start_bin=16)

np.testing.assert_array_equal(calibrator._calib_hist, expected_hist)
np.testing.assert_array_equal(calibrator._calib_bin_edges, expected_edges)
assert calibrator.compute_amax("percentile", percentile=50) == percentile
assert first_entropy == second_entropy
next_batch = torch.tensor([0.0, 2.0, 4.0, 64.0])
calibrator.collect(next_batch)
reference.collect(next_batch)
np.testing.assert_array_equal(calibrator._calib_hist, reference._calib_hist)
assert calibrator.compute_amax("percentile", percentile=50) == reference.compute_amax(
"percentile", percentile=50
)

@pytest.mark.skip(reason="TODO: Fix assertions in test_grow")
def test_grow(self, verbose):
x_1 = torch.tensor([0, 255, 255, 255, 255, 255])
Expand Down