Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions CHANGELOG.rst
Original file line number Diff line number Diff line change
Expand Up @@ -14,6 +14,7 @@ Changelog
*Quantization*

- Add IQ1_S and IQ2_XS weight-only quantization with GGML-compatible 256-value block encoders, built-in ``iq1_s`` / ``iq2_xs`` PTQ recipes, and unified HF and Megatron export of the packed blocks. Quantized weights must have a final dimension divisible by 256, and Megatron export requires tensor and pipeline parallel sizes of 1.
- Add ``iq2_xxs`` weight-only quantization with a CUDA encoder and a ``general/ptq`` recipe, at 2.0625 bits per weight between ``iq1_s`` and ``iq2_xs``. The same 256-value block constraint applies.
- A recipe can now **delegate its whole body to another recipe** with a top-level ``$import``; any top-level key given alongside it overrides the imported one. ``metadata.recipe_type`` became optional along with it: a recipe states its kind with a ``# modelopt-schema:`` comment, with ``metadata.recipe_type``, or by delegating to a recipe that does, and only a recipe that another file imports has to carry the schema comment. Whatever a recipe does state must be true: a schema comment and a ``recipe_type`` must agree, and so must a recipe and the recipe it delegates to. ``modelopt_recipes/models/`` uses this for checkpoint entries that a portable recipe already reproduces: the entry aliases that recipe instead of copying it.
- Backfill the recipes behind NVIDIA's already-published checkpoints under ``modelopt_recipes/models/``, so a released checkpoint's quantization scheme is reachable from its own model-hub path rather than only from the general tier. For example, ``moonshotai/Kimi-K2.6`` (published as ``nvidia/Kimi-K2.6-NVFP4``) and ``Qwen/Qwen3.5-397B-A17B`` (published as ``nvidia/Qwen3.5-397B-A17B-NVFP4-V2``) each alias a portable recipe wholesale -- the general expert-only NVFP4 recipe and the ``qwen3_5_moe`` architecture recipe respectively -- rather than copying its body; other checkpoints follow in separate changes.
- Add ``layerwise.export_dir``: layerwise calibration writes each decoder layer to its own quantized checkpoint shard as it finishes, so no separate ``export_hf_checkpoint()`` pass is needed and, with ``layerwise.checkpoint_dir``, an interrupted run resumes without redoing finished layers. Calibration writes the layer shards; ``finalize()`` on the exporter left on the model adds the tail shard, the index and the config artifacts, and the checkpoint does not load until it runs. ``examples/hf_ptq`` does this for you. Supports FP8 and NVFP4 on single-process models, resident or offloaded, including multimodal models and models with MTP layers; other formats and placements raise ``NotImplementedError`` before calibration starts.
Expand Down
22 changes: 4 additions & 18 deletions modelopt/torch/export/convert_hf_config.py
Original file line number Diff line number Diff line change
Expand Up @@ -19,14 +19,7 @@
from collections import defaultdict
from typing import Any

from modelopt.torch.quantization.ggml import (
IQ1_S_BLOCK_BYTES,
IQ1_S_BLOCK_SIZE,
IQ1_S_EFFECTIVE_BITS,
IQ2_XS_BLOCK_BYTES,
IQ2_XS_BLOCK_SIZE,
IQ2_XS_EFFECTIVE_BITS,
)
from .quant_format import IQ_BLOCK_METADATA, IQ_FORMATS


def _quant_algo_to_group_config(quant_algo: str, group_size: int | None = None) -> dict[str, Any]:
Expand Down Expand Up @@ -127,15 +120,8 @@ def _quant_algo_to_group_config(quant_algo: str, group_size: int | None = None)
},
"weights": {"dynamic": False, "num_bits": 8, "type": "float", "group_size": gs},
}
elif quant_algo in ("IQ1_S", "IQ2_XS"):
if quant_algo == "IQ1_S":
block_size = IQ1_S_BLOCK_SIZE
payload_bytes = IQ1_S_BLOCK_BYTES
effective_bits = IQ1_S_EFFECTIVE_BITS
else:
block_size = IQ2_XS_BLOCK_SIZE
payload_bytes = IQ2_XS_BLOCK_BYTES
effective_bits = IQ2_XS_EFFECTIVE_BITS
elif quant_algo.lower() in IQ_FORMATS:
block_size, payload_bytes, effective_bits = IQ_BLOCK_METADATA[quant_algo.lower()]
if group_size not in (None, block_size):
raise ValueError(f"{quant_algo} requires group size {block_size}, got {group_size}")
# IQ payloads are self-contained blocks, not compressed-tensors integer groups.
Expand Down Expand Up @@ -239,7 +225,7 @@ def convert_hf_quant_config_format(input_config: dict[str, Any]) -> dict[str, An
"targets": ["Linear"],
}
new_config["config_groups"] = {"group_0": config_group_details}
elif quant_algo_value in ("IQ1_S", "IQ2_XS"):
elif str(quant_algo_value).lower() in IQ_FORMATS:
# Forward the caller's group size so a mismatched one is rejected rather than rewritten
# to the format's block size.
iq_metadata = _quant_algo_to_group_config(
Expand Down
50 changes: 47 additions & 3 deletions modelopt/torch/export/quant_format.py
Original file line number Diff line number Diff line change
Expand Up @@ -19,6 +19,21 @@
constants, for example, are in :mod:`modelopt.torch.export.trtllm.model_config`.
"""

from modelopt.torch.quantization.ggml import (
IQ1_S_BLOCK_BYTES,
IQ1_S_BLOCK_SIZE,
IQ1_S_EFFECTIVE_BITS,
IQ2_XS_BLOCK_BYTES,
IQ2_XS_BLOCK_SIZE,
IQ2_XS_EFFECTIVE_BITS,
IQ2_XXS_BLOCK_BYTES,
IQ2_XXS_BLOCK_SIZE,
IQ2_XXS_EFFECTIVE_BITS,
quantize_iq1_s,
quantize_iq2_xs,
quantize_iq2_xxs,
)

QUANTIZATION_NONE = None
QUANTIZATION_FP8 = "fp8"
QUANTIZATION_INT8_SQ = "int8_sq"
Expand All @@ -37,16 +52,45 @@
QUANTIZATION_FP8_PB_WO = "fp8_pb_wo"
QUANTIZATION_FP8_PC_PT = "fp8_pc_pt"
QUANTIZATION_IQ1_S = "iq1_s"
QUANTIZATION_IQ2_XXS = "iq2_xxs"
QUANTIZATION_IQ2_XS = "iq2_xs"

# Every GGML IQ format. They share the weight-only, 256-value-block, per-module-scale
# shape, so export treats them as one family; adding a format means adding it here
# rather than extending a tuple at each use site.
IQ_FORMATS = frozenset(
{
QUANTIZATION_IQ1_S,
QUANTIZATION_IQ2_XXS,
QUANTIZATION_IQ2_XS,
}
)

# Block geometry per IQ format: (block size, packed bytes per block, bits per weight). Checkpoint
# metadata spells the algorithm in upper case, so consumers look up
# ``IQ_BLOCK_METADATA[algo.lower()]`` rather than carrying a second spelling of the family.
IQ_BLOCK_METADATA = {
QUANTIZATION_IQ1_S: (IQ1_S_BLOCK_SIZE, IQ1_S_BLOCK_BYTES, IQ1_S_EFFECTIVE_BITS),
QUANTIZATION_IQ2_XXS: (IQ2_XXS_BLOCK_SIZE, IQ2_XXS_BLOCK_BYTES, IQ2_XXS_EFFECTIVE_BITS),
QUANTIZATION_IQ2_XS: (IQ2_XS_BLOCK_SIZE, IQ2_XS_BLOCK_BYTES, IQ2_XS_EFFECTIVE_BITS),
}


# The packer each format's checkpoint weights are written with. Both exporters resolve through
# this one mapping so they cannot drift apart.
IQ_PACKERS = {
QUANTIZATION_IQ1_S: quantize_iq1_s,
QUANTIZATION_IQ2_XXS: quantize_iq2_xxs,
QUANTIZATION_IQ2_XS: quantize_iq2_xs,
}


# Formats whose scales are purely per-module, so export never merges them across the q/k/v
# and gate/up groups that share an input. Every other format unifies input_amax (and, for
# NVFP4, weight_scale_2) across such a group, which only a whole-model forward can discover.
FUSION_FREE_FORMATS = frozenset(
FUSION_FREE_FORMATS = IQ_FORMATS | frozenset(
{
QUANTIZATION_FP8,
QUANTIZATION_IQ1_S,
QUANTIZATION_IQ2_XS,
QUANTIZATION_NONE,
QUANTIZATION_FP8_PB_REAL,
}
Expand Down
28 changes: 6 additions & 22 deletions modelopt/torch/export/quant_utils.py
Original file line number Diff line number Diff line change
Expand Up @@ -27,14 +27,6 @@

from modelopt import __version__
from modelopt.torch.models import get_spec, list_all_possible
from modelopt.torch.quantization.ggml import (
IQ1_S_BLOCK_BYTES,
IQ1_S_BLOCK_SIZE,
IQ1_S_EFFECTIVE_BITS,
IQ2_XS_BLOCK_BYTES,
IQ2_XS_BLOCK_SIZE,
IQ2_XS_EFFECTIVE_BITS,
)
from modelopt.torch.quantization.model_calib import (
enable_stats_collection,
finish_stats_collection,
Expand All @@ -59,6 +51,8 @@
from ..quantization.nn import NVFP4StaticQuantizer, SequentialQuantizer, TensorQuantizer
from .model_utils import TiedWeightMap, get_language_model_from_vl
from .quant_format import (
IQ_BLOCK_METADATA,
IQ_FORMATS,
KV_CACHE_FP8,
KV_CACHE_FP8_K_NVFP4_V,
KV_CACHE_INT8,
Expand All @@ -71,8 +65,6 @@
QUANTIZATION_INT4_AWQ,
QUANTIZATION_INT8_SQ,
QUANTIZATION_INT8_WO,
QUANTIZATION_IQ1_S,
QUANTIZATION_IQ2_XS,
QUANTIZATION_MXFP4,
QUANTIZATION_MXFP8,
QUANTIZATION_NONE,
Expand Down Expand Up @@ -474,8 +466,7 @@ def uses_iq_quantization(module) -> bool:
if (
weight_quantizer is not None
and weight_quantizer.is_enabled
and getattr(weight_quantizer, "num_bits", None)
in (QUANTIZATION_IQ1_S, QUANTIZATION_IQ2_XS)
and getattr(weight_quantizer, "num_bits", None) in IQ_FORMATS
):
return True
return any(uses_iq_quantization(child) for _, child in module.named_children())
Expand Down Expand Up @@ -515,7 +506,7 @@ def _get_quantization_from_layer(layer, quantizer_attr_names: QuantizerAttrNames
return QUANTIZATION_W4A8_AWQ

# Handle individual num_bits cases
if weight_quantizer.num_bits in (QUANTIZATION_IQ1_S, QUANTIZATION_IQ2_XS):
if weight_quantizer.num_bits in IQ_FORMATS:
if weight_quantizer.backend != "ggml":
raise ValueError("IQ formats require the built-in 'ggml' quantization backend")
# Both exporters return before collecting input_scale and before the pre_quant_scale
Expand Down Expand Up @@ -781,15 +772,8 @@ def process_layer_quant_config(layer_config_dict):
"quant_algo": "MXFP8",
"group_size": block_size_value,
}
elif v in (QUANTIZATION_IQ1_S, QUANTIZATION_IQ2_XS):
if v == QUANTIZATION_IQ1_S:
block_size = IQ1_S_BLOCK_SIZE
payload_bytes = IQ1_S_BLOCK_BYTES
effective_bits = IQ1_S_EFFECTIVE_BITS
else:
block_size = IQ2_XS_BLOCK_SIZE
payload_bytes = IQ2_XS_BLOCK_BYTES
effective_bits = IQ2_XS_EFFECTIVE_BITS
elif v in IQ_FORMATS:

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Bot comment.

The new metadata reaches get_quant_config, but convert_hf_config.py still recognizes only ("IQ1_S", "IQ2_XS") in both convert_hf_quant_config_format and _quant_algo_to_group_config. Consequently a uniform IQ2_XXS export drops packing, block_payload_bytes, effective_bits, and group_size; mixed exports warn and omit those fields from the IQ2_XXS config group. Extend both conversion paths, preferably using shared IQ metadata rather than another format list. Add IQ2_XXS to test_iq_quantization_config and cover mixed conversion and rejection of a non-256 group size.

block_size, payload_bytes, effective_bits = IQ_BLOCK_METADATA[v]
if block_size_value != block_size:
raise ValueError(
f"{v.upper()} requires block size {block_size}, got {block_size_value}"
Expand Down
11 changes: 4 additions & 7 deletions modelopt/torch/export/unified_export_hf.py
Original file line number Diff line number Diff line change
Expand Up @@ -67,7 +67,6 @@
from modelopt.torch.opt.conversion import ModeloptStateManager, modelopt_state
from modelopt.torch.opt.plugins.huggingface import _MODELOPT_STATE_SAVE_NAME
from modelopt.torch.quantization import set_quantizer_by_cfg_context
from modelopt.torch.quantization.ggml import quantize_iq1_s, quantize_iq2_xs
from modelopt.torch.quantization.nn import SequentialQuantizer, TensorQuantizer
from modelopt.torch.quantization.qtensor import MXFP8QTensor, NVFP4QTensor
from modelopt.torch.quantization.qtensor.base_qtensor import QTensorWrapper
Expand Down Expand Up @@ -101,11 +100,11 @@
)
from .quant_format import (
FUSION_FREE_FORMATS,
IQ_FORMATS,
IQ_PACKERS,
QUANTIZATION_FP8,
QUANTIZATION_FP8_PB_REAL,
QUANTIZATION_FP8_PC_PT,
QUANTIZATION_IQ1_S,
QUANTIZATION_IQ2_XS,
QUANTIZATION_MXFP8,
QUANTIZATION_NONE,
QUANTIZATION_NVFP4,
Expand Down Expand Up @@ -630,15 +629,13 @@ def _export_quantized_weight(
"which dispatches to the streaming writer that materialises weights layer-by-layer."
)

if quantization_format in (QUANTIZATION_IQ1_S, QUANTIZATION_IQ2_XS):
if quantization_format in IQ_FORMATS:
if weight_name != "weight":
raise NotImplementedError(
"IQ unified export currently supports modules with a standard 'weight' "
f"attribute, got {weight_name!r} on {type(sub_module).__name__}"
)
quantize_iq = (
quantize_iq1_s if quantization_format == QUANTIZATION_IQ1_S else quantize_iq2_xs
)
quantize_iq = IQ_PACKERS[quantization_format]
packed_weight, _ = quantize_iq(weight.to(dtype))
setattr(sub_module, weight_name, nn.Parameter(packed_weight, requires_grad=False))
maybe_clear_cuda_cache()
Expand Down
26 changes: 13 additions & 13 deletions modelopt/torch/export/unified_export_megatron.py
Original file line number Diff line number Diff line change
Expand Up @@ -35,7 +35,6 @@
from safetensors.torch import save_file

from modelopt import __version__
from modelopt.torch.quantization.ggml import quantize_iq1_s, quantize_iq2_xs
from modelopt.torch.quantization.nn.modules.tensor_quantizer import GroupedQuantizer
from modelopt.torch.utils import import_plugin, warn_rank_0
from modelopt.torch.utils.plugins.hf_checkpoint_utils import (
Expand All @@ -57,13 +56,13 @@
)
from .plugins.megatron_importer import GPTModelImporter, _get_mamba_conv1d
from .quant_format import (
IQ_FORMATS,

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Do we have unit tests for megatron as well?

IQ_PACKERS,
KV_CACHE_FP8,
KV_CACHE_NVFP4,
QUANTIZATION_FP8,
QUANTIZATION_FP8_PB_REAL,
QUANTIZATION_FP8_PB_WO,
QUANTIZATION_IQ1_S,
QUANTIZATION_IQ2_XS,
QUANTIZATION_NONE,
QUANTIZATION_NVFP4,
QUANTIZATION_W4A16_NVFP4,
Expand All @@ -85,6 +84,7 @@
import transformers
from transformers import AutoProcessor


has_mcore = False
with import_plugin("megatron"):
from megatron.core.models.gpt import GPTModel
Expand Down Expand Up @@ -351,7 +351,7 @@ def save_pretrained(
quantization = "NVFP4"
elif quantization_format == QUANTIZATION_W4A16_NVFP4:
quantization = "W4A16_NVFP4"
elif quantization_format in (QUANTIZATION_IQ1_S, QUANTIZATION_IQ2_XS):
elif quantization_format in IQ_FORMATS:
Comment thread
coderabbitai[bot] marked this conversation as resolved.
quantization = quantization_format.upper()

if is_last_stage_main_rank:
Expand Down Expand Up @@ -1115,7 +1115,7 @@ def _get_quantized_state(
self._record_excluded_module(prefix)
block_size = get_weight_block_size(module)

is_iq = qformat in (QUANTIZATION_IQ1_S, QUANTIZATION_IQ2_XS)
is_iq = qformat in IQ_FORMATS
name_to_value = self._get_weight_bias(
module, dtype, name_to_value, keep_weight_device=is_iq
)
Expand Down Expand Up @@ -1185,7 +1185,7 @@ def _get_weight_scales(self, quantized_state: dict[str, Any], qformat: str):
@staticmethod
def _pack_iq_weight(weight: torch.Tensor, qformat: str) -> torch.Tensor:
"""Pack one ``[out, in]`` weight and return its CPU payload."""
quantize_iq = quantize_iq1_s if qformat == QUANTIZATION_IQ1_S else quantize_iq2_xs
quantize_iq = IQ_PACKERS[qformat]
packed_weight, _ = quantize_iq(weight)
return packed_weight.detach().cpu()

Expand All @@ -1210,7 +1210,7 @@ def _reject_unsupported_fused_iq_export(qformat: str) -> None:
The one gap left is a rank holding no local expert at all, which needs expert-parallel
size to exceed the expert count. Worth revisiting if that becomes a supported topology.
"""
if qformat in (QUANTIZATION_IQ1_S, QUANTIZATION_IQ2_XS):
if qformat in IQ_FORMATS:
raise NotImplementedError(
"Fused-MoE IQ export requires a deployment loader that supports "
"[num_experts, out_features, in_features // 256, payload_bytes]"
Expand Down Expand Up @@ -1280,7 +1280,7 @@ def _name_remapping(
weight = weight + 1.0
weight_scale, weight_scale_2 = self._get_weight_scales(name_to_value, qformat)

if qformat in (QUANTIZATION_IQ1_S, QUANTIZATION_IQ2_XS):
if qformat in IQ_FORMATS:
self._state_dict.update(self._get_iq_weight_state(prefix + "weight", weight, qformat))
elif weight_scale is None:
self._state_dict[prefix + "weight"] = weight
Expand Down Expand Up @@ -1327,7 +1327,7 @@ def _gated_mlp_slicing(
gate_proj_weight = weight[:ffn_hidden_size, :]
up_proj_weight = weight[ffn_hidden_size:, :]

if qformat in (QUANTIZATION_IQ1_S, QUANTIZATION_IQ2_XS):
if qformat in IQ_FORMATS:
self._state_dict.update(
self._get_iq_weight_state(gate_proj_prefix + "weight", gate_proj_weight, qformat)
)
Expand Down Expand Up @@ -1501,7 +1501,7 @@ def _grouped_mlp_slicing(
seen_qformat, seen_block_size = qformat, block_size

weight = state_dict[weight_key].to(self.dtype)
if qformat not in (QUANTIZATION_IQ1_S, QUANTIZATION_IQ2_XS):
if qformat not in IQ_FORMATS:
weight = weight.cpu()
weight_scale_cpu = (
weight_scale.detach().cpu().clone() if weight_scale is not None else None
Expand Down Expand Up @@ -1533,7 +1533,7 @@ def _grouped_mlp_slicing(
]

for shard_prefix, shard_weight, shard_scale in shards:
if qformat in (QUANTIZATION_IQ1_S, QUANTIZATION_IQ2_XS):
if qformat in IQ_FORMATS:
local_expert_state.update(
self._get_iq_weight_state(
shard_prefix + "weight", shard_weight, qformat
Expand Down Expand Up @@ -1702,7 +1702,7 @@ def _take(tensor, index, last_dim, with_gate=False):
proj_weights = [_take(weight, s, hidden_size, g) for s, g in zip(slices, gated)]
proj_keys = [p + "weight" for p in prefixes]

if qformat in (QUANTIZATION_IQ1_S, QUANTIZATION_IQ2_XS):
if qformat in IQ_FORMATS:
for key, weight in zip(proj_keys, proj_weights):
self._state_dict.update(self._get_iq_weight_state(key, weight, qformat))
elif weight_scale is None:
Expand Down Expand Up @@ -1820,7 +1820,7 @@ def _gated_delta_net_slicing(self, module, prefix, is_mtp=False):
proj_keys = [p + "weight" for p in proj_prefixes]
weight_scale, weight_scale_2 = self._get_weight_scales(name_to_value, qformat)

if qformat in (QUANTIZATION_IQ1_S, QUANTIZATION_IQ2_XS):
if qformat in IQ_FORMATS:
for proj_prefix, proj_weight in zip(proj_prefixes, proj_weights):
if proj_prefix in keep_bf16:
self._state_dict[proj_prefix + "weight"] = proj_weight.cpu()
Expand Down
1 change: 1 addition & 0 deletions modelopt/torch/kernels/quantization/ggml/common.cuh
Original file line number Diff line number Diff line change
Expand Up @@ -49,6 +49,7 @@ constexpr int kScaleBytes = 2;
// them cannot drift apart.
constexpr int kIq1sEntries = 2048;
constexpr int kIq2xsEntries = 512;
constexpr int kIq2xxsEntries = 256;

// One CUDA block encodes one GGML block. The reductions below fold over exactly this many warps,
// and each kernel static_asserts that its codebook divides evenly among the threads.
Expand Down
Loading
Loading