QuantConfig#

class tensorrt_llm.llmapi.QuantConfig(
*,
quant_algo: QuantAlgo | None = None,
kv_cache_quant_algo: QuantAlgo | None = None,
group_size: int | None = 128,
smoothquant_val: float = 0.5,
clamp_val: List[float] | None = None,
use_meta_recipe: bool = False,
has_zero_point: bool = False,
pre_quant_scale: bool = False,
exclude_modules: List[str] | None = None,
mamba_ssm_cache_dtype: str | None = None,
mamba_ssm_stochastic_rounding: bool = False,
mamba_ssm_philox_rounds: Annotated[int, Ge(ge=1)] = 10,
)[source]#

Bases: StrictBaseModel

Serializable quantization configuration class, part of the PretrainedConfig.

field clamp_val: List[float] | None = None#

Clamp values used in FP8 rowwise quantization.

field exclude_modules: List[str] | None = None#

Module name patterns that are skipped in quantization.

field group_size: int | None = 128#

Group size for group-wise quantization.

field has_zero_point: bool = False#

Whether to use zero point for quantization.

field kv_cache_quant_algo: QuantAlgo | None = None#

KV cache quantization algorithm.

field mamba_ssm_cache_dtype: str | None = None#

Data type for mamba SSM cache.

field mamba_ssm_philox_rounds: int = 10#

Number of Philox rounds for stochastic rounding PRNG. Higher values give better randomness.

Constraints:
  • ge = 1

field mamba_ssm_stochastic_rounding: bool = False#

Enable stochastic rounding for Mamba SSM state updates. Requires fp16 cache.

field pre_quant_scale: bool = False#

Whether to use pre-quant scale for quantization.

field quant_algo: QuantAlgo | None = None#

Quantization algorithm.

field smoothquant_val: float = 0.5#

Smoothing parameter alpha used in smooth quant.

field use_meta_recipe: bool = False#

Whether to use Meta’s recipe for FP8 rowwise quantization.

__init__(**data: Any) → None#

Create a new model by parsing and validating input data from keyword arguments.

Raises [ValidationError][pydantic_core.ValidationError] if the input data cannot be validated to form a valid model.

self is explicitly positional-only to allow self as a field name.

classmethod from_dict(
config: dict,
) → QuantConfig[source]#

Create a QuantConfig instance from a dict.

Parameters:

config (dict) – The dict used to create QuantConfig.

Returns:

The QuantConfig created from dict.

Return type:

tensorrt_llm.models.modeling_utils.QuantConfig

is_module_excluded_from_quantization(name: str) → bool[source]#

Check if the module is excluded from quantization.

A module is excluded if its own name or any ancestor (split on .) matches an entry in exclude_modules via fnmatch or a re: prefixed regex. The ancestor walk means listing a parent module (without a glob suffix) implicitly excludes all of its children.

A trailing .* subtree wildcard also matches the parent node itself, so an entry like model.layers.1.* excludes both model.layers.1 and everything under it. This keeps a subtree exclusion consistent regardless of whether the producer wrote it as model.layers.1 / model.layers.1* / model.layers.1.* (modelopt mixes these forms within a single checkpoint).

Parameters:

name (str) – The name of the module.

Returns:

True if the module is excluded from quantization, False otherwise.

Return type:

bool

property layer_quant_mode: QuantMode#
property quant_mode: QuantModeWrapper#