Skip to content

vllm.model_executor.layers.quantization.inc.schemes.inc_wna16_scheme

_check_xpu_w4a8_supported(layer_config, prefix)

Raise unless int4_gemm_w4a8 can serve this layer.

The backend is requested explicitly, so an unusable configuration is an error rather than something to silently fall back from.

Source code in vllm/model_executor/layers/quantization/inc/schemes/inc_wna16_scheme.py
def _check_xpu_w4a8_supported(layer_config: "INCLayerConfig", prefix: str) -> None:
    """Raise unless ``int4_gemm_w4a8`` can serve this layer.

    The backend is requested explicitly, so an unusable configuration is an
    error rather than something to silently fall back from.
    """
    import torch

    if not hasattr(torch.ops._xpu_C, "int4_gemm_w4a8"):
        raise NotImplementedError(
            "VLLM_XPU_INC_WNA16_BACKEND=w4a8 requires the int4_gemm_w4a8 op, "
            "which this build of vllm-xpu-kernels does not provide. "
            f"Layer: {prefix}."
        )
    assert isinstance(layer_config.group_size, int), (
        "WNA16 only supports integer group_size."
    )
    if layer_config.group_size <= 0 or layer_config.group_size % 32 != 0:
        raise NotImplementedError(
            "VLLM_XPU_INC_WNA16_BACKEND=w4a8 requires a group size that is a "
            f"positive multiple of 32, got {layer_config.group_size}. "
            f"Layer: {prefix}."
        )

_humming_weight_config(layer_config)

Build the humming weight-schema config for a WNA16 int checkpoint.

Source code in vllm/model_executor/layers/quantization/inc/schemes/inc_wna16_scheme.py
def _humming_weight_config(layer_config: "INCLayerConfig") -> dict:
    """Build the humming weight-schema config for a WNA16 int checkpoint."""
    if layer_config.is_gptq:
        return {
            "quant_method": "gptq",
            "bits": layer_config.bits,
            "group_size": layer_config.group_size,
            "desc_act": False,
            "sym": layer_config.sym,
        }
    if layer_config.is_awq:
        return {
            "quant_method": "awq",
            "bits": layer_config.bits,
            "group_size": layer_config.group_size,
            "zero_point": not layer_config.sym,
        }
    raise NotImplementedError(
        "INC humming dispatch only supports gptq/awq packed int checkpoints, "
        f"but found {layer_config}."
    )