Skip to content

vllm.v1.hisparse.layout

Functions:

create_hisparse_layout(vllm_config, groups, host_budget)

Size the host pool for groups laid out by get_hisparse_kv_cache_groups.

Source code in vllm/v1/hisparse/layout.py
def create_hisparse_layout(
    vllm_config: VllmConfig,
    groups: list[KVCacheGroupSpec],
    host_budget: int,
) -> HiSparseLayout:
    """Size the host pool for groups laid out by `get_hisparse_kv_cache_groups`."""
    (source_group,) = [group for group in groups if group.host_resident]
    shared_host_pool = use_shared_hisparse_host_pool(vllm_config)
    host_block_stride = get_hisparse_host_block_stride(
        source_group.kv_cache_spec.page_size_bytes,
        use_shared_host_pool=shared_host_pool,
    )
    host_num_blocks = host_budget // host_block_stride
    if host_num_blocks <= 0:
        raise ValueError("HiSparse has no allocatable host blocks.")
    # Every computed page needs a host block, so one request at max_model_len
    # must fit alongside the pool's null block and a copy-on-write tail.
    gpu_block_size = source_group.kv_cache_spec.block_size
    min_host_blocks = cdiv(vllm_config.model_config.max_model_len, gpu_block_size) + 2
    if host_num_blocks < min_host_blocks:
        raise ValueError(
            f"HiSparse host pool has {host_num_blocks} blocks but max_model_len "
            f"needs {min_host_blocks}; increase host_pool_gib."
        )

    return HiSparseLayout(
        source_group=source_group,
        device_groups=[group for group in groups if not group.host_resident],
        host_num_blocks=host_num_blocks,
        host_block_stride=host_block_stride,
        shared_host_pool=shared_host_pool,
    )

get_hisparse_kv_cache_groups(vllm_config, kv_cache_spec)

The groups HiSparse allocates: the host source group, then the GPU groups sharing one device pool. Derived resident/hot specs in kv_cache_spec (see resolve_hisparse_specs) are laid out again from the attention specs they belong to.

Source code in vllm/v1/hisparse/layout.py
def get_hisparse_kv_cache_groups(
    vllm_config: VllmConfig, kv_cache_spec: dict[str, KVCacheSpec]
) -> list[KVCacheGroupSpec] | None:
    """The groups HiSparse allocates: the host source group, then the GPU
    groups sharing one device pool. Derived resident/hot specs in
    `kv_cache_spec` (see `resolve_hisparse_specs`) are laid out again
    from the attention specs they belong to."""
    attention_config = getattr(vllm_config, "attention_config", None)
    if attention_config is None or attention_config.hisparse_config is None:
        return None

    mla_specs: dict[str, KVCacheSpec] = {
        name: spec
        for name, spec in kv_cache_spec.items()
        if isinstance(spec, MLAAttentionSpec)
    }
    other_specs = {
        name: spec
        for name, spec in kv_cache_spec.items()
        if not isinstance(
            spec, (MLAAttentionSpec, HiSparseResidentSpec, HiSparseHotSpec)
        )
    }
    if not mla_specs:
        return None

    from vllm.v1.core.kv_cache_utils import get_kv_cache_groups

    mla_group_spec = UniformTypeKVCacheSpecs.from_specs(mla_specs)
    assert mla_group_spec is not None
    mla_group = KVCacheGroupSpec(list(mla_specs), mla_group_spec)
    regular_groups = (
        get_kv_cache_groups(vllm_config, other_specs) if other_specs else []
    )
    return _lay_out_hisparse_groups(vllm_config, [mla_group, *regular_groups])