RocketSparseAttentionConfig#

class tensorrt_llm.llmapi.RocketSparseAttentionConfig(
*,
algorithm: Literal['rocket'] = 'rocket',
seq_len_threshold: int | None = None,
window_size: int | None = 32,
kernel_size: int | None = 63,
topr: int | float | None = 128,
topk: int | None = 64,
prompt_budget: int | None = 2048,
page_size: int | None = 4,
kt_cache_dtype: str | None = 'float8_e5m2',
)[source]#

Bases: SeqLenAwareSparseAttentionConfig

Configuration for RocketKV sparse attention.

field algorithm: Literal['rocket'] = 'rocket'#
field kernel_size: int | None = 63#

The kernel size for RocketKV.

field kt_cache_dtype: str | None = 'float8_e5m2'#

KT cache dtype

field page_size: int | None = 4#

Page size

field prompt_budget: int | None = 2048#

Prompt budget

field seq_len_threshold: int | None = None#

The sequence length threshold for separating short and long sequences.

field topk: int | None = 64#

Top-k

field topr: int | float | None = 128#

Top-r

field window_size: int | None = 32#

The window size for RocketKV.

__init__(**data: Any) → None#

Create a new model by parsing and validating input data from keyword arguments.

Raises [ValidationError][pydantic_core.ValidationError] if the input data cannot be validated to form a valid model.

self is explicitly positional-only to allow self as a field name.

get_indices_block_size() → int[source]#
needs_separate_short_long_cuda_graphs() → bool#

Whether to capture separate CUDA graphs for short and long sequences.

supports_backend(backend: str) → bool[source]#

Override if the sparse attention algorithm does not support a subset of the possible backends.

to_sparse_metadata_params(**kwargs)[source]#

Lower user-facing config into SparseMetadataParams.

to_sparse_params(**kwargs)[source]#

Lower user-facing config into SparseParams.