Managed KV Cache Request#

class ManagedKVCacheRequest#

One runtime request facade over orthogonal execution paging and context-reuse clients.

The current support matrix selects exactly one backend: bounded SWA when reuse is off, or context reuse with full KV storage. Keeping the runtime lifecycle behind this facade allows a later adapter to compose both without moving SWA rotation into the context-reuse manager.

In-flight admission, context-reuse backend only.

The bounded-SWA backend has no live-slot admission (its per-slot window state cannot join a running batch), and supportsSeatedAdmission() refuses such deployments before an engine that would call these is ever constructed; the kFailed/false returns here are the backstop.

ContextCacheRequest::AdmitSequenceStatus admitSequence(
std::vector<int32_t> const &tokenIds,
std::string const &loraWeightsName,
DecodingKvHeadroom const &headroom,
int32_t &prefillStart,
ResidentRef resident,
cudaStream_t stream,
std::vector<int32_t> const &mediaTokenIds = {},
std::vector<imageUtils::ImageData> const &imageBuffers = {},
std::vector<audioUtils::AudioData> const &audioBuffers = {}
)#
bool finalizeSequenceAdmission(
int32_t slot,
int32_t const &lookaheadToken,
int32_t fullInputLength
)#
bool retractSequenceAdmission() noexcept#

Public Functions

ManagedKVCacheRequest(ManagedKVCacheRequest&&) noexcept = default#
ManagedKVCacheRequest &operator=(ManagedKVCacheRequest&&) = delete#
ManagedKVCacheRequest(ManagedKVCacheRequest const&) = delete#
ManagedKVCacheRequest &operator=(
ManagedKVCacheRequest const&
) = delete#
~ManagedKVCacheRequest() = default#
bool hasContextReuse() const noexcept#
std::vector<int32_t> const *prefillStarts() const noexcept#
int32_t reuseTokenLength(int32_t slot) const#
bool publishHybridMtpEndpoint(
int32_t slot,
int32_t residentStateLength,
Tensor const &baseHiddenStates,
int32_t boundaryHiddenRow
)#
bool restoreHybridMtpBoundaryHidden(
int32_t slot,
Tensor &baseHiddenStates,
int32_t destinationRow
)#
bool preparePrefill()#
bool prepareSwaPrefill(
std::vector<int32_t> const &effectiveInputLengths
)#
bool enqueuePrefillCaptures()#
bool completePrefill(
DecodingInferenceContext const &context,
std::vector<int32_t> const &commonStateLengths
)#
bool prepareDecodeStep(
DecodingInferenceContext const &context,
DecodingKvHeadroom const &headroom
)#
bool completeDecodeStep(
DecodingInferenceContext const &context,
std::vector<int32_t> const &commonStateLengths
)#
bool beginBatchCompaction(
std::vector<int32_t> const &oldToNew,
int32_t newBatchSize,
Tensor &deviceBatchMapping
)#
bool completeBatchCompaction(
std::vector<int32_t> const &keepMapping
)#
bool finish()#

Public Static Functions

static std::optional<ManagedKVCacheRequest> begin(
ContextCacheCoordinator *contextCache,
BoundedSwaKVPageManager *swaPageManager,
LLMGenerationRequest const &request,
DecodingInferenceContext const &context,
bool speculativeRequest,
DecodingKvHeadroom const &headroom,
DecodingTokenStateContract tokenStateContract,
ContextCacheCommitPolicy commitPolicy,
std::vector<int32_t> const &mediaTokenIds = {}
)#