LLM Rank Runtime#

class LLMRankRuntime#

Public Types

using TokenBroadcastFn = std::function<bool(void *buffer, int32_t count, cudaStream_t stream)>#

Callback type for broadcasting GPU int32 buffers across parallel ranks. Rank 0 sends; peer ranks receive.

Public Functions

LLMRankRuntime(
std::string const &engineDir,
std::string const &multimodalEngineDir,
std::unordered_map<std::string, std::string> const &loraWeightsMap,
std::optional<SpecDecodeDraftingConfig> const &draftingConfig,
cudaStream_t stream,
ParallelMapping const &mapping,
tokenizer::Tokenizer &tokenizer,
chat_template::ChatTemplate const &chatTemplate,
ContextCacheConfig const &contextCacheConfig,
std::string const &checkpointDir,
std::string const &draftCheckpointDir
)#

Construct one rank-local runtime from a fully-resolved parallel mapping. Preferred entry point — carries tensor/context/expert coordinates so future CP/EP support needs no further constructor changes.

LLMRankRuntime(
ModelArtifacts &&artifacts,
std::string const &engineDir,
std::string const &multimodalEngineDir,
std::unordered_map<std::string, std::string> const &loraWeightsMap,
std::optional<SpecDecodeDraftingConfig> const &draftingConfig,
cudaStream_t stream,
ParallelMapping const &mapping,
tokenizer::Tokenizer &tokenizer,
chat_template::ChatTemplate const &chatTemplate,
ContextCacheConfig const &contextCacheConfig
)#
~LLMRankRuntime()#

Destructor.

bool captureDecodingCUDAGraph(cudaStream_t stream)#

Capture CUDA graphs for decoding stages to optimize performance.

When draft model is present, captures graphs for draft proposal, draft accept token, base verification, and base vanilla decoding. Without draft model, captures only vanilla decoding graphs.

Note

If capture fails for any stage, the inference can proceed without CUDA graph capture, but at cost of performance degradation.

Parameters:

stream – CUDA stream

Throws:

std::runtime_error – if a tensor reshape operation fails

Returns:

True if all stage captures succeed, false otherwise

bool handleRequest(
LLMGenerationRequest const &request,
LLMGenerationResponse &response,
cudaStream_t stream,
bool outputThinkerEmbeddings = false,
TokenBroadcastFn tokenBroadcast = nullptr,
int32_t parallelRank = -1,
GenerationBoundaryHook const &boundaryHook = {}
)#
std::unique_ptr<SteppedGeneration> beginGeneration(
LLMGenerationRequest const &request,
LLMGenerationResponse &response,
cudaStream_t stream,
bool outputThinkerEmbeddings,
TokenBroadcastFn tokenBroadcast,
int32_t parallelRank,
RequestId founderRequestId = 0
)#

Everything handleRequest does before the generation loop: validation, context population, the context-cache admit, prefill-execution setup, streaming setup, and the founding prefill itself. Returns null on any refusal (the same conditions that returned false), and throws where handleRequest threw. request must outlive the returned object.

bool finishGeneration(
SteppedGeneration &generation,
LLMGenerationRequest const &request,
LLMGenerationResponse &response,
cudaStream_t stream
)#

Everything handleRequest does after the loop: the drained-batch check, the context-cache finish, metrics, and response assembly from whatever the loop’s harvests left in completedBatches.

bool genAndSaveSystemPromptKVCache(
std::string const &prompt,
std::string const &loraWeightsName,
cudaStream_t stream
)#

Generate and save system prompt KV cache (public API matching standard runtime signature)

Parameters:
  • prompt – The system prompt to generate the KVCache

  • loraWeightsName – The name of the LoRA weights

  • stream – The CUDA stream used for the generation

Throws:

std::runtime_error – if a CUDA operation fails

Returns:

True if the KVCache is generated and saved successfully, false otherwise

void setActionNoiseSeed(int32_t seed) noexcept#

Set the random seed used when initializing the action diffusion noise trajectory.

Parameters:

seed – Random seed value; has no effect if no action runner is loaded

void setVisualPrunerConfig(VisualPrunerConfig const &config)#

Enable visual-token pruning for supported VLM prefill execution.

inline metrics::LLMPrefillMetrics const &getPrefillMetrics(
) const noexcept#

Get LLM prefill stage metrics.

inline metrics::SpecDecodeGenerationMetrics const &getSpecDecodeGenerationMetrics(
) const noexcept#

Get speculative decoding generation stage metrics (only meaningful when draft model is present)

inline char const *getSpeculativeDecodingStrategyName(
) const noexcept#
inline metrics::LLMGenerationMetrics const &getGenerationMetrics(
) const noexcept#

Get vanilla generation stage metrics (only meaningful when no draft model / vanilla path)

std::optional<ContextCacheMetrics> getContextCacheMetrics(
) const noexcept#

Get context-cache metrics, or nullopt when the runtime cache is disabled.

inline metrics::MultimodalMetrics getMultimodalMetrics(
) const noexcept#

Get multimodal metrics (returns empty metrics if no multimodal runner)

inline rt::Tensor const &getEmbeddingTable() const#

Get the embedding table (for Talker streaming pipeline)

inline rt::Tensor const *getBaseModelHiddenStates(
int32_t layerIdx
) const noexcept#

Get a base model hidden-states buffer for the requested layer index.

Buffers are owned by the runtime and reused across requests. Layer 0 corresponds to the post-multimodal input embeddings (backed up before the decode loop reshapes them); other layer indices correspond to engine-output hidden states (e.g. acceptHiddenLayer for the Qwen3-Omni Talker, or future MTP layers).

Lifetime contract:

  • Buffers are sized to {maxRuntimeBatchSize, maxSupportedInputLength, hiddenSize}.

  • Contents are cleared (overwritten) at the start of each handleRequest() call and remain valid until the next handleRequest() begins. The buffer is reshaped to {activeBatchSize, prefillLength, hiddenSize} for the most recent request — use getBaseModelPrefillLength() to query the valid prefill length.

  • The caller is responsible for consuming the data within that window.

Parameters:

layerIdx – Layer index. 0 = input embeddings (post-multimodal); other indices are model-specific (e.g. acceptHiddenLayer for Qwen3-Omni Talker).

Returns:

Pointer to the buffer, or nullptr if no buffer is registered for that layer.

inline int32_t getBaseModelPrefillLength() const noexcept#

Number of valid prefill tokens in the hidden-states buffers from the most recent handleRequest() call. Returns 0 if no hidden-states output was requested.

inline std::vector<std::vector<int32_t>> const &getBaseModelInputTokenIds(
) const noexcept#

Per-batch input token IDs from the most recent handleRequest() call. Cleared at the start of each handleRequest(); valid until the next one begins.

inline bool hasDraftModel() const noexcept#

Check if draft model is loaded and spec-decode is available.

bool supportsSeatedAdmission() const noexcept#

Whether this deployment can take boundary admissions via prefillSlotInPlace.

False for the static refusals prefillSlotInPlace would otherwise throw on mid-request: draft (speculative) engines, diffusion backbones, and bounded-SWA deployments. Probed once at RequestEngine construction so an unsupported deployment falls back to the blocking path instead of failing admissions.

int32_t maxBatchSize() const noexcept#

The batch dimension the engine was built with: the physical bound on how many sequences a batch can seat, and therefore on any scheduler’s maxBatchSize.

class GenerationSession : public trt_edgellm::rt::GenerationBoundary#

Handle generation request.

One request’s generation, advanced a step at a time.

Exists because the decode loop carries state between steps: a thinking-done flag per slot, several tokenizer ids, and a stop predicate that closes over them. While those were locals in one long function, that state and the loop were forced to share a lifetime by construction. As an object they still share one, but it is now the object’s, and a caller can hold it across steps instead of being obliged to run the loop to the end in a single call.

The prefill preceding the first step also produces a token on most backbones, so a session is primed once before it is stepped &#8212; see primeFromPrefill().

Note

Calls on the same runtime must be externally serialized. An accidental overlap with another handleRequest() is rejected before runtime or response state is mutated; this is not a general thread-safety guarantee.

Param request:

Generation request with prompts and parameters

Param response:

Output response with generated tokens and text

Param stream:

CUDA stream

Throws std::runtime_error:

if an LLM or CUDA operation fails

Return:

True on success, false on failure

Public Functions

GenerationSession(
LLMRankRuntime &runtime,
DecodingInferenceContext &context,
DecodingStrategy &strategy,
ManagedKVCacheRequest *managedRequest,
LLMGenerationRequest const &request,
DecodingKvHeadroom const &kvHeadroom,
cudaStream_t stream,
bool boundarySchedulingActive
)#
~GenerationSession()#

Clears the stop predicate the context holds, which closes over members of this object.

GenerationSession(GenerationSession const&) = delete#
GenerationSession &operator=(GenerationSession const&) = delete#
bool primeFromPrefill()#

Consume the token prefill produced, before any decode step runs.

Prefill emits a token on every backbone except the diffusion one, and that token has to travel the same cancel/decode/finalize/emit path a decode step would give it. Skipping this drops the first token of every request.

bool finished() const#

True once every slot has reached a terminal state, or none are left.

bool advance()#

Advance every live slot by one decode step.

virtual AdmitDecision admitSequence(SlotSeed seed) override#

Join one new sequence to this running batch and prefill it, without touching the sequences in flight.

Admission plus seating: the context grows a slot (appendSlot), the session’s own per-slot state grows with it, and prefillSlotInPlace runs the new slot’s prompt as a seated batch-1 pass. On a failed prefill the slot is marked terminal with kError so the next eviction files its result; the batch’s other sequences are unaffected either way.

Whether the arriving request may share this batch at all (sampling parameters, adapter, step budget) is BatchCompatibility’s question, answered before a seed is built.

Throws:

std::runtime_error – for seeds appendSlot rejects, and for deployments prefillSlotInPlace refuses; nothing is modified on those paths.

virtual std::unordered_map<int32_t, BatchResult> takeCompletedAtOrAbove(
int32_t firstIndex
) override#

Move out the results of finished sequences whose original index is >= firstIndex.

Sequences admitted mid-flight carry indices above the founding request’s range, so this is how their results leave the batch without appearing in the founding caller’s response. Harvested at every boundary by whoever drives the loop.

virtual AdmitDecision admitRequest(
LLMGenerationRequest const &request,
int32_t originalIndex,
RequestId requestId
) override#

Admit a caller-level request: tokenization included, so a scheduler never builds a SlotSeed by hand or holds a tokenizer.

Whether the request may share this batch at all &#8212; sampling parameters, adapter, budgets &#8212; is the scheduler’s question (BatchCompatibility), answered before calling this.

Parameters:

request – A single-sequence request. Its stream channel, stop strings and logit bias ride along; its sampling parameters are ignored in favour of the batch’s. Scheduler policy keeps media requests founder-only until live admission has dedicated coverage.

Throws:

std::runtime_error – for an invalid request or an unsupported media deployment; nothing is modified.

virtual LLMGenerationResponse materializeResult(
BatchResult const &result,
std::vector<std::string> const &stopStrings
) const override#

Turn one finished sequence’s result into the caller-facing response, decoding included, exactly as the founding request’s own assembly would have.

Lives here because the conversion needs the tokenizer and the stop-string trimming that the runtime owns; a scheduler holding raw BatchResults would either skip the text or grow its own copy of both.

inline virtual int32_t residentCount() const override#

Sequences currently resident, finished or not. Admission capacity is judged on this.

struct SteppedGeneration#

Everything one request owns between beginGeneration and finishGeneration: the same state handleRequest used to keep on its stack, packaged so a stepped control plane can hold a request open across ticks. The founding request must outlive this object; member order is load-bearing (the stream finalizer references the context and must be destroyed first).

Public Functions

inline explicit SteppedGeneration(
std::atomic<bool> &activeFlag,
Tensor &hostTokenStorage
) noexcept#
inline ~SteppedGeneration() noexcept#
SteppedGeneration(SteppedGeneration const&) = delete#
SteppedGeneration &operator=(SteppedGeneration const&) = delete#
inline ManagedKVCacheRequest *managedRequest() noexcept#

Public Members

InProgressGuard guard#
DecodingInferenceContext context#
DecodingStrategy *strategy = {nullptr}#
std::optional<ManagedKVCacheRequest> managedKVCacheRequest#
DecodingKvHeadroom kvHeadroom = {}#
std::optional<StreamChannelFinalizer> streamFinalizer#
bool enableSpecDecode = {false}#
bool hasActionRequest = {false}#
struct InProgressGuard#

Releases the runtime’s one-request-at-a-time latch, whatever path retires the request.

Public Functions

inline explicit InProgressGuard(
std::atomic<bool> &active
) noexcept#
inline ~InProgressGuard() noexcept#

Public Members

std::atomic<bool> &mActive#
class GenerationSession : public trt_edgellm::rt::GenerationBoundary

Handle generation request.

One request’s generation, advanced a step at a time.

Exists because the decode loop carries state between steps: a thinking-done flag per slot, several tokenizer ids, and a stop predicate that closes over them. While those were locals in one long function, that state and the loop were forced to share a lifetime by construction. As an object they still share one, but it is now the object’s, and a caller can hold it across steps instead of being obliged to run the loop to the end in a single call.

The prefill preceding the first step also produces a token on most backbones, so a session is primed once before it is stepped &#8212; see primeFromPrefill().

Note

Calls on the same runtime must be externally serialized. An accidental overlap with another handleRequest() is rejected before runtime or response state is mutated; this is not a general thread-safety guarantee.

Param request:

Generation request with prompts and parameters

Param response:

Output response with generated tokens and text

Param stream:

CUDA stream

Throws std::runtime_error:

if an LLM or CUDA operation fails

Return:

True on success, false on failure

Public Functions

GenerationSession(
LLMRankRuntime &runtime,
DecodingInferenceContext &context,
DecodingStrategy &strategy,
ManagedKVCacheRequest *managedRequest,
LLMGenerationRequest const &request,
DecodingKvHeadroom const &kvHeadroom,
cudaStream_t stream,
bool boundarySchedulingActive
)
~GenerationSession()

Clears the stop predicate the context holds, which closes over members of this object.

GenerationSession(GenerationSession const&) = delete
GenerationSession &operator=(GenerationSession const&) = delete
bool primeFromPrefill()

Consume the token prefill produced, before any decode step runs.

Prefill emits a token on every backbone except the diffusion one, and that token has to travel the same cancel/decode/finalize/emit path a decode step would give it. Skipping this drops the first token of every request.

bool finished() const

True once every slot has reached a terminal state, or none are left.

bool advance()

Advance every live slot by one decode step.

virtual AdmitDecision admitSequence(SlotSeed seed) override

Join one new sequence to this running batch and prefill it, without touching the sequences in flight.

Admission plus seating: the context grows a slot (appendSlot), the session’s own per-slot state grows with it, and prefillSlotInPlace runs the new slot’s prompt as a seated batch-1 pass. On a failed prefill the slot is marked terminal with kError so the next eviction files its result; the batch’s other sequences are unaffected either way.

Whether the arriving request may share this batch at all (sampling parameters, adapter, step budget) is BatchCompatibility’s question, answered before a seed is built.

Throws:

std::runtime_error – for seeds appendSlot rejects, and for deployments prefillSlotInPlace refuses; nothing is modified on those paths.

virtual std::unordered_map<int32_t, BatchResult> takeCompletedAtOrAbove(
int32_t firstIndex
) override

Move out the results of finished sequences whose original index is >= firstIndex.

Sequences admitted mid-flight carry indices above the founding request’s range, so this is how their results leave the batch without appearing in the founding caller’s response. Harvested at every boundary by whoever drives the loop.

virtual AdmitDecision admitRequest(
LLMGenerationRequest const &request,
int32_t originalIndex,
RequestId requestId
) override

Admit a caller-level request: tokenization included, so a scheduler never builds a SlotSeed by hand or holds a tokenizer.

Whether the request may share this batch at all &#8212; sampling parameters, adapter, budgets &#8212; is the scheduler’s question (BatchCompatibility), answered before calling this.

Parameters:

request – A single-sequence request. Its stream channel, stop strings and logit bias ride along; its sampling parameters are ignored in favour of the batch’s. Scheduler policy keeps media requests founder-only until live admission has dedicated coverage.

Throws:

std::runtime_error – for an invalid request or an unsupported media deployment; nothing is modified.

virtual LLMGenerationResponse materializeResult(
BatchResult const &result,
std::vector<std::string> const &stopStrings
) const override

Turn one finished sequence’s result into the caller-facing response, decoding included, exactly as the founding request’s own assembly would have.

Lives here because the conversion needs the tokenizer and the stop-string trimming that the runtime owns; a scheduler holding raw BatchResults would either skip the text or grow its own copy of both.

inline virtual int32_t residentCount() const override

Sequences currently resident, finished or not. Admission capacity is judged on this.

struct SteppedGeneration

Everything one request owns between beginGeneration and finishGeneration: the same state handleRequest used to keep on its stack, packaged so a stepped control plane can hold a request open across ticks. The founding request must outlive this object; member order is load-bearing (the stream finalizer references the context and must be destroyed first).

Public Functions

inline explicit SteppedGeneration(
std::atomic<bool> &activeFlag,
Tensor &hostTokenStorage
) noexcept
inline ~SteppedGeneration() noexcept
SteppedGeneration(SteppedGeneration const&) = delete
SteppedGeneration &operator=(SteppedGeneration const&) = delete
inline ManagedKVCacheRequest *managedRequest() noexcept

Public Members

InProgressGuard guard
DecodingInferenceContext context
DecodingStrategy *strategy = {nullptr}
std::optional<ManagedKVCacheRequest> managedKVCacheRequest
DecodingKvHeadroom kvHeadroom = {}
std::optional<StreamChannelFinalizer> streamFinalizer
bool enableSpecDecode = {false}
bool hasActionRequest = {false}
struct InProgressGuard

Releases the runtime’s one-request-at-a-time latch, whatever path retires the request.

Public Functions

inline explicit InProgressGuard(
std::atomic<bool> &active
) noexcept
inline ~InProgressGuard() noexcept

Public Members

std::atomic<bool> &mActive
struct InProgressGuard

Releases the runtime’s one-request-at-a-time latch, whatever path retires the request.

Public Functions

inline explicit InProgressGuard(
std::atomic<bool> &active
) noexcept
inline ~InProgressGuard() noexcept

Public Members

std::atomic<bool> &mActive