Runtime Coordinator#

class RuntimeCoordinator#

Owns local LLM rank workers and communication groups for a runtime coordinator.

Public Functions

explicit RuntimeCoordinator(Config config)#
~RuntimeCoordinator()#
RuntimeCoordinator(RuntimeCoordinator const&) = delete#
RuntimeCoordinator &operator=(RuntimeCoordinator const&) = delete#
RuntimeCoordinator(RuntimeCoordinator&&) = delete#
RuntimeCoordinator &operator=(RuntimeCoordinator&&) = delete#
bool captureDecodingCUDAGraph(cudaStream_t stream = nullptr)#
bool supportsBoundaryScheduling() const noexcept#

True when this coordinator can host in-flight admission: the single-rank inline path, or thread-launched tensor parallelism where the boundary-decision relay can cross ranks by shared memory. MPI-launched ranks live in other processes and are not covered.

bool supportsSteppedExecution() const noexcept#

True when this coordinator can hand out stepped requests: inline single-rank execution only in this release; a tensor-parallel deployment answers false.

std::unique_ptr<SteppedExecution> beginStepped(
LLMGenerationRequest const &request,
RequestId requestId,
bool enableProfiling,
cudaStream_t stream
)#

Open one request under the stepped control plane. Prepares request state, runs the founding prefill, and returns the handle the scheduler drives tick by tick; null on the refusals dispatchRequest would have reported as failure.

bool dispatchRequest(
LLMGenerationRequest const &request,
bool enableProfiling,
bool outputThinkerEmbeddings = false,
cudaStream_t stream = nullptr,
GenerationBoundaryHook const &boundaryHook = {}
)#
bool genAndSaveSystemPromptKVCache(
std::string const &prompt,
std::string const &loraWeightsName,
cudaStream_t stream = nullptr
)#
std::vector<int32_t> countPromptTokens(
LLMGenerationRequest const &request
) const#

Return the token count produced by the same preparation path used for inference. Only text requests are supported.

void setVisualPrunerConfig(VisualPrunerConfig const &config)#
LLMGenerationResponse takeRankResponse(int32_t globalRank)#
LLMRankRuntime &rootRuntime()#
LLMRankRuntime const &rootRuntime() const#
bool ownsGlobalRank(int32_t globalRank) const noexcept#
bool localRanksSucceeded() const noexcept#
inline int32_t worldSize() const noexcept#

Ranks this plan spans. 1 for a single-device plan.

struct Config#

Public Members

std::string engineDir#
std::string multimodalEngineDir#
std::unordered_map<std::string, std::string> loraWeightsMap#
ParallelConfig parallelConfig#
std::optional<SpecDecodeDraftingConfig> draftingConfig#
ContextCacheConfig contextCacheConfig#
std::string checkpointDir#
std::string draftCheckpointDir#
std::vector<int32_t> localRanks#
std::unordered_map<int32_t, int32_t> localRankDevices#
std::vector<cudaStream_t> localStreams#
bool ownsLocalStreams = {true}#
std::vector<ParallelBackendHandles> backendHandles#
std::unique_ptr<ModelArtifacts> modelArtifacts#
struct Config

Public Members

std::string engineDir
std::string multimodalEngineDir
std::unordered_map<std::string, std::string> loraWeightsMap
ParallelConfig parallelConfig
std::optional<SpecDecodeDraftingConfig> draftingConfig
ContextCacheConfig contextCacheConfig
std::string checkpointDir
std::string draftCheckpointDir
std::vector<int32_t> localRanks
std::unordered_map<int32_t, int32_t> localRankDevices
std::vector<cudaStream_t> localStreams
bool ownsLocalStreams = {true}
std::vector<ParallelBackendHandles> backendHandles
std::unique_ptr<ModelArtifacts> modelArtifacts