Gptq Marlin Repack#
- cudaError_t trt_edgellm::kernel::launchGptqMarlinRepackSourceBatch(
- int32_t const *const *firstQweights,
- int32_t const *const *secondQweights,
- int32_t const *const *firstQzeros,
- int32_t const *const *secondQzeros,
- int32_t *marlinOutput,
- int32_t count,
- int32_t projectionN,
- int32_t K,
- int32_t numGroups,
- int32_t groupSize,
- int32_t zeroPointOffset,
- cudaStream_t stream
GPU GPTQ → Marlin repack for MoE experts (vLLM
gptq_marlin_repackstyle). ! ! Input: GPTQqweight[E, K/8, N]int32 (8 INT4 values packed along K). ! Optionalqzeros[E, G, N/8]int32 — when non-null, remaps to Marlin’s !(q - 8) * scaleconvention viaq' = clamp(q - z - zpOffset + 8, 0, 15). ! Output: Marlin[E, K/16, 2*N]int32 in the Int4MoePlugin layout. ! ! RequiresK % 16 == 0,N % 64 == 0. Whenqzerosis non-null, !G * groupSize == K. ! ! Algorithm mirrors vLLMcsrc/.../marlin/gptq_marlin_repack.cu(packed GPTQ ! → Marlin on device, no host nibble expand). Packing indices match Edge-LLM !pack_int4_awq_marlin/marlinPackSwizzle. cudaError_t launchGptqMarlinRepack(int32_t const* dQweightE_K8_N, int32_t const* dQzerosE_G_N8_orNull, int32_t* dMarlin,
int32_t E, int32_t N, int32_t K, int32_t numGroups, int32_t groupSize, int32_t zeroPointOffset,
cudaStream_t stream);
! Repack non-contiguous expert projections directly from CUDA-mapped ! checkpoint storage.
secondQweightsis null for a single projection.