mirror of
https://github.com/NVIDIA/TensorRT-LLM.git
synced 2026-01-14 06:27:45 +08:00
* add MNNVL memory mapping support Signed-off-by: Dongxu Yang <78518666+dongxuy04@users.noreply.github.com> * add more MPI environment for trtllm-llmapi-launch Signed-off-by: Dongxu Yang <78518666+dongxuy04@users.noreply.github.com> * add MoE communication and prepare kernels Signed-off-by: Dongxu Yang <78518666+dongxuy04@users.noreply.github.com> * add MNNVL AlltoAll support for DeepSeekV3 Signed-off-by: Dongxu Yang <78518666+dongxuy04@users.noreply.github.com> * add output dump for throughput benchmark Signed-off-by: Dongxu Yang <78518666+dongxuy04@users.noreply.github.com> * support dynamic kernel launch grid Signed-off-by: Dongxu Yang <78518666+dongxuy04@users.noreply.github.com> * address review comments Signed-off-by: Dongxu Yang <78518666+dongxuy04@users.noreply.github.com> * address review comments #2 Signed-off-by: Dongxu Yang <78518666+dongxuy04@users.noreply.github.com> --------- Signed-off-by: Dongxu Yang <78518666+dongxuy04@users.noreply.github.com> |
||
|---|---|---|
| .. | ||
| allgatherOp.cpp | ||
| allreduceOp.cpp | ||
| attentionOp.cpp | ||
| CMakeLists.txt | ||
| convertSpecDecodingMaskToPackedMaskOp.cpp | ||
| cublasScaledMM.cpp | ||
| cutlassScaledMM.cpp | ||
| deepseekAllreduceFusionOp.cpp | ||
| dynamicDecodeOp.cpp | ||
| dynamicDecodeOp.h | ||
| fmhaPackMaskOp.cpp | ||
| fp4BatchedQuantize.cpp | ||
| fp4BlockScaleMoe.cpp | ||
| fp4Gemm.cpp | ||
| fp4GemmTrtllmGen.cpp | ||
| fp4Op.cpp | ||
| fp4Quantize.cpp | ||
| fp8BatchedGemmTrtllmGen.cpp | ||
| fp8BlockScaleMoe.cpp | ||
| fp8BlockScalingGemm.cpp | ||
| fp8Op.cpp | ||
| fp8Quantize.cpp | ||
| fusedTopkSoftmax.cpp | ||
| gatherTreeOp.cpp | ||
| logitsBitmaskOp.cpp | ||
| loraOp.cpp | ||
| mambaConv1dOp.cpp | ||
| moeCommOp.cpp | ||
| moeOp.cpp | ||
| mtpOp.cpp | ||
| ncclCommunicatorOp.cpp | ||
| ncclCommunicatorOp.h | ||
| noAuxTcOp.cpp | ||
| parallelDecodeKVCacheUpdateOp.cpp | ||
| redrafterCurandOp.cpp | ||
| reducescatterOp.cpp | ||
| relativeAttentionBiasOp.cpp | ||
| selectiveScanOp.cpp | ||
| thUtils.cpp | ||
| thUtils.h | ||
| userbuffersFinalizeOp.cpp | ||
| userbuffersTensor.cpp | ||
| userbuffersTensor.h | ||
| weightOnlyQuantOp.cpp | ||