mirror of
https://github.com/NVIDIA/TensorRT-LLM.git
synced 2026-02-16 07:53:55 +08:00
Signed-off-by: Harris Nover <249353502+hnover-nv@users.noreply.github.com> Co-authored-by: Claude Sonnet 4.5 <noreply@anthropic.com> |
||
|---|---|---|
| .. | ||
| cubin | ||
| decoderXQAImplJIT | ||
| instantiation | ||
| CMakeLists.txt | ||
| copy_cu.py | ||
| decoderMaskedMultiheadAttentionLaunch.h | ||
| decoderMaskedMultiheadAttentionTemplate.h | ||
| decoderXQAConstants.h | ||
| decoderXQAImpl.cpp | ||
| decoderXQAImpl.h | ||
| decoderXQAImplCommon.cpp | ||
| decoderXQAImplCommon.h | ||
| decoderXQAImplPrecompiled.cpp | ||
| decoderXQAImplPrecompiled.h | ||
| decoderXQARunner.cpp | ||
| decoderXQARunner.h | ||
| mmha_notes.md | ||
| tensorMapUtils.cpp | ||
| tensorMapUtils.h | ||
| xqaParams.h | ||