Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
241 commits
Select commit Hold shift + click to select a range
63aac74
[None][chore] Bump version to 1.2.0 (#10686)
yiqingy0 Jan 15, 2026
bc5094c
[None][chore] Setup the code review rule on release/1.2 branch (#10694)
yiqingy0 Jan 15, 2026
fe65847
[None][infra] Waive failed cases for release branch on 01/16 (#10748)
EmmaQiaoCh Jan 16, 2026
22b5f0c
[None][fix] Disable short profile for tunable ops with MERGE strategy…
hyukn Jan 16, 2026
06f7970
[https://nvbugs/5800521][ci] Move test_openai_chat_guided_decoding to…
syuoni Jan 16, 2026
3f98409
[None][doc] 1.2 Release Notes Headers (#10722)
pcastonguay Jan 16, 2026
e87a406
[https://nvbugs/5669671][fix] Support GuidedDecoder with sharded logi…
syuoni Jan 16, 2026
c772f3a
[https://nvbugs/5803813][fix] Fix llama 4 min latency (#10724)
mikeiovine Jan 16, 2026
18fe915
[TRTLLM-5366][chore] Add dgx-spark beta notes (#10766)
farazkh80 Jan 17, 2026
a48466f
[None][infra] Update upgrade related docs for release 1.2 (#10760)
EmmaQiaoCh Jan 17, 2026
6a8f18e
[None][infra] Waive failed cases for release on 10/18 (#10781)
EmmaQiaoCh Jan 18, 2026
368cb15
[None][doc] update doc (add minimax model) (#10749)
jmydurant Jan 19, 2026
83be9bb
[None][fix] Fix tmp dir being deleted too early in unit test. (#10741)
hyukn Jan 19, 2026
17f419f
[https://nvbugs/5811697][fix] Fix buffer reuse. (#10716)
yuxianq Jan 19, 2026
bc712a0
[https://nvbugs/5782112][fix] Cherry-pick #10633: Fix hanging issue f…
hyukn Jan 19, 2026
6185464
[None][infra] Waive failed case for release branch on 01/19 (#10795)
EmmaQiaoCh Jan 19, 2026
460b8a5
[None][test] modify ctx config in 128k8k disagg cases (#10779)
ruodil Jan 19, 2026
e5ade20
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Jan 19, 2026
c51c75c
[None][test] Update test case for release (#10763)
crazydemo Jan 19, 2026
b979a02
[https://nvbugs/5748664][fix] Increasing disagg acc test timeout (#10…
pcastonguay Jan 19, 2026
6c0b080
[https://nvbugs/5791242][fix] workaround for flashinfer.sampling.samp…
ixlmar Jan 20, 2026
ed6df35
[https://nvbugs/5636916][fix] Fix accuracy issue of TWOSHOT AllReduce…
hyukn Jan 20, 2026
307ea14
[None][infra] Waive failed cases for release branch on 01/20 (#10828)
EmmaQiaoCh Jan 20, 2026
87ff718
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Jan 20, 2026
89bb6f3
[None][chore] Reduce tedious logs (#10819)
chzblych Jan 20, 2026
837579c
[None][test] Update case for release (#10811)
crazydemo Jan 21, 2026
226274a
[None][fix] Fix the potential access issue of cache operations betwee…
hyukn Jan 21, 2026
4791ee7
[None][chore] Revert NVIDIA/TensorRT-LLM#10819 (#10870)
chzblych Jan 21, 2026
07a2878
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Jan 21, 2026
7a3f264
[https://nvbugs/5814203][fix] Fix port 8000 being used issue in stres…
dominicshanshan Jan 21, 2026
8bf736a
[https://nvbugs/5747938][infra] Unwaive trtllm serve example test (#1…
LinPoly Jan 21, 2026
4e38b99
[https://nvbugs/5754977][fix] Use free port for serve test (#10878)
JunyiXu-nv Jan 21, 2026
0362a42
[https://nvbugs/5740377][fix] Prevent out-of-bounds read (#10868)
HuiGao-NV Jan 22, 2026
8ecfc29
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Jan 22, 2026
42c2a41
[https://nvbugs/5769425][fix] add syncthreads for tinygemm to resolve…
dc3671 Jan 22, 2026
9488cbc
[https://nvbugs/5821433][fix] fix test_auto_scaling for 2 GPUs (#10866)
reasonsolo Jan 22, 2026
8cc475c
[https://nvbugs/5779536][fix] Unwaive Llama 3.3 related multi GPU tes…
pengbowang-nv Jan 22, 2026
7539452
[https://nvbugs/5826962][fix] Fix PD disaggregation for VLMs that use…
2ez4bz Jan 22, 2026
b7011a8
[https://nvbugs/5691730][fix] Have LoRa bf16 ckpts work with Llama 3.…
moraxu Jan 22, 2026
d36bd45
[https://nvbugs/5814409][fix] fix pp loop hang because of i-sending n…
reasonsolo Jan 22, 2026
40a8b2a
[None][fix] Always reset drafting states for GuidedDecoder (#10899)
syuoni Jan 22, 2026
6433764
[https://nvbugs/5814247][fix] AutoDeploy: skip mxfp4_moe test unless …
lucaslie Jan 22, 2026
13103af
[https://nvbugs/5769712][fix] fix timeout in AutoDeploy llama accurac…
lucaslie Jan 22, 2026
7f11072
[https://nvbugs/5784543][chore] unwaive test. (#10906)
yuxianq Jan 23, 2026
8cf294a
[https://nvbugs/5701445][chore] unwaive tests. (#10913)
yuxianq Jan 23, 2026
e1d61d1
[https://nvbugs/5748600][ci] Update guided decoding waive list (#10904)
syuoni Jan 23, 2026
00856de
[https://nvbugs/5819021][fix] unwaive some tests (#10836)
byshiue Jan 23, 2026
4689f83
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Jan 23, 2026
27feb0c
[https://nvbugs/5680911][fix] Remove @cache decorator to enhance CI s…
zheyuf Jan 23, 2026
9205a4d
[https://nvbugs/5814309][fix] Use NCCL as fallback to avoid crash due…
hyukn Jan 23, 2026
ea4b8c9
[None][test] Update test list (#10883)
crazydemo Jan 23, 2026
133076a
[https://nvbugs/5833795][chore] Waive test test_e2e.py::test_ptp_quic…
yihwang-nv Jan 23, 2026
09e4a94
[https://nvbugs/5741304][chore] Update flashinfer-python to 0.6.1 (#1…
yihwang-nv Jan 23, 2026
cab453f
[https://nvbugs/5814914][fix] Fix llama sm120 spec dec (#10765)
mikeiovine Jan 23, 2026
571521b
[None][fix] Fix MTP 1-model sampler (#10369)
mikeiovine Jan 23, 2026
8159925
[None][ci] Remove long-running sanity check tests on GH200 (#10924)
chzblych Jan 24, 2026
02cb227
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Jan 24, 2026
01ba1a1
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Jan 24, 2026
fe27e9a
[https://nvbugs/5804146][fix] Enable responses tests and remove ds to…
JunyiXu-nv Jan 24, 2026
da7d1d3
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Jan 25, 2026
6d6d0fb
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Jan 25, 2026
829170a
[https://nvbugs/5779536][fix] Unwaive DeepSeekR1 nvfp4 pp4 mtp test c…
pengbowang-nv Jan 25, 2026
f048ade
[https://nvbugs/5829097][fix] Re-init TRTLLM sampler to use sample st…
yuxianq Jan 26, 2026
8f4b8f2
[https://nvbugs/5769890][fix] enable system memory to transfer active…
yuxianq Jan 26, 2026
453b7d1
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Jan 26, 2026
11f2081
[None][feat] Cherry-pick #10335: Use XQA JIT impl by default and miti…
pengbowang-nv Jan 26, 2026
57a4fb9
[#10614][fix] gpt_oss first iteration streaming in trtllm-serve (#10884)
LinPoly Jan 26, 2026
4b319f0
[TRTLLM-9581][infra] Use /home/scratch.trt_llm_data_ci in computelab …
ZhanruiSunCh Jan 26, 2026
15cb430
[None][infra] Waive failed cases for release branch on 01/26 (#10999)
EmmaQiaoCh Jan 26, 2026
cfddd21
[https://nvbugs/5826689][fix] replace etcd3 with etcd-sdk-python (#10…
reasonsolo Jan 26, 2026
59c8225
[https://nvbugs/5769815][fix] Fix offset calculation in _are_stop_wor…
stnie Jan 26, 2026
fa0b317
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Jan 27, 2026
0f0787b
[None][test] Fix missing test cases (#10881)
yufeiwu-nv Jan 27, 2026
bb56001
[None][feat] support Lyris GB200 and increase disagg test timeout (#1…
yingguo-trt Jan 27, 2026
8802eaf
[https://nvbugs/5800646][fix] Fix hang issue by avoid exposing UB buf…
liji-nv Jan 27, 2026
e2957d6
[None][chore] Unwaive helix tests (#11008)
brb-nv Jan 27, 2026
ce36c8a
[https://nvbugs/5835925][fix] Add EPD disagg support for Qwen3 VL MoE…
2ez4bz Jan 28, 2026
6dcac08
[https://nvbugs/5819452][ci] Unwaive LLaMA2 7B FP8 case (#10997)
syuoni Jan 28, 2026
b8318b2
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Jan 28, 2026
96c0ae0
[https://nvbugs/5811087][chore] Unwaive Gemma3 27B multimodal test (#…
brb-nv Jan 28, 2026
219e5db
[https://nvbugs/5821433][fix] WAR for popen in QA env (#10989)
reasonsolo Jan 28, 2026
dde9545
[None][infra] Waive failed case for release on 1/28 (#11055)
EmmaQiaoCh Jan 28, 2026
4ebe56f
[https://nvbugs/5839569][test] update test constraint (#11054)
crazydemo Jan 28, 2026
18c4ff6
[None][infra] cherry pick lock file fix to release/1.2 (#10975)
yuanjingx87 Jan 28, 2026
bbc6462
[TRTLLM-10669][fix] Fix Eagle3 draft model weight loading for through…
cascade812 Jan 28, 2026
4f5c9ce
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Jan 29, 2026
1bb0ead
[None][infra] Fix TRT-LLM data scratch mount point for gb10x (#10880)…
EmmaQiaoCh Jan 29, 2026
ffaa62b
[None][chore] unwaive qwen3 235B accuracy test (#11058)
kris1025 Jan 29, 2026
1519191
[None][doc] Hardware support update (#10719)
pcastonguay Jan 29, 2026
8e3d1be
[https://nvbugs/5829830][fix] Declare the var in the correct scope (#…
ziyixiong-nv Jan 30, 2026
569947a
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Jan 30, 2026
6f7ff1b
[https://nvbugs/5815136][fix] Cherry-pick #11042: nccl symmetric with…
hyukn Jan 30, 2026
fe1ea30
[None][feat] Add documentation on configuring CPU affinity in TRT-LLM…
dhansen-nvidia Jan 30, 2026
3d3b30c
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Jan 31, 2026
66c017f
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Feb 1, 2026
8d6bb0a
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Feb 2, 2026
0fa4d4e
[https://nvbugs/5823465][fix] Add CUTEDSL moe backend for deepseek r1…
dominicshanshan Feb 2, 2026
a459360
[https://nvbugs/5787904][fix] update mig tests (#11014)
xinhe-nv Feb 2, 2026
272fee9
[https://nvbugs/5819444][fix] Unwaive gpt-oss test (#10927)
LinPoly Feb 2, 2026
59e30a6
[None][infra] Waive failed cases for release branch on 02/02 (#11182)
EmmaQiaoCh Feb 2, 2026
343b608
[https://nvbugs/5739981][fix] unwaive tests using opt-125M (#11099)
ixlmar Feb 2, 2026
c4b2bbc
[None][chore] Add warning about 2-model MTP deprecation (#11043)
mikeiovine Feb 2, 2026
ed3b831
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Feb 3, 2026
41656da
[https://nvbugs/5854419][fix] Fix Qwen3-VL-Dense/MoE accuracy drop (#…
yechank-nvidia Feb 3, 2026
dcff503
[https://nvbugs/5761391][fix] Cherry-pick #10471: Include triton-kern…
anish-shanbhag Feb 3, 2026
cf67667
[TRTLLM-10803][fix] Cherry-pick of #11200: Fix mocking of HuggingFace…
anish-shanbhag Feb 4, 2026
5a36856
[https://nvbugs/5815025][fix] Fix spec-dec mode flag and related cpp …
pengbowang-nv Feb 4, 2026
a3a293b
[TRTLLM-8425][doc] Update sampling documentation (#10083) (#11270)
stnie Feb 4, 2026
adb133b
[https://nvbugs/5821433][fix] complete WAR for popen in QA env (#11214)
crazydemo Feb 5, 2026
2ffc068
[None][chore] Pass without_comm to cutlass and deepgemm (#11245)
xxi-nv Feb 5, 2026
b60949f
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Feb 5, 2026
7f4a4cd
[None][chore] Fix slurm job name (#11265)
yingguo-trt Feb 5, 2026
baa2abf
[https://nvbugs/5845769][fix] B300(sm103) support on VLMs (#11274)
yechank-nvidia Feb 5, 2026
46b890a
[https://nvbugs/5830877][fix] Use the best (correct) config for GPTOS…
dongfengy Feb 5, 2026
adb8bcb
[https://nvbugs/5826890][fix] Warm-up before disagg benchmarking (Che…
bo-nv Feb 5, 2026
1d875d8
[https://nvbugs/5863443][fix] Fix message truncation in Helix CP cach…
brb-nv Feb 5, 2026
8e01450
[https://nvbugs/5688721][fix] AutoDeploy: unwaive fixed NemotronH acc…
lucaslie Feb 5, 2026
b6f2582
[TRTLLM-10118][fix] Fix vulnerabilities urllib3 nbconvert jaraco-cont…
yiqingy0 Feb 6, 2026
60d53fb
[None][chore] Resolve a conflict in the md file (#11255)
ziyixiong-nv Feb 6, 2026
2027b55
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Feb 7, 2026
a22f63a
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Feb 8, 2026
82bfc5f
[https://nvbugs/5831976][chore] Move test_trtllm_flashinfer_symbol_co…
yihwang-nv Feb 10, 2026
78544d6
[https://nvbugs/5624818][fix] Fix GPT-OSS with non-paged_context_fmha…
pengbowang-nv Feb 10, 2026
daabeab
[https://nvbugs/5814504][fix] Add skip_pre_hopper flag on NVILA & Nan…
yechank-nvidia Feb 10, 2026
7167f2b
[None][infra] Disable release spark stage due to migration of spark c…
EmmaQiaoCh Feb 10, 2026
5038e96
[None][infra] Enable spark stage for release since the spark cloud mi…
EmmaQiaoCh Feb 10, 2026
57c1ecf
[https://nvbugs/5820922][perf] Improve TorchSampler performance by re…
stnie Feb 11, 2026
c679cb5
[None][infra] Pin the version for torchao (#11446)
EmmaQiaoCh Feb 11, 2026
d3b26dc
[https://nvbugs/5889564][fix] fix kwargs name (#11496)
reasonsolo Feb 13, 2026
3cddddd
[https://nvbugs/5833795][fix] Remove test waive and try CI (#11464)
dongfengy Feb 13, 2026
de294fc
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Feb 14, 2026
3cc9e47
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Feb 15, 2026
d1e5956
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Feb 15, 2026
0149e89
[https://nvbugs/5860137][fix] Adjust deepgemm tuning buckets to cover…
dc3671 Feb 15, 2026
1c207c1
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Feb 16, 2026
4bcc441
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Feb 16, 2026
aa4c677
[https://nvbugs/5875296][fix] Fix TritonMOE test for Qwen3_30B_A3B (#…
dongfengy Feb 16, 2026
fbda477
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Feb 17, 2026
cf1b00f
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Feb 17, 2026
4a110dd
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Feb 17, 2026
261627c
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Feb 18, 2026
c426c49
[None][infra] Cherry pick plc pipeline for 1.2 (#11546)
yuanjingx87 Feb 18, 2026
82f6878
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Feb 19, 2026
7d08591
[https://nvbugs/839137][fix] Unwaive disagg unexpected ucx error (#11…
pcastonguay Feb 19, 2026
5006b5f
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Feb 20, 2026
5bd8661
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Feb 21, 2026
1d229e7
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Feb 21, 2026
73db5c0
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Feb 22, 2026
8e9c39a
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Feb 22, 2026
275f615
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Feb 23, 2026
fe91481
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Feb 23, 2026
c5127f6
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Feb 23, 2026
4e96ed1
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Feb 23, 2026
48c03d3
[https://nvbugs/5823783][fix] Fix multi-node trust_remote_code hang i…
JunyiXu-nv Feb 23, 2026
c42ba80
[None][infra] Waive failures on release 1.2 (#11639)
jieli-matrix Feb 23, 2026
4f6acbb
[None][chore] Fix gpu memory requirement in stress test (#11404)
dominicshanshan Feb 23, 2026
1510a14
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Feb 24, 2026
c9a6df9
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Feb 24, 2026
08318ab
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Feb 25, 2026
172ab2a
[https://nvbugs/5839155][test] Unwaive DeepSeekR1 fp8_blockscale thro…
kaiyux Feb 26, 2026
ea5d0b5
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Feb 26, 2026
00de565
[https://nvbugs/5809169][unwaive] Unwaive TestGPTOSS test (#11416)
peaceh-nv Feb 26, 2026
dd89617
[https://nvbugs/5859881][fix] Unwaive test (#11716)
hyukn Feb 26, 2026
6c5f1da
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Feb 26, 2026
aa5e1d9
[https://nvbugs/5799917][fix] Recover from CUTLASS MoE doActivation p…
rosenrodt Feb 26, 2026
11dba1b
[None][feat] add sanity tests for release1.2 version (#11738)
yingguo-trt Feb 26, 2026
6225e34
[https://nvbugs/5889841][fix] Add custom option class to allow subcom…
FrankD412 Feb 26, 2026
95e3682
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Feb 27, 2026
98c7afd
[None][chore]: Add waives for nvbug 5936273 and 5936322 (#11775)
jieli-matrix Feb 27, 2026
1648ce6
[https://nvbugs/5875522][docs] Add known issue for disaggregated serv…
Tabrizian Feb 27, 2026
76f011e
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Feb 28, 2026
c693d41
[https://nvbugs/5756028][fix] Fix VSWA initialization with spec-dec a…
cascade812 Feb 28, 2026
06e6ef6
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Mar 1, 2026
ae32812
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Mar 1, 2026
7a6d551
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Mar 2, 2026
db3af40
[https://nvbugs/5775256] [fix] Reopen fp8_dsl_fused_moe ut. (#11779)
limin2021 Mar 2, 2026
9b0c020
[TRTLLM-11135][fix] Fix vulnerabilities protobuf (#11702)
yiqingy0 Mar 2, 2026
a4ee00f
[https://nvbugs/5762822][chore] Unwaive longbenchV2 test (#11647)
heyuhhh Mar 2, 2026
6c542e9
[https://nvbugs/5823212][fix] Warmup maybe_compiled_cat in forward_co…
yuantailing Mar 2, 2026
76dd900
[https://nvbugs/5747920][bug] Cherry pick 11296 from main (#11771)
yechank-nvidia Mar 3, 2026
527ce26
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Mar 3, 2026
96c09fb
[TRTLLM-11176][fix] Security Issue Fix cherry pick (#11683)
yibinl-nvidia Mar 4, 2026
c8f3331
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Mar 4, 2026
3768b35
[TRTLLM-11135][fix] Fix vulnerability aiohttp (#11778)
yiqingy0 Mar 4, 2026
9c53176
[https://nvbugs/5936273][fix] Fix bugs of Mistral Large3 (#11885)
byshiue Mar 4, 2026
f06eaaa
[https://nvbugs/5949098][doc] Fixing docs links (#11912)
pcastonguay Mar 4, 2026
9f32f48
[None][doc] Replace the TensorRT-LLM with TensorRT LLM (#11914)
nv-guomingz Mar 5, 2026
3d97567
[None][chore] Fix/disagg perf failure detection (#11904)
yingguo-trt Mar 5, 2026
36e34ee
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Mar 5, 2026
14bf96b
[None][test] cherry-pick: add concurrency override and fix for 128k8k…
ruodil Mar 6, 2026
df5d831
[None][infra] Waive 3 failed cases for release/1.2 in post-merge 40 (…
ZhanruiSunCh Mar 6, 2026
e8b1956
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Mar 6, 2026
3429ad6
[None][infra] Waive 2 failed cases for release/1.2 in post-merge 42 (…
ZhanruiSunCh Mar 6, 2026
5e23368
[https://nvbugs/5948878][fix] Fix ClientPayloadError (#11973)
yingguo-trt Mar 6, 2026
8e77388
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Mar 7, 2026
e691be9
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Mar 8, 2026
a19f0f0
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Mar 9, 2026
02e5f84
[None][doc] Release notes for 1.2 release (#11955)
pcastonguay Mar 9, 2026
4988ba0
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Mar 9, 2026
d9cbb34
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Mar 10, 2026
c15f57d
[None][test] Fix disagg test sku for release 1.2 (#12066)
fredricz-20070104 Mar 10, 2026
484f4fe
[https://nvbugs/5924136][fix] Fix bug by add env var (#11974)
benzh-2025 Mar 10, 2026
51f5ef3
[None][test] Fix disagg test gpu (#12078)
fredricz-20070104 Mar 10, 2026
21796cd
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Mar 11, 2026
5bcd108
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Mar 12, 2026
dbdb13b
[TRTLLM-9911] [doc] Update Perf-Overview.md for Release 1.2 (#12098)
zbpatel Mar 12, 2026
9647698
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Mar 13, 2026
0a141c0
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Mar 14, 2026
e5fe628
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Mar 15, 2026
75c8de8
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Mar 16, 2026
7321457
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Mar 17, 2026
bd3a0c1
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Mar 18, 2026
0dd9181
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Mar 19, 2026
78f820a
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Mar 20, 2026
31e8fe5
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Mar 21, 2026
5ecb150
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Mar 22, 2026
15462b1
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Mar 23, 2026
45c819e
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Mar 24, 2026
be4ca51
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Mar 25, 2026
b30e906
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Mar 26, 2026
76ccc6e
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Mar 27, 2026
9e0572c
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Mar 28, 2026
e5c1a58
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Mar 30, 2026
a569b63
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Mar 31, 2026
704d036
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Apr 1, 2026
3c13d8c
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Apr 3, 2026
d607bf5
[None][infra] use public torch index as CI backup (#12261) (#12804)
niukuo Apr 8, 2026
4403635
[None][doc] Update release note 1.2.1 (#12825)
VALLIS-NERIA Apr 8, 2026
9532fed
[None][infra] Bump xgrammar (#12811)
yuanjingx87 Apr 8, 2026
52d7250
[None][infra] Bump version to 1.2.1 (#12755)
yuanjingx87 Apr 8, 2026
833aceb
[None][infra] Move B200 tests from lbd to computelabsc01 (#12874)
yuanjingx87 Apr 9, 2026
f461db0
[None][infra] Bump black and tornado (#12876)
yuanjingx87 Apr 9, 2026
bfb9d38
[None][infra] Bump etcd to 3.6.9 to involve grpc fix (#12594) (#12872)
niukuo Apr 9, 2026
ce0a11c
Revert "[None][infra] Bump black and tornado (#12876)" (#12894)
niukuo Apr 9, 2026
dbf33bb
[None][infra] Tag Image (#12925)
yuanjingx87 Apr 10, 2026
47c8c6b
[https://nvbugs/6025177][fix] Cherry-pick KV cache corruption fix to …
thorjohnsen Apr 12, 2026
220cdd7
[None][infra] More vulnerability fix for release/1.2.1 (#12904)
yuanjingx87 Apr 12, 2026
d05d608
[None][infra] Check in most recent lock file from nightly pipeline
tensorrt-cicd Apr 13, 2026
7d44e91
[https://nvbugs/6071380][fix] Update the invalid dynamo urls in doc. …
nv-guomingz Apr 15, 2026
376f7e1
[None][infra] Regenerate lock file to fix pygments vulnerability (#13…
yuanjingx87 Apr 16, 2026
ca1fa90
fix(test): pass explicit streaming=False in base_metrics_verification…
mc-nv Apr 20, 2026
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
The table of contents is too big for display.
Diff view
Diff view
  •  
  •  
  •  
2 changes: 1 addition & 1 deletion .devcontainer/docker-compose.override-example.yml
Original file line number Diff line number Diff line change
Expand Up @@ -5,4 +5,4 @@ services:
volumes:
# Uncomment the following lines to enable
# # Mount TRTLLM data volume:
# - /home/scratch.trt_llm_data/:/home/scratch.trt_llm_data/:ro
# - /home/scratch.trt_llm_data_ci/:/home/scratch.trt_llm_data_ci/:ro
8 changes: 7 additions & 1 deletion .github/CODEOWNERS
Original file line number Diff line number Diff line change
Expand Up @@ -219,6 +219,12 @@ docs/source/performance/perf-benchmarking.md @NVIDIA/trtllm-bench-reviewers
## Any changes to versions, additions, or removals of third-party libraries
/3rdparty/** @NVIDIA/trt-llm-oss-compliance

### Vendored Third-Party Code (triton-kernels)
## This is a temporary vendored copy of triton-kernels from the Triton project (MIT License).
## Do not accept contributions to this directory - it should only be updated via scripts/vendor_triton_kernels.py
## This can be removed if and when triton-kernels is published as a separate wheel.
/triton_kernels/** @NVIDIA/trt-llm-oss-compliance

### Docker & Installation Scripts
## These scripts install and pin dependency versions
/docker/common/** @NVIDIA/trt-llm-setup-infra-devs @NVIDIA/trt-llm-infra-devs @NVIDIA/trt-llm-oss-compliance
Expand All @@ -233,4 +239,4 @@ docs/source/performance/perf-benchmarking.md @NVIDIA/trtllm-bench-reviewers
# The rule below requires that any PR to release/**/* branches must be approved by at least one member
# of the NVIDIA/trt-llm-release-branch-approval team, regardless of who else approves the PR.
# Without approval from a member of this team, PRs cannot be merged to release branches.
# * @NVIDIA/trt-llm-release-branch-approval
* @NVIDIA/trt-llm-release-branch-approval
3 changes: 3 additions & 0 deletions .pre-commit-config.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -1369,6 +1369,9 @@ common-files: &common_files |
triton_backend/tools/whisper/client.py |
)$

# Global exclude pattern for vendored third-party code
exclude: '^triton_kernels/'

default_install_hook_types: [pre-commit, commit-msg]
repos:
- repo: https://github.com/pycqa/isort
Expand Down
9 changes: 7 additions & 2 deletions 3rdparty/CMakeLists.txt
Original file line number Diff line number Diff line change
Expand Up @@ -108,11 +108,16 @@ FetchContent_Declare(
SOURCE_SUBDIR
dont-add-this-project-with-add-subdirectory)

set(_patch_file "${CMAKE_CURRENT_SOURCE_DIR}/patches/xgrammar_constexpr.patch")
FetchContent_Declare(
xgrammar
GIT_REPOSITORY https://github.com/mlc-ai/xgrammar
GIT_TAG v0.1.25 # e4e816f5f0fe39f5b1601a17a4552307fa3b70ff
GIT_TAG v0.1.32 # 62e13551b9b63251114894c5ee638564b160dd48
GIT_SHALLOW TRUE
# NOTE: TensorRT-LLM only uses the headers
SOURCE_SUBDIR
dont-add-this-project-with-add-subdirectory)
dont-add-this-project-with-add-subdirectory
PATCH_COMMAND
bash -c "patch -p1 --forward --batch --dry-run -i '${_patch_file}' && \
patch -p1 --forward --batch -i '${_patch_file}' || \
echo 'Patch already applied, skipping.'")
19 changes: 19 additions & 0 deletions 3rdparty/patches/xgrammar_constexpr.patch
Original file line number Diff line number Diff line change
@@ -0,0 +1,19 @@
--- a/cpp/grammar_functor.cc
+++ b/cpp/grammar_functor.cc
@@ -1750,11 +1750,11 @@
void Apply(Grammar* grammar);
static std::optional<uint64_t> HashSequence(const Grammar& grammar, int32_t sequence_id);

- static const int16_t kNotEndStateFlag = -0x100;
- static const int16_t kEndStateFlag = -0x200;
- static const int16_t kSelfRecursionFlag = -0x300;
- static const int16_t kSimpleCycleFlag = -0x400;
- static const int16_t kUnKnownFlag = -0x500;
+ static constexpr int16_t kNotEndStateFlag = -0x100;
+ static constexpr int16_t kEndStateFlag = -0x200;
+ static constexpr int16_t kSelfRecursionFlag = -0x300;
+ static constexpr int16_t kSimpleCycleFlag = -0x400;
+ static constexpr int16_t kUnKnownFlag = -0x500;

private:
Grammar* grammar_;
42 changes: 36 additions & 6 deletions ATTRIBUTIONS-Python.md
Original file line number Diff line number Diff line change
Expand Up @@ -5261,7 +5261,7 @@ For more information, please refer to <http://unlicense.org>
- `Tracker`: https://github.com/tox-dev/py-filelock/issues


## flashinfer-python (0.3.1.post1)
## flashinfer-python (0.6.1)

### Licenses
License: `Apache-2.0`
Expand Down Expand Up @@ -33478,7 +33478,6 @@ limitations under the License.
- `Homepage`: https://github.com/NVIDIA/Model-Optimizer


<<<<<<< HEAD
## nvidia-modelopt-core (0.33.1)

### Licenses
Expand Down Expand Up @@ -33506,10 +33505,7 @@ limitations under the License.
- `Homepage`: https://github.com/NVIDIA/Model-Optimizer


## nvidia-nccl-cu12 (2.27.3)
=======
## nvidia-nccl-cu13 (2.27.7)
>>>>>>> fd7624b32 (modify ATTRIBUTIONS-Python.md)

### Licenses
License: `BSD-3-Clause`
Expand Down Expand Up @@ -62379,7 +62375,7 @@ Copyright 2018- The Hugging Face team. All rights reserved.
- `Homepage`: https://github.com/huggingface/transformers


## triton (3.5.0)
## triton (3.5.1)

### Licenses
License: `MIT License`
Expand Down Expand Up @@ -62417,6 +62413,40 @@ License: `MIT License`
- `Homepage`: https://github.com/triton-lang/triton/


## triton-kernels (3.5.1)

### Licenses
License: `MIT License`

- `LICENSE` (from triton repository root):
```
Copyright 2018-2020 Philippe Tillet
Copyright 2020-2022 OpenAI

Permission is hereby granted, free of charge, to any person obtaining
a copy of this software and associated documentation files
(the "Software"), to deal in the Software without restriction,
including without limitation the rights to use, copy, modify, merge,
publish, distribute, sublicense, and/or sell copies of the Software,
and to permit persons to whom the Software is furnished to do so,
subject to the following conditions:

The above copyright notice and this permission notice shall be
included in all copies or substantial portions of the Software.

THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT.
IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY
CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT,
TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE
SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
```

### URLs
- `Source`: https://github.com/triton-lang/triton/tree/v3.5.1/python/triton_kernels


## tritonclient (2.63.0)

### Licenses
Expand Down
6 changes: 3 additions & 3 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -8,9 +8,9 @@ state-of-the-art optimizations to perform inference efficiently on NVIDIA GPUs.<
[![Documentation](https://img.shields.io/badge/docs-latest-brightgreen.svg?style=flat)](https://nvidia.github.io/TensorRT-LLM/)
[![python](https://img.shields.io/badge/python-3.12-green)](https://www.python.org/downloads/release/python-3123/)
[![python](https://img.shields.io/badge/python-3.10-green)](https://www.python.org/downloads/release/python-31012/)
[![cuda](https://img.shields.io/badge/cuda-13.0.0-green)](https://developer.nvidia.com/cuda-downloads)
[![torch](https://img.shields.io/badge/torch-2.9.0-green)](https://pytorch.org)
[![version](https://img.shields.io/badge/release-1.2.0rc8-green)](https://github.com/NVIDIA/TensorRT-LLM/blob/main/tensorrt_llm/version.py)
[![cuda](https://img.shields.io/badge/cuda-13.1.0-green)](https://developer.nvidia.com/cuda-downloads)
[![torch](https://img.shields.io/badge/torch-2.9.1-green)](https://pytorch.org)
[![version](https://img.shields.io/badge/release-1.2.1-green)](https://github.com/NVIDIA/TensorRT-LLM/blob/main/tensorrt_llm/version.py)
[![license](https://img.shields.io/badge/license-Apache%202-blue)](https://github.com/NVIDIA/TensorRT-LLM/blob/main/LICENSE)

[Architecture](https://nvidia.github.io/TensorRT-LLM/developer-guide/overview.html)&nbsp;&nbsp;&nbsp;|&nbsp;&nbsp;&nbsp;[Performance](https://nvidia.github.io/TensorRT-LLM/developer-guide/perf-overview.html)&nbsp;&nbsp;&nbsp;|&nbsp;&nbsp;&nbsp;[Examples](https://nvidia.github.io/TensorRT-LLM/quick-start-guide.html)&nbsp;&nbsp;&nbsp;|&nbsp;&nbsp;&nbsp;[Documentation](https://nvidia.github.io/TensorRT-LLM/)&nbsp;&nbsp;&nbsp;|&nbsp;&nbsp;&nbsp;[Roadmap](https://github.com/NVIDIA/TensorRT-LLM/issues?q=is%3Aissue%20state%3Aopen%20label%3Aroadmap)
Expand Down
17 changes: 13 additions & 4 deletions constraints.txt
Original file line number Diff line number Diff line change
@@ -1,5 +1,14 @@
# These vulnerabilities were inherited from the base image (pytorch:25.10-py3) and should be removed when the base image
# These vulnerabilities were inherited from the base image (pytorch:25.12-py3) and should be removed when the base image
# is updated.
# WAR against https://github.com/advisories/GHSA-gm62-xv2j-4w53
# WAR against https://github.com/advisories/GHSA-2xpw-w6gg-jr37
urllib3>=2.6.0
# WAR against https://github.com/advisories/GHSA-38jv-5279-wg99
urllib3>=2.6.3
# WAR against https://github.com/advisories/GHSA-8rrh-rw8j-w5fx
wheel>=0.46.2
# WAR against https://github.com/advisories/GHSA-7gcm-g887-7qv7
protobuf>=6.33.5
# WAR against https://github.com/advisories/GHSA-6mq8-rvhq-8wgg
aiohttp>=3.13.3
# WAR against https://github.com/advisories/GHSA-qjxf-f2mg-c6mc
tornado>=6.5.5
# WAR against https://github.com/advisories/GHSA-3936-cmfr-pm3m
black>=26.3.1
9 changes: 9 additions & 0 deletions cpp/conan.lock
Original file line number Diff line number Diff line change
@@ -0,0 +1,9 @@
{
"version": "0.5",
"requires": [
"libnuma/system#65d9e0e45ccc1e477b97d678fa7d56bb%1769137587.8781652"
],
"build_requires": [],
"python_requires": [],
"config_requires": []
}
3 changes: 3 additions & 0 deletions cpp/include/tensorrt_llm/batch_manager/kvCacheManager.h
Original file line number Diff line number Diff line change
Expand Up @@ -288,6 +288,9 @@ class KVCacheBlock

void removeNextBlock(BlockKey const& blockKey);

void freeDescendantsRecursively();
void freeBlockAndAllDescendants();

//! \brief Find block matching blockKey. If allowPartial is true, the returned block may match only a prefix of
//! blockKey.
//! @return tuple of [partialMatch, numMatched, block], partialMatch is true if not all the tokens of the block were
Expand Down
2 changes: 2 additions & 0 deletions cpp/kernels/fmha_v2/src/fused_multihead_attention.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -342,6 +342,8 @@ static inline void set_params(bert::Fused_multihead_attention_params_v2& params,

// Attention sinks.
params.attention_sinks = reinterpret_cast<float*>(attention_sinks_d);
assert((attention_sinks_d == nullptr || launch_params.flash_attention)
&& "attention sinks are only supported with flash attention");

#if defined(STORE_P)
params.p_ptr = p_d;
Expand Down
10 changes: 6 additions & 4 deletions cpp/kernels/xqa/gen_cubins.py
Original file line number Diff line number Diff line change
Expand Up @@ -89,7 +89,8 @@

#include "tensorrt_llm/common/config.h"

TRTLLM_NAMESPACE_BEGIN
namespace tensorrt_llm
{
namespace kernels
{
// clang-format off
Expand All @@ -98,7 +99,7 @@
cpp_file_suffex_text = R"""
// clang-format on
} // namespace kernels
TRTLLM_NAMESPACE_END
}
"""

cubin_meta_info_struct_prefix_text = R"""
Expand Down Expand Up @@ -438,8 +439,9 @@ def generate_header_file_contents(
CompileMacroOption('HEAD_ELEMS', 'd', [128]),
CompileMacroOption('BEAM_WIDTH', 'beam', [1]),
CompileMacroOption('CACHE_ELEM_ENUM', 'kvt', [0, 1, 2]),
CompileMacroOption('TOKENS_PER_PAGE', 'pagedKV',
[0, 64, 128]), # 0 denotes contiguous kv cache.
CompileMacroOption(
'TOKENS_PER_PAGE', 'pagedKV',
[0, 32, 64, 128]), # 0 denotes contiguous kv cache.
CompileMacroOption('HEAD_GRP_SIZE', 'nqpkv', [0]),
CompileMacroOption('M_TILESIZE', 'm', [16, 32]),
]]
Expand Down
89 changes: 71 additions & 18 deletions cpp/kernels/xqa/mha.cu
Original file line number Diff line number Diff line change
Expand Up @@ -465,33 +465,24 @@ using WarpAcc = WarpAccT<warpTile.y, warpTile.x>;
#if SPEC_DEC
#define MMAS_N_PER_MASK 2

__device__ inline void applyMaskFromInput(Warp const& warp, WarpAcc& acc, MaskType const* mask, uint32_t rowOffset,
uint32_t nbValidCols, uint32_t qSeqLen, uint32_t actualQSeqLen, uint32_t headGrpSize
#if SLIDING_WINDOW && !IS_SPEC_DEC_TREE
,
int32_t tok0WinBeg, uint32_t seqIter, uint32_t const cacheSeqLen, uint32_t const warpTileTokenBeg
#endif
)
__device__ inline void applyMaskFromInputSlidingAndSpecDec(Warp const& warp, WarpAcc& acc, MaskType const* mask,
uint32_t rowOffset, uint32_t nbValidCols, uint32_t qSeqLen, uint32_t actualQSeqLen, uint32_t headGrpSize,
int32_t tok0WinBeg, uint32_t seqIter, uint32_t const cacheSeqLen, uint32_t const warpTileTokenBeg)
{
uint32_t const idxInQuad = laneId() % 4;
uint32_t const idxQuad = laneId() / 4;
// Packed mask is aligned with 32 bits (2 uint16_t).
uint32_t const nbPackedMasksPerRow = divUp(qSeqLen, 32u) * 2u;
uint16_t const* uint16Mask = reinterpret_cast<uint16_t const*>(mask);
constexpr uint64_t fullMask = ~uint64_t{0};
#if SLIDING_WINDOW && !IS_SPEC_DEC_TREE
Range const tileRange = {warpTileTokenBeg, warpTileTokenBeg + warpTile.x};
Range const maxMaskOutRange = {0, mha::max(0, tok0WinBeg) + (nbValidRows / MMAS_N_PER_MASK - 1)};
bool const ctaNeedBegMask = tileRange.beg < maxMaskOutRange.end;
assert(ctaNeedBegMask == overlap(tileRange, maxMaskOutRange));
int32_t const tok0NbMaskOut = int32_t(tok0WinBeg) - int32_t(warpTileTokenBeg);
uint32_t const nbSeqItersWithoutSpecDecMask = (cacheSeqLen - actualQSeqLen) / ctaTile.x;
bool const ctaNeedSpecDecMask = (seqIter >= nbSeqItersWithoutSpecDecMask);
#else
constexpr bool ctaNeedBegMask = false;
bool const ctaNeedSpecDecMask = true;
int32_t const tok0NbMaskOut = -2147483648;
#endif
bool const needMask = ctaNeedBegMask || ctaNeedSpecDecMask;

if (!needMask)
Expand Down Expand Up @@ -559,6 +550,61 @@ __device__ inline void applyMaskFromInput(Warp const& warp, WarpAcc& acc, MaskTy
}
#endif

__device__ inline void applyMaskFromInput(Warp const& warp, WarpAcc& acc, MaskType const* mask, uint32_t rowOffset,
uint32_t nbValidCols, uint32_t qSeqLen, uint32_t actualQSeqLen, uint32_t headGrpSize)
{
uint32_t const idxInQuad = laneId() % 4;
uint32_t const idxQuad = laneId() / 4;
// Packed mask is aligned with 32 bits (2 uint16_t).
uint32_t const nbPackedMasksPerRow = divUp(qSeqLen, 32u) * 2u;
uint16_t const* uint16Mask = reinterpret_cast<uint16_t const*>(mask);
#pragma unroll
for (uint32_t m = 0; m < acc.rows; m++)
{
#pragma unroll
for (uint32_t i = 0; i < InstAcc::rows; i++)
{
uint32_t const tokenRow = min((rowOffset + instM * m + idxQuad + i * 8) / headGrpSize, actualQSeqLen - 1);
#pragma unroll
for (uint32_t mask_n = 0; mask_n < acc.cols / MMAS_N_PER_MASK; mask_n++)
{
uint32_t const firstCol = instN * mask_n * MMAS_N_PER_MASK + InstAcc::cols * idxInQuad;
uint32_t const lastCol = firstCol + instN * (MMAS_N_PER_MASK - 1) + InstAcc::cols - 1;
uint32_t const maskPos0 = firstCol + actualQSeqLen < nbValidCols
? 0u
: min(firstCol + actualQSeqLen - nbValidCols, actualQSeqLen - 1);
uint32_t const maskPos1 = lastCol + actualQSeqLen < nbValidCols
? 0u
: min(lastCol + actualQSeqLen - nbValidCols, actualQSeqLen - 1);
uint32_t packedMask = 0u;
uint32_t const maskPosStart = (maskPos0 / 16) * 16;
reinterpret_cast<uint16_t*>(&packedMask)[0]
= uint16Mask[tokenRow * nbPackedMasksPerRow + (maskPos0 / 16)];
reinterpret_cast<uint16_t*>(&packedMask)[1]
= uint16Mask[tokenRow * nbPackedMasksPerRow + (maskPos1 / 16)];
#pragma unroll
for (uint32_t nj = 0; nj < MMAS_N_PER_MASK; nj++)
{
#pragma unroll
for (uint32_t j = 0; j < InstAcc::cols; j++)
{
uint32_t const n = (mask_n * MMAS_N_PER_MASK + nj);
uint32_t const col = instN * n + InstAcc::cols * idxInQuad + j;
// bool const maskFlag = col + qSeqLen < nbValidCols ? true : mask[tokenRow * qSeqLen + (col +
// qSeqLen - nbValidCols)];
bool const maskFlag = col + actualQSeqLen < nbValidCols
? true
: packedMask & (1u << ((col + actualQSeqLen - nbValidCols) - maskPosStart));
acc(m, n)(i, j) = maskFlag && col < nbValidCols ? acc(m, n)(i, j) : safeInitRowMax;
}
}
}
}
}
}

#endif

__device__ inline QuadRegRowMax warpTileOnlineSoftmax(Warp const& warp, QuadRegRowMax const& rowMaxHint, WarpAcc& acc)
{
QuadRegRowMax rowMax = rowMaxHint;
Expand Down Expand Up @@ -1655,7 +1701,7 @@ CUBIN_EXPORT __global__
uint32_t const tok0SeqLen = cacheSeqLen - actualQSeqLen + 1 + idxHeadTokenInGrp; // ctaTokOffset;
int32_t const tok0WinBeg = int32_t(tok0SeqLen) - int32_t(slidingWinSize);
uint32_t const nbTotalSkipTokens = mha::max(0, tok0WinBeg);

bool const rtIsReallySliding = (cacheSeqLen + actualQSeqLen > slidingWinSize);
#elif SLIDING_WINDOW
bool const rtIsReallySliding = (cacheSeqLen > slidingWinSize);
assert(!SPEC_DEC || !rtIsReallySliding);
Expand All @@ -1673,7 +1719,8 @@ CUBIN_EXPORT __global__

uint32_t const nbSeqIters = useKVCache ? divUp(cacheSeqLen, ctaTile.x) : 0;
#if SLIDING_WINDOW && SPEC_DEC && !IS_SPEC_DEC_TREE
uint32_t const nbSeqItersWithoutMask = nbSkipLeadingTiles;
uint32_t const nbSeqItersWithoutMask
= rtIsReallySliding ? nbSkipLeadingTiles : (cacheSeqLen - actualQSeqLen) / ctaTile.x;
#elif SPEC_DEC
uint32_t const nbSeqItersWithoutMask = (cacheSeqLen - actualQSeqLen) / ctaTile.x;
#endif
Expand Down Expand Up @@ -1960,12 +2007,18 @@ CUBIN_EXPORT __global__
if (seqIter >= nbSeqItersWithoutMask)
{
uint32_t const nbValidCols = (warpTileTokenBeg < cacheSeqLen ? cacheSeqLen - warpTileTokenBeg : 0U);
applyMaskFromInput(warp, acc, mask, idxHeadTokenInGrp, nbValidCols, qSeqLen, actualQSeqLen, headGrpSize
#if SLIDING_WINDOW && !IS_SPEC_DEC_TREE
,
tok0WinBeg, seqIter, cacheSeqLen, warpTileTokenBeg
if (rtIsReallySliding)
{
applyMaskFromInputSlidingAndSpecDec(warp, acc, mask, idxHeadTokenInGrp, nbValidCols, qSeqLen,
actualQSeqLen, headGrpSize, tok0WinBeg, seqIter, cacheSeqLen, warpTileTokenBeg);
}
else
#endif
);
{
applyMaskFromInput(
warp, acc, mask, idxHeadTokenInGrp, nbValidCols, qSeqLen, actualQSeqLen, headGrpSize);
}
}
#else
bool const isFirstIter = (seqIter == nbSkipLeadingTiles);
Expand Down
Loading
Loading