Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
1349 commits
Select commit Hold shift + click to select a range
407f406
[ROCm][CI] Adding metadata (#47477)
AndreasKaratzas Jul 3, 2026
6768fbc
[ROCm][CI] Adding qwen3 dp4 eplb (#47480)
AndreasKaratzas Jul 3, 2026
442ccc6
[ROCm][CI] Adding extract hs 2gpu (#47482)
AndreasKaratzas Jul 3, 2026
4c3c64f
Add Laguna XS.2.1 DFlash drafter support (#46853)
adamkbaranowski Jul 3, 2026
34bf7b4
[CI] intel CI: add quantization and awq case for xpu (#46456)
wendyliu235 Jul 3, 2026
276b837
[ModelRunner V2][BugFix] Free all model refs on shutdown (#47483)
njhill Jul 3, 2026
d85601c
[CI] Pin modelscope version to fix test breakage (#47465)
njhill Jul 3, 2026
41de138
[BugFix] Derive FlashInfer Q dtype from resolved per-group builder st…
mgoin Jul 3, 2026
979f551
[Bugfix][Gemma4] Keep image bidirectional attention within the slidin…
lucianommartins Jul 3, 2026
1aeabec
[Bugfix][Rust Frontend] Tolerate out-of-vocab prompt ids in detokeniz…
Sunt-ing Jul 3, 2026
9b8e765
[Rust Frontend] Recover buffered text from incomplete tool calls at E…
reidliu41 Jul 3, 2026
3f0b773
[XPU][CI]Mv huggingface cache to larger disk in Intel GPU CI (#47405)
zxd1997066 Jul 3, 2026
bd8d902
[CPU][Build] Enable oneDNN ITT task collection by default for CPU pri…
eparshut Jul 3, 2026
2dfaae7
[XPU][CI]Fix dependency typo in Intel GPU CI (#47510)
zxd1997066 Jul 3, 2026
fbc9ba6
New stable abi cleanup (#46656)
cleonard530 Jul 3, 2026
6429d5f
[Rust Frontend] add repetition_detection support to sampling params (…
yangyang-cs95 Jul 3, 2026
b790c84
[CI] Enable sccache for Rust build under CUDA/ROCm (#45246)
BugenZhao Jul 3, 2026
1f486d9
Add Triton Backend for Unlimited-OCR R-SWA (#47102)
andakai Jul 3, 2026
4875b44
[Doc] Fix VLM2Vec benchmark chat template path (#47517)
kalyanamdewri Jul 3, 2026
bbdcbe4
Move Roberta remaining nn.Embedding to VocabParallelEmbedding (#47452)
maxdebayser Jul 3, 2026
400a9c3
[Rust Frontend] Bump llm-multimodal version (#47530)
Isotr0py Jul 3, 2026
18f658b
[Bugfix][Frontend] Fix batch chat endpoint corrupting logprobs when r…
fenghourun Jul 3, 2026
a14f57a
[Frontend] Refine the entrypoint class's inheritance hierarchy. (#47498)
noooop Jul 3, 2026
978de83
[Bugfix][CPU] Ship examples/ in the CPU release image (#47447)
AgenticSpark Jul 3, 2026
d7192cf
[CI Bugfix] Lazily import Qwen warmup dependencies (#47539)
LopezCastroRoberto Jul 3, 2026
3775d5f
[ROCm][CI] Adding test groups for parity with upstream (#47479)
AndreasKaratzas Jul 3, 2026
8651f04
[Rust Frontend] Speed up chat roundtrip tests (#47523)
BugenZhao Jul 3, 2026
f63dca6
[ROCm] Fix encoder-decoder cross-attention KV layout aliasing (#47035)
djramic Jul 3, 2026
f006e5a
[CI][AMD] Allow git operations on previously created work trees (#47554)
tpopp Jul 3, 2026
576bf75
[AMD][EPLB] Enable EPLB for Quark OCP MXFP4 MoE (#47220)
okorzh-amd Jul 3, 2026
3799501
[Bugfix][Multimodal] Normalize direct PIL image inputs (#47566)
Sunt-ing Jul 3, 2026
d6d39c1
[GLM4V] Avoid GLM4V processor init during startup metadata reads (#47…
labAxiaoming Jul 3, 2026
fb5291b
[Frontend] [Parser] Port DeepSeek V4 to streaming parser engine frame…
bbrowning Jul 4, 2026
ab3b6d9
[Frontend] Limit `SO_REUSEPORT` to multi-worker serving (#47529)
BugenZhao Jul 4, 2026
67ff0ae
Support nvfp4 kv with kv-cache-dtype-skip-layers sliding_window (#42890)
sychen52 Jul 4, 2026
07516fd
[MRV2][SD] Make Dynamic SD comatible with Full Cuda Graphs (#45953)
ekagra-ranjan Jul 4, 2026
f329ce4
[ROCm][CI][Bugfix] Use VllmRunner for `voxtral_realtime` tests to avo…
shen-shanshan Jul 4, 2026
4c3c17d
[ROCm] Disable persistent sparse-MLA kernel for chunked-prefill conti…
Rohan138 Jul 4, 2026
26eb872
[Bugfix] Fix CPU split-KV scratchpad sizing (#45844)
gausah01 Jul 4, 2026
e7c9df9
[Bugfix][Structured Output][Spec Decode] Constrain bitmask and trim g…
yuyue0225sc Jul 4, 2026
1a308c4
[XPU] Add W8A8 FP8 linear kernel with multi-granularity quant support…
chaojun-zhang Jul 4, 2026
6eac8e0
[Misc] Preserve cross-encoder pooling extra kwargs (#47082)
taneem-ibrahim Jul 4, 2026
fa1fa96
[Misc] Forward request-level prompt extras for cross-encoder scoring …
taneem-ibrahim Jul 4, 2026
2f21224
[Misc] Update request-extras parity for batch chat completion (#47333)
taneem-ibrahim Jul 4, 2026
1d354c6
[Misc] Validate Pooling cache_salt Values (#46966)
taneem-ibrahim Jul 4, 2026
f1445f6
[CI] Bump `huggingface-hub` from `v1.10.2` to `v1.22.0` (#47551)
hmellor Jul 4, 2026
0cd6f76
[Bugfix][Frontend][gpt-oss] Recover raw tail when Harmony parser ends…
yzong-rh Jul 4, 2026
2a9113f
[Perf] Remove redundant op for GLM 5.2 (#47198)
yewentao256 Jul 4, 2026
d2afe39
[Bugfix][Frontend] Preserve default sampling params in batch chat (#4…
Sunt-ing Jul 4, 2026
4a6bf3c
[ROCm][CI] Fix Kernels and Kernels attention test failures (#47519)
cpersson-amd Jul 4, 2026
91b5647
[Bugfix][Model] Allow Run:ai memory_limit sentinel values (#47337)
Sunt-ing Jul 5, 2026
34b560b
[Bugfix][Gemma4] Fix FA4 mm_prefix mask: add sliding window and absol…
lucianommartins Jul 5, 2026
9226613
[Bugfix][Pooling] Forward instruction to Jina reranker scoring prompt…
Sunt-ing Jul 5, 2026
fa4321d
[Bugfix][TurboQuant] Preserve KV cache dtype in backend shape (#47609)
LucasWilkinson Jul 5, 2026
b6cc46e
[Feature] Support sequence parallel without the need for DP, 1.9%~5.0…
yewentao256 Jul 5, 2026
fb2face
[Bugfix][Model] Fix crash loading Mamba/Mamba2 checkpoints without an…
Sunt-ing Jul 5, 2026
8974ed8
[Bugfix][Voxtral Realtime] Fix token feedback timeout silent hang (#4…
Sunt-ing Jul 5, 2026
cc1d020
[MRV2] Enable mm prefix bidi attention support on MRV2 (#46942)
Isotr0py Jul 5, 2026
b712181
[ROCm][Test] Fix test_per_token_group_quant_fp8 tolerance for 1-ULP F…
spandantiwari Jul 5, 2026
78a04c2
[XPU] Fix CUDA API shims breaking Torch Dynamo during AOT compile (#4…
lslusarczyk Jul 6, 2026
d2ec433
[XPU] Fix Eagle3 initialization on XPU (#43957)
chaojun-zhang Jul 6, 2026
95a248f
[Attention Backend] HPC_ATTN backend support mtp and dynamic schedule…
thisjiang Jul 6, 2026
f2aaf59
[Feature] Support MTP speculative decoding for Bailing hybrid models …
alex101-ops Jul 6, 2026
6569df6
[Test][LoRA] Use lightweight CPU reference and skip heavy cleanup in …
chaojun-zhang Jul 6, 2026
6971582
[Test][XPU] Skip fork in kv_sharing_fast_prefill test on XPU (#47406)
Liangliang-Ma Jul 6, 2026
394edc8
[XPU] limit max-num-seqs in test_lmeval.py for XPU (#47682)
mayuyuace Jul 6, 2026
f1073c0
[CPU][BugFix] Multiple fixes to w4a8_int8 CPU MoE path (#46739)
fadara01 Jul 6, 2026
e9cc1fd
[CI/Build][CPU] Remove global extra index (#47687)
bigPYJ1151 Jul 6, 2026
d9c1767
[INC][ARK] Direct Register Custom Op for ARK (#46361)
Zhenzhong1 Jul 6, 2026
16f8110
[Bugfix][CPU][RISC-V] Fix VLEN detection for RVV attention path (#47532)
I3eg1nner Jul 6, 2026
e433634
[Performance][Hardware][RISC-V] Reduce LMUL pressure in INT4 LUT dequ…
I3eg1nner Jul 6, 2026
990c2a0
[RISC-V] Enable BF16 on VLEN=256 hardware (#45243)
velonica0 Jul 6, 2026
98ba9b9
[Frontend] Support OpenAI Responses API namespace tools (#47024)
zhongjing123 Jul 6, 2026
8f0e75e
[ROCm][CI] Adding nixl multiconn (#47481)
AndreasKaratzas Jul 6, 2026
fb265fc
[ROCm][CI] Increasing parallelism in Basic Models Tests (Extra Initia…
AndreasKaratzas Jul 6, 2026
2fa1056
[Core][DP] Rotate load-balancer tie-break to avoid systematic engine …
mayuyuace Jul 6, 2026
cdab283
[XPU][CI]Add agent tags for Basic Models Tests (Initialization) in In…
zxd1997066 Jul 6, 2026
d039c17
[Bugfix] Recycle post-final-norm hidden in GLM MTP (single norm) (#47…
zhou9402 Jul 6, 2026
344609a
[CI/Build] Fix pre-commit check (#47695)
bigPYJ1151 Jul 6, 2026
736f1a5
[XPU] Route mm_prefix models to Triton attention backend (#47688)
zhenwei-intel Jul 6, 2026
3d7f357
[Doc] docs: fix note formatting for pooling models (#47701)
llsj14 Jul 6, 2026
26c754d
[XPU][Bugfix] Do not transpose weight_scale_inv at load time (#47116)
majian4work Jul 6, 2026
90ce3a0
[bugfix] fix MOSS-Audio deepstack_input_embeds initialization in PP (…
yma11 Jul 6, 2026
ba22152
fix(security): block request-level GPU video backend selection withou…
jperezdealgaba Jul 6, 2026
40cc2e8
[Bugfix] Return HTTP 422 for unprocessable image URLs instead of 500 …
akinsella Jul 6, 2026
740f379
[ROCm][AITER] Directly Implement AITER Custom All-reduce in CudaCommu…
BadrBasowid Jul 6, 2026
98e4726
[fix][run_batch]: respect proxy env vars when downloading media URLs …
mayuyuace Jul 6, 2026
f676808
[CI] Use TTY for AMD CI tests for colored buildkite logs (#47730)
njhill Jul 6, 2026
8b79971
attention: pass None for unused args in unified attention TD path (#4…
afierka-intel Jul 6, 2026
8f4c69b
[Rust Frontend] Cache metric handles for scheduler & request stats (#…
BugenZhao Jul 6, 2026
7a90eb9
[Bugfix] [Gemma4] Fix Gemma4 MTP draft model layers ignoring quant_co…
ayush1399 Jul 6, 2026
07f9baf
Revert "[Platform] Replace `torch.cuda.Event` with `torch.Event` (#47…
jikunshang Jul 6, 2026
641cb59
[Doc] Clarify fastokens availability (#45813)
LiJzd Jul 6, 2026
373eb31
[Bugfix][Core] Fix num_output_placeholders underflow with async sched…
Sunt-ing Jul 6, 2026
51ee564
[CI] Skip test for checkpoint that was deleted (#47748)
hmellor Jul 6, 2026
095adf1
[Bugfix] Fix int32 overflow in triton_decode_attention page offsets (…
ivanium Jul 6, 2026
598d511
[Bugfix][Distributed] Delegate MNNVL allreduce one-shot selection (#4…
jesco-absolut Jul 6, 2026
b1c6dba
[Refactor] Remove multiple dead code (#47329)
yewentao256 Jul 6, 2026
8d8ec38
[Bugfix][Spec Decode] Add missing draft_id_to_target_id to DSparkDeep…
Laurent-Zhang Jul 6, 2026
f70caef
[Perf] Cache `token_to_req_indices` for dsv4, 5x~6x kernel performanc…
yewentao256 Jul 6, 2026
5ad1117
[perf]Add fused Kimi image preprocessing (#47416)
Kevin-XiongC Jul 6, 2026
5bce653
Make the Transformers modeling backend as fast as native vLLM (#47187)
hmellor Jul 6, 2026
3ee9eea
[macOS][CPU][Installation] Fix the broken installation of vllm 0.24.0…
WindChimeRan Jul 6, 2026
24dd2ae
[Bugfix] Preserve FP8 indexer WK pairs across incremental load_weight…
lcheng321 Jul 6, 2026
9fde043
[Kernel][Helion][1/N] Add Helion kernel for silu_and_mul_per_block_qu…
xiaohongchen1991 Jul 6, 2026
b136cc2
[Bugfix][Model] Add stability window to DiffusionGemma to match HF st…
NathanielMcVicar Jul 6, 2026
b1384f5
Enable B12x backend for non-gated MoEs (like Nemotron) (#43328)
askliar Jul 6, 2026
ae098ab
[CI] Fix some errors on `main` (#47726)
hmellor Jul 6, 2026
04adc88
[Bugfix]Fix DeepSeek-V4 fp8_ds_mla KV cache reshape (#47716)
ACEEE-1222 Jul 6, 2026
d891b9b
[Quantization] add humming moe backend to all dense/moe oracles (#41652)
jinzhen-lin Jul 6, 2026
567a784
[Bugfix] Fix dp mtp hang (#40589)
SherryC41 Jul 6, 2026
482e552
[Bugfix][ROCm] Fix memory access fault in AITER MLA backend for DPA+F…
simondanielsson Jul 6, 2026
8484ca5
[ROCm][CI] Adding Rust parity (#47478)
AndreasKaratzas Jul 6, 2026
5769a73
[ROCm][CI][Bugfix] Fix flaky parallel tool-call streaming (test asser…
akii96 Jul 6, 2026
86db6c3
[Frontend] add per-request timing `metrics` field to response body of…
nv-nedelman-1 Jul 7, 2026
69f3150
[XPU] Fix PP accuracy on XPU device (#47253)
yisustc Jul 7, 2026
445321f
[Bugfix] [Quantization] Fix loading for CT DSV2 (#47780)
kylesayrs Jul 7, 2026
a46c932
[Rust Frontend] Add DeepSeek V3.2 roundtrip fixture (#47619)
reidliu41 Jul 7, 2026
9dd2465
feat(cpu): add CPU support for Mamba ShortConv (#35059)
rahulssv-ibm Jul 7, 2026
a4f019f
fix(distributed): propagate distributed_timeout_seconds to NCCL devic…
jialoop-git Jul 7, 2026
700e882
Add TorchCodec as a video decoding backend (#46609)
NicolasHug Jul 7, 2026
39a1d32
[Rust Frontend] Avoid extra copies for multimodal tensors (#47581)
reidliu41 Jul 7, 2026
34e6dfc
[Rust Frontend] Stamp `arrival_time` at the frontend entry (#47787)
tahsintunan Jul 7, 2026
c64c356
[Perf] Bound DiffusionGemma sampler transient via request-tiled logit…
guan404ming Jul 7, 2026
32ab064
[UX] Add `model_class_overrides` for development and debugging (#47148)
jeejeelee Jul 7, 2026
2f71b2b
[ROCm] Align mixed encoder-decoder KV cache views in V2 runner (#47685)
AndreasKaratzas Jul 7, 2026
cbe9c40
[Bugfix] Forward callable hf_overrides to the draft model config (#45…
HumphreySun98 Jul 7, 2026
6db31c8
[XPU][CI]Adjust memory request for tests in Intel GPU CI (#47758)
zxd1997066 Jul 7, 2026
dd5c299
[ROCm][Bugfix] Convert ModelOpt FP8 per-channel weights to e4m3fnuz o…
micah-wil Jul 7, 2026
e040899
[KV Offloading] Add basic offloading metrics (#45958)
Srinivasoo7 Jul 7, 2026
8e61b64
fix(security): add resource bounds validation to derender endpoints (…
jperezdealgaba Jul 7, 2026
1e823dc
[docs update] Update usage of `hf` cli for cache list and removal (#4…
ariG23498 Jul 7, 2026
b4cfbc2
[Bugfix][Core] Fix host memory leak from undrained new_block_ids (#44…
Sunt-ing Jul 7, 2026
ba50b97
[Bugfix] Match the mapped filename in find_loaded_library (#47586)
lucifer1004 Jul 7, 2026
e55cc59
[Rust Frontend][CI] Unblock more end-to-end test cases (#47735)
BugenZhao Jul 7, 2026
5d23ca4
[Kernel] Applies routed_scaling_factor internally (#47408)
jeejeelee Jul 7, 2026
066f02a
[MoE] FI autotuning: max bucket = max token count [e.g. `DP_size*MNBT…
netanel-haber Jul 7, 2026
c5b6623
[Bugfix][Spec Decode] Skip uniform spec-decode padding for diffusion …
kl527 Jul 7, 2026
c85d720
[HARDWARE][POWER] optimize math functions of VSX power (#47321)
Rukhaiya2004 Jul 7, 2026
d3e69fd
[Perf] Use blocking CUDA events to avoid busy polling cuda driver loc…
GirasoleY Jul 7, 2026
b3e85be
fix: use configured max_logprobs instead of hardcoded 20 in derender …
jperezdealgaba Jul 7, 2026
cbb5f04
[ROCm][CI] Refresh ROCm base images when docker rocm_base changes (#4…
AndreasKaratzas Jul 7, 2026
3354dba
[Bugfix][KV offload] Store interior chunk-boundary blocks under MTP/E…
drakosha Jul 7, 2026
48fcfc9
[KV Offload] Add `ParentManager` ABC for secondary tier callbacks (#4…
ronensc Jul 7, 2026
ed051fa
[Bugfix] Reject sampling params unsupported by diffusion models (#45418)
guan404ming Jul 7, 2026
0ed05b6
[CI] Fix Transformers modeling backend LoRA test (#47832)
hmellor Jul 7, 2026
0a2965b
[BugFix] Fix ModelOpt mixed-precision quantization for sparse `quanti…
danielafrimi Jul 7, 2026
7ff656c
fix: ensure no double load of lm head in nemotron mtp (#47440)
shaunkotek Jul 7, 2026
dd94484
Bump Transformers version to 5.10.4 (#41359)
hmellor Jul 7, 2026
8b91cd5
[Bugfix][Core] Close underlying iterator in merge_async_iterators sin…
Sunt-ing Jul 7, 2026
9204699
[UX] Log worker exit code when process dies unexpectedly (#38641)
NickCao Jul 7, 2026
8b74552
[Bugfix] Fix UBatchWrapper CUDA graph key to sum all ubatches, not ju…
liulanze Jul 7, 2026
93e2ab7
Disable dynamic speculative decoding when DP is enabled (#45963)
tlrmchlsmth Jul 7, 2026
65a7b46
[KV-Offloading] Support workload identity for objectstore secondary t…
pierDipi Jul 7, 2026
65dcde1
[Bugfix] Fix PD disagg + MTP correctness for Qwen3.5(GDN) (#47466)
andakai Jul 7, 2026
c46ced1
[kv_offload] Establish tier-owned KV event handling (#46544)
Change72 Jul 7, 2026
beb4327
Enable causal masking for SWA in vllm-project/speculators models (#47…
eldarkurtic Jul 7, 2026
bdaf275
[XPU] Fix Event init failure w/ blocking (#47868)
zhenwei-intel Jul 7, 2026
21b396a
AGENTS MD: Add suggestion on how to incorporate tests (#47784)
simon-mo Jul 7, 2026
392d1b4
[BugFix][LoRA] Refresh punica metadata when LoRA slots are reassigned…
AmeenP Jul 7, 2026
bdc6f3b
[Bug] Fix tmp directory for `lm_eval` (#47755)
yewentao256 Jul 7, 2026
b93cbd7
[XPU] Fix topk_sigmoid arg mismatch on XPU (#47858)
zhenwei-intel Jul 7, 2026
c74e751
[Doc] Fix grammatically incorrect error message in gpu_worker and xpu…
robinguo23 Jul 7, 2026
c3284c3
[Perf][3/N] Expand Triton kernel warmup coverage, Qwen (#47546)
LopezCastroRoberto Jul 7, 2026
abe41f2
Upgrade tpu-inference to v0.24.0 (#47835)
CienetStingLin Jul 7, 2026
d687519
[Bugfix] Exclude kv_cache_memory_bytes from CacheConfig.compute_hash …
matteso1 Jul 7, 2026
2f3f441
fix: include topic frame in KV events replay response (#45177)
RishabhSaini Jul 7, 2026
7bd1543
[Bugfix] Fix mamba+dflash for MRV2 (#47698)
benchislett Jul 7, 2026
3dd910d
[Bugfix] Allow non-contiguous query in FlashInfer FP8 query quantizat…
MatthewBonanni Jul 7, 2026
47c40bf
[Doc] Fix manylinux tag in installation guide (#47913)
NickCao Jul 7, 2026
3f99883
[CI Bug Fix] Temp fix for v3.2 accuracy (#47902)
yewentao256 Jul 7, 2026
55da232
[Bugfix] Pad Mamba page size instead of scaling block_size in unify_k…
Sahil170595 Jul 7, 2026
dd0d74c
[Doc] Surface the --kv-cache-memory suggestion at INFO and document f…
matteso1 Jul 7, 2026
c8c2f83
Add tuned selective_state_update config for AMD Instinct MI355 (#47767)
vanshbhatia-amd Jul 7, 2026
d99adce
[BugFix] Fix ModelOpt quantization inference for fused siblings (#47445)
jasonlizhengjian Jul 7, 2026
675f429
fix(security): bound completion prompt list to prevent unbounded engi…
jperezdealgaba Jul 7, 2026
7d2ce57
[Bugfix] Patch Hopper MXFP4 OOB scales reads leading to NaN (#47910)
yzong-rh Jul 7, 2026
aad0fb7
[CI/Build] Accept ready-run-all-tests label in pre-commit gate (#47897)
AmeenP Jul 7, 2026
6e35c5e
[ROCm][CI] Minimize comment in RocmAttention q_scale check (#47731)
stefankoncarevic Jul 8, 2026
4aceabf
[ROCm][Bugfix] Key sparse-MLA persistent metadata on per-request cont…
Rohan138 Jul 8, 2026
e97c3cb
[Core] Persist and reuse the memory-profiling result across boots (op…
matteso1 Jul 8, 2026
f7efab5
[CPU][Bugfix] Fix flaky ShortConv prefill test on ARM (uninitialized …
rahulssv-ibm Jul 8, 2026
f7fc0ca
[Frontend] Add endpoint plugins framework (#47454)
hickeyma Jul 8, 2026
5e975ea
[Bugfix] Avoid blocking model launching when no system ffmpeg availab…
Isotr0py Jul 8, 2026
0ca6eee
[Core] Pass request context to CPU offload cache policy touch (#47744)
jacklin78911-collab Jul 8, 2026
dd127d8
[Core][Engine] only materialize tokens when thinking budget is in req…
walterbm Jul 8, 2026
0303f37
[Bugfix][Pooling] Align CrossEncoder token type ids after truncation …
Sunt-ing Jul 8, 2026
9021589
[Minimax-M3] Using tok_sparse_select from MSA instead of triton kerne…
zyongye Jul 8, 2026
d9e57ea
[ROCm][Perf] MXFP8 dense-linear + grouped-MoE GEMM optimizations for …
amd-ethany Jul 8, 2026
80eb01e
[Bugfix] DSV4 TP16 garbage output (#47493)
majunze2001 Jul 8, 2026
2afa3f7
[Perf] Minimax M3 - Support cross-layer allreduce-norm fusion (#47631)
wzhao18 Jul 8, 2026
5d5fab0
[Bugfix][Frontend] Fix http_requests_total metric recording some 4xx …
zqzten Jul 8, 2026
c0e8e1f
[Bugfix] Register VLLM_BUILD_* and VLLM_IMAGE_TAG provenance env vars…
nicklasfrahm Jul 8, 2026
d35eba3
[Bugfix] Avoid leaking Pydantic repr in tool_choice error message (#4…
muhammadfawaz1 Jul 8, 2026
2c64b4c
[ROCm] fixed aiter master flag and expert parallelism compatibility o…
hongxiayang Jul 8, 2026
7cc2e8e
fix: hash speculative draft model config (#47911)
alexeldeib Jul 8, 2026
51e5372
[Model][HunyuanVL] Use native transformers processor and adapt to tra…
ManaEstras Jul 8, 2026
d79855e
[Docs] `kv_sharing_fast_prefill` correction (#47044)
NickLucche Jul 8, 2026
7c67da9
Remove unused _get_kv_cache_config_deepseek_v4 alias (#47969)
NickLucche Jul 8, 2026
99a8561
[Test] Skip DeepEP MoE layer tests without P2P access (#47946)
tlrmchlsmth Jul 8, 2026
4400025
[XPU] [Fusion passes] Disable fuse_rope_kvcache_cat_mla & qk_norm_rop…
chaojun-zhang Jul 8, 2026
bd3bb4e
[Misc][Docs] Add human-readable integer support for more cli-args (#…
NickLucche Jul 8, 2026
04a703e
[Frontend] Support bad_words in the /v1/completions endpoint (#46793)
sungbin1015 Jul 8, 2026
1f4ad05
[ROCm] Add tuned selective_state_update float16 config for AMD Instin…
vanshbhatia-amd Jul 8, 2026
285c08c
[Model] Support MOSS-Transcribe-Diarize (#47729)
gcanlin Jul 8, 2026
eeaf231
[ROCm] Add tuned selective_state_update float32 config for AMD Instin…
vanshbhatia-amd Jul 8, 2026
e7b3853
Remove router weight upcast for DSv2-related models (#47970)
gau-nernst Jul 8, 2026
a1ab51a
[Bugfix] Allocate HY V3 expert_bias in float32 to prevent silent down…
aoright Jul 8, 2026
db39d60
Add tuned selective_state_update float32 config for AMD Instinct MI35…
vanshbhatia-amd Jul 8, 2026
2cae98d
[Rust Frontend] Handle `continue_final_message` with renderer sentine…
BugenZhao Jul 8, 2026
934eeae
[CI/Build][BugFix][The Rock] Fix get_ssm_device_name to return saniti…
rasmith Jul 8, 2026
cd0de48
[Bugfix][V1] Free out-of-window blocks on the processed-token basis u…
Saddss Jul 8, 2026
9f2b3b0
Improvement of Docker image build for IBM Power using prebuilt wheels…
vivek8123 Jul 8, 2026
572b25b
[Bug] Fix Batched DeepGEMM (#47884)
robertgshaw2-redhat Jul 8, 2026
68b4a1d
Fix NVML capability lookup for visible devices (#47892)
tlrmchlsmth Jul 8, 2026
0d12618
[Spec Decode] Support hybrid (SWA + full attention) DFlash drafters (…
mgoin Jul 8, 2026
c2ecd0f
Fix FlashAttention MLA prefill V unpadding (#42642)
voipmonitor Jul 8, 2026
f05603f
[Bugfix][DCP] Cast LSE to fp32 in a2a combine to fix bf16 bitcast cra…
shawntsai Jul 8, 2026
d1f1d86
[Bugfix] Re-enable benchmarking of librispeech dataset. (#47033)
almayne Jul 8, 2026
b2cf70e
[CI] BugFix Eval Small Models Distributed test for DiffusionGemma (#4…
ilmarkov Jul 8, 2026
8347c6e
updated flash_attn GIT_TAG to point to torch Stable ABI FA3 commit (#…
cleonard530 Jul 8, 2026
a5d19cb
[Core] Move MRV1 `late_interaction_runner.py` out of MRV2 subtree (#4…
njhill Jul 8, 2026
089e412
[Perf] Integrate TRTLLM BF16 MoE Modular Kernel (#45182)
kjiang249 Jul 8, 2026
49abada
[ROCm][Bugfix] Fix empty-tensor .max() crash in AITER FA (#47894)
djramic Jul 8, 2026
0d2f4e7
Allow FlashInfer A2A backends for TRTLLM FP8 MoE Modular (#46661)
gau-nernst Jul 8, 2026
dcdd756
[CI] GSM8K eval integration test for KV offloading (#46893)
tlrmchlsmth Jul 8, 2026
5f85975
[Feat] Add runtime monitor for post-warmup TileLang compilation (#46718)
LopezCastroRoberto Jul 8, 2026
6cf7b26
[docs] Fix the docs build (#48008)
hickeyma Jul 8, 2026
2683194
[ROCm] Fix pooling startup workspace lock (#47912)
AndreasKaratzas Jul 8, 2026
56da398
Fix embed scaling + CUDA graphs in Transformers modelling backend (#4…
hmellor Jul 8, 2026
95d6d6f
[Bugfix] Use int8 workspace for FlashInfer MLA decode (#48046)
njhill Jul 8, 2026
7802c20
[KVConnector][NIXL] Support pipeline-parallel prefill in push mode (#…
zixi-qi Jul 8, 2026
bc44f9f
[ROCm][CI][MoE] Fix double-transpose of fused w3 expert weights (#47874)
stefankoncarevic Jul 8, 2026
2c17d33
[Bugfix][ROCm] Change AttentionCGSuppoort in TritonMLA to UNIFORM_SIN…
music-dino Jul 9, 2026
b8c7c86
[XPU][LoRA] Fix torch.compile DEVICE_LOST by avoiding view-mutation i…
chaojun-zhang Jul 9, 2026
529af88
[KV Offloading] Add free block iterator for CPU offload scheduling (#…
chaunceyjiang Jul 9, 2026
1171467
[CPU] Fix Qwen-Next SSM type for AMX GDN (#48073)
bigPYJ1151 Jul 9, 2026
a07765c
[Bugfix] Fix Qwen3-ASR transcription streaming postprocessing (#42478)
BWAAEEEK Jul 9, 2026
ab7961a
Remove TeleChatForCausalLM (#47989)
xianbaoqian Jul 9, 2026
0206f10
Add Intel XPU Docker release pipeline (#47880)
wendyliu235 Jul 9, 2026
1cd75b3
[Bugfix] Fix race condition in KVBlockZeroer (#48085)
benchislett Jul 9, 2026
e875216
Sanitize server file paths from validation error responses (#46415)
muhammadfawaz1 Jul 9, 2026
ae6170f
[P/D][Bugfix] Fix PD async KV load lookahead handling for MTP spec de…
chaunceyjiang Jul 9, 2026
412414d
Remove PersimmonForCausalLM and FuyuForCausalLM model architectures (…
xianbaoqian Jul 9, 2026
b83be00
Migrate Olmo and Olmo2 to the Transformers modeling backend (#48100)
hmellor Jul 9, 2026
85b3a72
[Bugfix][Model Runner V2] Order uniform decodes first so spec decodes…
WoosukKwon Jul 9, 2026
299d2b5
[CI] Annotate built Docker image tags on the Buildkite build page (#4…
khluu Jul 9, 2026
753c503
Pin PyNvVideoCodec to tested 2.0.4 wheel (#48056)
brandonpelfrey Jul 9, 2026
429f405
[Bugfix] Guard CUDA-only rms_norm_per_block_quant in FUSED_OPS for no…
tsvikas Jul 9, 2026
7aa9d4e
Merge branch 'main' into feature/top-n-sigma-logits-processor
Codekiing Jul 9, 2026
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
The table of contents is too big for display.
Diff view
Diff view
  •  
  •  
  •  
6 changes: 3 additions & 3 deletions .buildkite/ci_config_intel.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -2,17 +2,17 @@ name: vllm_intel_ci
job_dirs:
- ".buildkite/intel_jobs"
run_all_patterns:
- ".buildkite/ci_config_intel.yaml"
- ".buildkite/scripts/hardware_ci/run-intel-test.sh"
- "docker/Dockerfile"
- "docker/Dockerfile.xpu"
- "CMakeLists.txt"
- "requirements/common.txt"
- "requirements/xpu.txt"
- "requirements/build/cuda.txt"
- "requirements/test/cuda.txt"
- "setup.py"
- "csrc/"
- "cmake/"
run_all_exclude_patterns:
- "docker/Dockerfile."
- "csrc/cpu/"
- "csrc/rocm/"
- "cmake/hipify.py"
Expand Down
1 change: 1 addition & 0 deletions .buildkite/ci_config_rocm.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -8,6 +8,7 @@ run_all_patterns:
- "docker/docker-bake-rocm.hcl"
- ".buildkite/hardware_tests/amd.yaml"
- ".buildkite/scripts/ci-bake-rocm.sh"
- ".buildkite/scripts/rocm/"
- ".buildkite/scripts/hardware_ci/run-amd-test.py"
- ".buildkite/scripts/hardware_ci/run-amd-test.sh"
- "CMakeLists.txt"
Expand Down
63 changes: 34 additions & 29 deletions .buildkite/hardware_tests/amd.yaml
Original file line number Diff line number Diff line change
@@ -1,18 +1,45 @@
group: Hardware - AMD Build

# ROCm image flow:
# 1. Refresh the long-lived ROCm base image only when Dockerfile.rocm_base changes.
# 2. Build ci_base from either the stable base or the freshly refreshed base.
# 3. Build the per-commit ROCm CI image and smoke-test it before GPU jobs run.
steps:
- label: "AMD: :docker: refresh ROCm base"
key: refresh-rocm-base-amd
depends_on: []
device: amd_cpu
no_plugin: true
commands:
- bash .buildkite/scripts/rocm/refresh-base-image.sh
env:
DOCKER_BUILDKIT: "1"
BUILDKIT_PROGRESS: "tty"
TERM: "xterm-256color"
retry:
automatic:
- exit_status: -1 # Agent was lost
limit: 1
- exit_status: -10 # Agent was lost
limit: 1

# Ensure ci_base is up-to-date before building the test image.
# Compares a content hash of ci_base-affecting files against the remote
# image label. If hashes match the build is skipped (< 30 s); if they
# differ ci_base is rebuilt and pushed automatically.
- label: "AMD: :docker: ensure ci_base"
key: ensure-ci-base-amd
depends_on: []
soft_fail: false
depends_on:
- refresh-rocm-base-amd
device: amd_cpu
no_plugin: true
commands:
- bash .buildkite/scripts/ci-bake-rocm.sh ci-base-rocm-ci-with-deps
- bash .buildkite/scripts/rocm/build-ci-base.sh
env:
DOCKER_BUILDKIT: "1"
BUILDKIT_PROGRESS: "tty"
TERM: "xterm-256color"
VLLM_BAKE_FILE: "docker/docker-bake-rocm.hcl"
PYTORCH_ROCM_ARCH: "gfx90a;gfx942;gfx950"
REMOTE_VLLM: "1"
Expand All @@ -26,40 +53,18 @@ steps:

- label: "AMD: :docker: build test image and artifacts"
key: image-build-amd
soft_fail: false
depends_on:
- ensure-ci-base-amd
device: amd_cpu
no_plugin: true
commands:
- |
if [[ "${ROCM_CI_ARTIFACT_ONLY:-0}" == "1" ]]; then
echo "ROCM_CI_ARTIFACT_ONLY=1; building ROCm wheel artifact only"
IMAGE_TAG="" bash .buildkite/scripts/ci-bake-rocm.sh test-rocm-ci-with-artifacts
else
bash .buildkite/scripts/ci-bake-rocm.sh test-rocm-ci-with-wheel
fi
- |
docker run --rm --network=none --entrypoint /bin/bash "rocm/vllm-ci:${BUILDKITE_COMMIT}" -ec '
if [ ! -d /vllm-workspace ]; then echo Missing directory: /vllm-workspace >&2; exit 1; fi
if [ ! -d /vllm-workspace/tests ]; then echo Missing directory: /vllm-workspace/tests >&2; exit 1; fi
if [ ! -d /vllm-workspace/src/vllm ]; then echo Missing directory: /vllm-workspace/src/vllm >&2; exit 1; fi
if [ ! -x /vllm-workspace/src/vllm/vllm-rs ]; then echo Missing executable: /vllm-workspace/src/vllm/vllm-rs >&2; exit 1; fi
command -v python3
command -v uv
command -v pytest
if ! command -v amd-smi >/dev/null 2>&1 && ! command -v rocminfo >/dev/null 2>&1; then
echo No ROCm CLI found in image >&2
exit 1
fi
python3 - <<PY
import torch, vllm
print(torch.__version__)
print(vllm.__version__)
PY
echo AMD image smoke OK
'
- bash .buildkite/scripts/rocm/build-test-image.sh
- bash .buildkite/scripts/rocm/smoke-test-image.sh
env:
DOCKER_BUILDKIT: "1"
BUILDKIT_PROGRESS: "tty"
TERM: "xterm-256color"
VLLM_BAKE_FILE: "docker/docker-bake-rocm.hcl"
PYTORCH_ROCM_ARCH: "gfx90a;gfx942;gfx950"
IMAGE_TAG: "rocm/vllm-ci:$BUILDKITE_COMMIT"
Expand Down
41 changes: 24 additions & 17 deletions .buildkite/hardware_tests/cpu.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -17,29 +17,32 @@ steps:
- tests/kernels/test_awq_int4_to_int8.py
- tests/kernels/quantization/test_cpu_fp8_scaled_mm.py
- tests/kernels/mamba/cpu/test_cpu_gdn_ops.py
- tests/kernels/mamba/test_cpu_short_conv.py
commands:
- |
bash .buildkite/scripts/hardware_ci/run-cpu-test.sh 30m "
pytest -x -v -s tests/kernels/attention/test_cpu_attn.py
pytest -x -v -s tests/kernels/moe/test_cpu_fused_moe.py
pytest -x -v -s tests/kernels/moe/test_cpu_quant_fused_moe.py
pytest -x -v -s tests/kernels/mamba/test_cpu_short_conv.py
pytest -x -v -s tests/kernels/test_onednn.py
pytest -x -v -s tests/kernels/test_awq_int4_to_int8.py
pytest -x -v -s tests/kernels/quantization/test_cpu_fp8_scaled_mm.py
pytest -x -v -s tests/kernels/mamba/cpu/test_cpu_gdn_ops.py"

- label: CPU-Compatibility Tests
depends_on: []
device: intel_cpu
no_plugin: true
source_file_dependencies:
- cmake/cpu_extension.cmake
- setup.py
- vllm/platforms/cpu.py
commands:
- |
bash .buildkite/scripts/hardware_ci/run-cpu-test.sh 20m "
bash .buildkite/scripts/hardware_ci/run-cpu-compatibility-test.sh"
# Note: SDE can't be downloaded from CI host because of AWS WAF
# - label: CPU-Compatibility Tests
# depends_on: []
# device: intel_cpu
# no_plugin: true
# source_file_dependencies:
# - cmake/cpu_extension.cmake
# - setup.py
# - vllm/platforms/cpu.py
# commands:
# - |
# bash .buildkite/scripts/hardware_ci/run-cpu-test.sh 20m "
# bash .buildkite/scripts/hardware_ci/run-cpu-compatibility-test.sh"

- label: CPU-Language Generation and Pooling Model Tests
depends_on: []
Expand All @@ -52,7 +55,7 @@ steps:
- tests/models/language/pooling/
commands:
- |
bash .buildkite/scripts/hardware_ci/run-cpu-test.sh 40m "
bash .buildkite/scripts/hardware_ci/run-cpu-test.sh 50m "
pytest -x -v -s tests/models/language/generation -m cpu_model
pytest -x -v -s tests/models/language/pooling -m cpu_model"

Expand All @@ -67,13 +70,15 @@ steps:
- vllm/v1/sample/ops/topk_topp_triton.py
- vllm/v1/sample/ops/topk_topp_sampler.py
- tests/v1/sample/test_topk_topp_sampler.py
- tests/v1/e2e/test_cpu_linear_attn_chunked_prefix.py
commands:
- |
bash .buildkite/scripts/hardware_ci/run-cpu-test.sh 45m "
uv pip install git+https://github.com/triton-lang/triton-cpu.git@270e696d
VLLM_USE_V2_MODEL_RUNNER=1 pytest -x -v -s tests/models/language/generation/test_granite.py -m cpu_model
# TODO: move to CPU-Kernel Tests once triton-cpu has a pre-built wheel
pytest -x -v -s tests/v1/sample/test_topk_topp_sampler.py::TestTritonTopkTopp"
pytest -x -v -s tests/v1/sample/test_topk_topp_sampler.py::TestTritonTopkTopp
pytest -x -v -s tests/v1/e2e/test_cpu_linear_attn_chunked_prefix.py"

- label: CPU-Quantization Model Tests
depends_on: []
Expand All @@ -88,11 +93,13 @@ steps:
- vllm/model_executor/layers/fused_moe/experts/cpu_moe.py
- tests/quantization/test_compressed_tensors.py
- tests/quantization/test_cpu_wna16.py
- tests/quantization/test_cpu_w8a8.py
commands:
- |
bash .buildkite/scripts/hardware_ci/run-cpu-test.sh 30m "
bash .buildkite/scripts/hardware_ci/run-cpu-test.sh 45m "
pytest -x -v -s tests/quantization/test_compressed_tensors.py::test_compressed_tensors_w8a8_logprobs
pytest -x -v -s tests/quantization/test_cpu_wna16.py"
pytest -x -v -s tests/quantization/test_cpu_wna16.py
pytest -x -v -s tests/quantization/test_cpu_w8a8.py"

- label: CPU-Distributed Tests (PP+TP)
depends_on: []
Expand Down Expand Up @@ -135,7 +142,7 @@ steps:
- |
bash .buildkite/scripts/hardware_ci/run-cpu-test.sh 45m "
pytest -x -v -s tests/models/multimodal/generation --ignore=tests/models/multimodal/generation/test_pixtral.py -m cpu_model --num-shards=$$BUILDKITE_PARALLEL_JOB_COUNT --shard-id=$$BUILDKITE_PARALLEL_JOB"
parallelism: 3
parallelism: 4

- label: "Arm CPU Test"
depends_on: []
Expand Down
80 changes: 80 additions & 0 deletions .buildkite/hardware_tests/intel_xpu_ci/test-intel.yaml
Original file line number Diff line number Diff line change
@@ -0,0 +1,80 @@
group: Intel
steps:
- label: ":docker: Build XPU image"
soft_fail: true
optional: true
depends_on: []
key: image-build-xpu
commands:
- bash -lc '.buildkite/image_build/image_build_xpu.sh "public.ecr.aws/q9t5s3a7" "vllm-ci-test-repo" "$BUILDKITE_COMMIT"'
env:
DOCKER_BUILDKIT: "1"
retry:
automatic:
- exit_status: -1 # Agent was lost
limit: 2
- exit_status: -10 # Agent was lost
limit: 2
- label: "XPU example Test"
depends_on:
- image-build-xpu
timeout_in_minutes: 30
optional: true
device: intel_gpu
agent_tags:
label: production
gpu: 2+
mem: 24+
no_plugin: true
env:
REGISTRY: "public.ecr.aws/q9t5s3a7"
REPO: "vllm-ci-test-repo"
source_file_dependencies:
- .buildkite/hardware_tests/intel_xpu_ci/test-intel.yaml
- .buildkite/scripts/hardware_ci/run-intel-ci-test.sh
commands:
- >-
bash .buildkite/scripts/hardware_ci/run-intel-test.sh
'bash .buildkite/scripts/hardware_ci/run-intel-ci-test.sh example'
- label: "XPU V1 test"
depends_on:
- image-build-xpu
timeout_in_minutes: 30
optional: true
device: intel_gpu
agent_tags:
label: production
gpu: 1+
mem: 24+
no_plugin: true
env:
REGISTRY: "public.ecr.aws/q9t5s3a7"
REPO: "vllm-ci-test-repo"
source_file_dependencies:
- .buildkite/hardware_tests/intel_xpu_ci/test-intel.yaml
- .buildkite/scripts/hardware_ci/run-intel-ci-test.sh
commands:
- >-
bash .buildkite/scripts/hardware_ci/run-intel-test.sh
'bash .buildkite/scripts/hardware_ci/run-intel-ci-test.sh v1'
- label: "XPU server test"
depends_on:
- image-build-xpu
timeout_in_minutes: 30
optional: true
device: intel_gpu
agent_tags:
label: production
gpu: 1+
mem: 16+
no_plugin: true
env:
REGISTRY: "public.ecr.aws/q9t5s3a7"
REPO: "vllm-ci-test-repo"
source_file_dependencies:
- .buildkite/hardware_tests/intel_xpu_ci/test-intel.yaml
- .buildkite/scripts/hardware_ci/run-intel-ci-test.sh
commands:
- >-
bash .buildkite/scripts/hardware_ci/run-intel-test.sh
'bash .buildkite/scripts/hardware_ci/run-intel-ci-test.sh server'
8 changes: 8 additions & 0 deletions .buildkite/image_build/image_build.sh
Original file line number Diff line number Diff line change
Expand Up @@ -79,12 +79,18 @@ setup_buildx_builder() {
docker buildx ls | grep -E '^\*|^NAME' || docker buildx ls
}

annotate_image_tags() {
.buildkite/scripts/annotate-image-build.sh \
"${IMAGE_TAG:-}" "${IMAGE_TAG_LATEST:-}"
}

check_and_skip_if_image_exists() {
if [[ -n "${IMAGE_TAG:-}" ]]; then
echo "--- :mag: Checking if image exists"
if docker manifest inspect "${IMAGE_TAG}" >/dev/null 2>&1; then
echo "Image already exists: ${IMAGE_TAG}"
echo "Skipping build"
annotate_image_tags
exit 0
fi
echo "Image not found, proceeding with build"
Expand Down Expand Up @@ -254,3 +260,5 @@ echo "--- :docker: Building ${TARGET}"
docker --debug buildx bake -f "${VLLM_BAKE_FILE_PATH}" -f "${CI_HCL_PATH}" --progress plain "${TARGET}"

echo "--- :white_check_mark: Build complete"

annotate_image_tags
37 changes: 19 additions & 18 deletions .buildkite/image_build/image_build_arm64.sh
Original file line number Diff line number Diff line change
Expand Up @@ -9,29 +9,30 @@ fi
REGISTRY=$1
REPO=$2
BUILDKITE_COMMIT=$3
IMAGE="$REGISTRY/$REPO:$BUILDKITE_COMMIT-arm64"

# authenticate with AWS ECR
aws ecr-public get-login-password --region us-east-1 | docker login --username AWS --password-stdin "$REGISTRY" || true

# skip build if image already exists
if [[ -z $(docker manifest inspect "$REGISTRY"/"$REPO":"$BUILDKITE_COMMIT"-arm64) ]]; then
echo "Image not found, proceeding with build..."
else
if docker manifest inspect "$IMAGE" >/dev/null 2>&1; then
echo "Image found"
exit 0
else
echo "Image not found, proceeding with build..."
# build for arm64 GPU targets: Grace/GH200 (sm_90) and DGX Spark/GB10
# (sm_121, family-covered by 12.0 under CUDA 13)
docker build --file docker/Dockerfile \
--platform linux/arm64 \
--build-arg max_jobs=16 \
--build-arg nvcc_threads=4 \
--build-arg torch_cuda_arch_list="9.0 12.0" \
--build-arg USE_SCCACHE=1 \
--build-arg buildkite_commit="$BUILDKITE_COMMIT" \
--tag "$IMAGE" \
--target test \
--progress plain .
# push
docker push "$IMAGE"
fi

# build (Grace/GH200 is the arm64 GPU target; sm_90)
docker build --file docker/Dockerfile \
--platform linux/arm64 \
--build-arg max_jobs=16 \
--build-arg nvcc_threads=4 \
--build-arg torch_cuda_arch_list="9.0" \
--build-arg USE_SCCACHE=1 \
--build-arg buildkite_commit="$BUILDKITE_COMMIT" \
--tag "$REGISTRY"/"$REPO":"$BUILDKITE_COMMIT"-arm64 \
--target test \
--progress plain .

# push
docker push "$REGISTRY"/"$REPO":"$BUILDKITE_COMMIT"-arm64
.buildkite/scripts/annotate-image-build.sh "$IMAGE"
Loading
Loading