[{"url":"https://api.github.com/repos/vllm-project/vllm/releases/385359183","assets_url":"https://api.github.com/repos/vllm-project/vllm/releases/385359183/assets","upload_url":"https://uploads.github.com/repos/vllm-project/vllm/releases/385359183/assets{?name,label}","html_url":"https://github.com/vllm-project/vllm/releases/tag/v0.29.0","id":385359183,"author":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"node_id":"RE_kwDOI7xefs4W-B1P","tag_name":"v0.29.0","target_commitish":"main","name":"v0.29.0","draft":false,"immutable":false,"prerelease":false,"created_at":"2026-09-08T08:50:34Z","updated_at":"2026-09-09T09:40:07Z","published_at":"2026-09-09T08:54:49Z","assets":[{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/552379255","id":552379255,"node_id":"RA_kwDOI7xefs4g7KN3","name":"vllm-0.29.0+cpu-cp312-cp312-macosx_11_0_arm64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":27952289,"digest":"sha256:7133cb494664c502b07b114fe915847f0e71296d502f67f6fa76172ba46978df","download_count":13,"created_at":"2026-09-09T08:56:05Z","updated_at":"2026-09-09T08:56:07Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.29.0/vllm-0.29.0%2Bcpu-cp312-cp312-macosx_11_0_arm64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/552379257","id":552379257,"node_id":"RA_kwDOI7xefs4g7KN5","name":"vllm-0.29.0+cpu-cp38-abi3-manylinux_2_34_aarch64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":68308832,"digest":"sha256:f14a4c18c0ba55f6932f41064f8fe345387e99e3ec76bb9c4a63b6069e876ca0","download_count":10,"created_at":"2026-09-09T08:56:05Z","updated_at":"2026-09-09T08:56:10Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.29.0/vllm-0.29.0%2Bcpu-cp38-abi3-manylinux_2_34_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/552379258","id":552379258,"node_id":"RA_kwDOI7xefs4g7KN6","name":"vllm-0.29.0+cpu-cp38-abi3-manylinux_2_34_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":137859053,"digest":"sha256:4d22dac8259e7e24dbc615531c73cbd1a6d4b4214a3329ccddb7854f26c9dac8","download_count":67,"created_at":"2026-09-09T08:56:05Z","updated_at":"2026-09-09T08:56:15Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.29.0/vllm-0.29.0%2Bcpu-cp38-abi3-manylinux_2_34_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/552379259","id":552379259,"node_id":"RA_kwDOI7xefs4g7KN7","name":"vllm-0.29.0+cu129-cp38-abi3-manylinux_2_28_aarch64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":521697691,"digest":"sha256:b4efb6f657a1cc2ea8f898d39127974c53bd72206f95a5879611ade5c986e61f","download_count":13,"created_at":"2026-09-09T08:56:05Z","updated_at":"2026-09-09T08:56:40Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.29.0/vllm-0.29.0%2Bcu129-cp38-abi3-manylinux_2_28_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/552379252","id":552379252,"node_id":"RA_kwDOI7xefs4g7KN0","name":"vllm-0.29.0+cu129-cp38-abi3-manylinux_2_28_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":548493389,"digest":"sha256:22e8d8fec755986b3ad964004a1f8c65a55626ec948354bdbe95993e6b0289fe","download_count":108,"created_at":"2026-09-09T08:56:05Z","updated_at":"2026-09-09T08:56:41Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.29.0/vllm-0.29.0%2Bcu129-cp38-abi3-manylinux_2_28_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/552379317","id":552379317,"node_id":"RA_kwDOI7xefs4g7KO1","name":"vllm-0.29.0+xpu-cp38-abi3-manylinux_2_34_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":31591009,"digest":"sha256:aed7bcbe1084bc1d416f41d82ffc0451fbbe1398fa6e479549749051f998a87c","download_count":14,"created_at":"2026-09-09T08:56:09Z","updated_at":"2026-09-09T08:56:11Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.29.0/vllm-0.29.0%2Bxpu-cp38-abi3-manylinux_2_34_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/552379378","id":552379378,"node_id":"RA_kwDOI7xefs4g7KPy","name":"vllm-0.29.0-cp38-abi3-manylinux_2_28_aarch64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":310033787,"digest":"sha256:e6b0dfc2b6fd307315e9b34b73cd2bfe7b6b08958eda721828e61732bba426b0","download_count":14,"created_at":"2026-09-09T08:56:11Z","updated_at":"2026-09-09T08:56:30Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.29.0/vllm-0.29.0-cp38-abi3-manylinux_2_28_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/552379387","id":552379387,"node_id":"RA_kwDOI7xefs4g7KP7","name":"vllm-0.29.0-cp38-abi3-manylinux_2_28_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":315961042,"digest":"sha256:09d48617fc2be9c6cdcd5db480651ab0d84817b257204f2cc2e3ecbb70bbb635","download_count":62,"created_at":"2026-09-09T08:56:12Z","updated_at":"2026-09-09T08:56:32Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.29.0/vllm-0.29.0-cp38-abi3-manylinux_2_28_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/552379493","id":552379493,"node_id":"RA_kwDOI7xefs4g7KRl","name":"vllm-0.29.0.tar.gz","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/x-gtar","state":"uploaded","size":40906181,"digest":"sha256:c03feb07943ecae9e69e7dc424d7c5569ecf8d1292513aef2a4de67af381cd97","download_count":19,"created_at":"2026-09-09T08:56:16Z","updated_at":"2026-09-09T08:56:19Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.29.0/vllm-0.29.0.tar.gz"}],"tarball_url":"https://api.github.com/repos/vllm-project/vllm/tarball/v0.29.0","zipball_url":"https://api.github.com/repos/vllm-project/vllm/zipball/v0.29.0","body":"# v0.29.0\r\n\r\n## Highlights\r\n\r\nThis release features 594 commits from 277 contributors (91 new)!\r\n\r\n* **Model Runner V2 is now the default for all models** (#53183), completing the rollout that began with pooling models (#48290). MRV2 also gained CUDA graph memory profiling for KV cache auto-sizing (#53306), batch-sharded sampling that cuts per-step logits memory by 1/TP (#50465), prompt embeds (#42963), `extract_hidden_states` speculation (#49811), padded FULL cudagraph dispatch for uniform decode under spec decode (#53407), and DP-sync skipping before EAGLE/MTP draft prefill (#53694). MRV1 remains in use for a few ROCm models and features MRV2 does not yet support.\r\n* **New models**: Hy4-preview, Tencent's 770B/49B-active MoE with Gated DeepSeek Sparse Attention and native MTP (#54160); Qwen3.8-Flash-Next with BF16/FP8/NVFP4 and MTP (#53896); GraniteSWA and GraniteMoeSWA (#52706); NemotronH_Omni_Reasoning_V3 with MTP (#52929, #53121); Kimi K3 NVFP4 checkpoints (#53132).\r\n* **Kimi-K3 and DeepSeek V4 performance**: fused MXFP4 top-k finalization in the K3 latent tail (about 5% E2E latency, #53152), K3 Mamba metadata preparation in one Triton launch (6.6-7.6x kernel speedup, #52388), tuned Hopper low-latency GEMM (#54088) now also dispatched on SM100 (#53534) and used for `eh_proj` (12.9-25.2% kernel speedup, #53942), GEMM-RS extended to GEMM-AR (#53053), MLA gate merged into the QKV-A projection (#54015), K3 DCP with DSpark (#52188) and DCP partial prefix cache hits (#50493); DeepSeek V4 shared experts fused into MegaMoE (#53040), adaptive top-k width re-landed (#52823), a native SwiGLU clamp kernel for Humming MoE (#53685), and an opt-in FlashInfer `moe_ep` expert backend (#49636).\r\n* **Speculative decoding**: per-request acceptance stats in OpenAI API responses via `--per-request-spec-decode-metrics` (#48915), adaptive verification extended to logprobs (#52242), SM100 sparse MLA for GLM-5.2 (#52783) and DeepSeek V4 on SM90 (#52795), Qwen3-Omni DSpark drafts (#52560), PLaMo3 EAGLE-3/DFlash (#54239), and DFlash2 loading from speculators format (#53797).\r\n* **RL weight sync**: a new `sharded_rdt` P2P backend where each worker pulls only its TP/EP slice over NIXL or Ray Direct Transport (#43375), rank-local IPC weight updates (#52497), sparse checkpoint-coordinate updates through native weight loaders (#50723, #53751), and routed expert loading for gpt-oss (#52209).\r\n* **Mamba prefix caching**: internal prefill checkpoints deliver a 9%-25% TTFT improvement (#52789); `prefix_cache_retention_interval` is now a CLI argument defaulting to 0 (#52216), with dense retention automatically restored for hybrid models using EAGLE/MTP (#55760, #55861).\r\n* **New defaults**: FlashInfer all-reduce enabled by default for TP CUDA groups, opt out with `VLLM_ALLREDUCE_USE_FLASHINFER=0` (#52998); prefix-cache `NONE_HASH` is deterministic by default so distributed KV cache users no longer need to pin `PYTHONHASHSEED` (#51875); new `--max-num-queued-reqs` / `--max-num-queued-tokens` admission-control flags (#49445).\r\n* **Breaking changes**: ten deprecated model architectures removed (#53608); FlexOlmo, Olmo3 and Hunyuan V1/VL migrated to the Transformers modeling backend (#53615); PyAV video decoder backend removed (#54231); `python -m vllm.entrypoints.openai.api_server` deprecated in favor of `vllm serve` (#52131); `VLLM_TEST_FORCE_FP8_MARLIN` (#52182) and `VLLM_ROCM_USE_AITER_FP4_ASM_GEMM` (#53141) removed.\r\n\r\n## Release Artifacts\r\n\r\n### Python Wheels\r\n\r\n| Platform | Install |\r\n|---|---|\r\n| PyPI (CUDA 13.0) | `pip install vllm` |\r\n| PyPI (CUDA 13.0, uv) | `uv pip install vllm --torch-backend=auto` |\r\n| ROCm | `pip install vllm --extra-index-url https://wheels.vllm.ai/rocm/0.29.0/rocm723` |\r\n| XPU | `uv pip install vllm --extra-index-url https://wheels.vllm.ai/0.29.0/xpu --extra-index-url https://download.pytorch.org/whl/xpu --index-strategy unsafe-best-match` |\r\n\r\n### Docker Images\r\n\r\n| Platform | Docker Image |\r\n|---|---|\r\n| CUDA 13.0 (Default) | `docker pull vllm/vllm-openai:v0.29.0` (`v0.29.0-cu130` also works) |\r\n| CUDA 12.9 | `docker pull vllm/vllm-openai:v0.29.0-cu129` |\r\n| CUDA 13.0 + Ubuntu 24.04 | `docker pull vllm/vllm-openai:v0.29.0-ubuntu2404` |\r\n| CUDA 12.9 + Ubuntu 24.04 | `docker pull vllm/vllm-openai:v0.29.0-cu129-ubuntu2404` |\r\n| ROCm | `docker pull vllm/vllm-openai-rocm:v0.29.0` |\r\n| CPU | `docker pull vllm/vllm-openai-cpu:v0.29.0` |\r\n| XPU | `docker pull vllm/vllm-openai-xpu:v0.29.0` |\r\n\r\n### Other Artifacts\r\n\r\nPre-built release artifacts are available in the **Assets** section at the bottom of this page, including:\r\n- Source distribution tarball\r\n- CUDA 12.9 Python wheels for x86_64 and arm64\r\n- CUDA 13.0 Python wheels for x86_64 and arm64\r\n- CPU Python wheels for x86_64, arm64, and macOS\r\n- XPU Python wheel for x86_64\r\n\r\n## Model Support\r\n* **New models**: Hy4-preview (#54160), Qwen3.8-Flash-Next (#53896), GraniteSWA and GraniteMoeSWA via the existing Granite implementation (#52706), NemotronH_Omni_Reasoning_V3 (#52929) with MTP for Nemotron VL models (#53121), bidirectional attention for DeepSeek-backbone embedding models (#52948), FP8 ModernBERT (#53101), and Kimi K3 NVFP4 checkpoints (#53132).\r\n* **Transformers modeling backend**: FlexOlmo, Olmo3 and Hunyuan V1/VL migrated off native implementations (#53615), multimodal path hardened (#51827), and `RMSNormFuser.fuse` performance fixed (#52766).\r\n* **Weight loading**: weight tying now inspects the checkpoint so a real `lm_head` is loaded even when the config claims tied embeddings (#51665, #53170).\r\n* **LoRA**: DeepSeek V4 (#53361), Qwen3-Omni multimodal LoRA (#52786, #53557), tower/connector LoRA for LLaVA-NeXT (#49788) and LFM2-VL (#51498), plus fixes for Muse-Glimmer (#53513), Qwen3.5 embedding modules (#48850), partial LoRA on Qwen3.5/3.6 GatedDeltaNet (#47640), int32 overflow in punica kernels at long context (#53034), false target matches on unsupported module types (#52313), and base-layer/routed-expert prefix ordering (#52552).\r\n* **Multimodal performance**: ViT full CUDA graph for Idefics3 and SmolVLM (#47625), packed encoder attention for Pixtral (#52185), fused Kimi vision Q/K RoPE kernel (#50400), Dots3 NOTE runtime (#53517) and Omni encoder optimizations (#53460), MM tensors no longer broadcast to workers for prefix-cache-covered items (#52041), redundant placeholder scans skipped (#52925), common token sequences cached (#53560), async media resolved concurrently across modalities (#54537), and H2D copies pinned to avoid stream stalls (#53412, #54292, #54299).\r\n* **Multimodal inputs**: video embeds accepted by the Python frontend (#54242), modality-scoped `mm_processor_kwargs` honored (#53808), Qwen3-VL profiling honors `cap_pixels_per_frame` (#54380), video frame sampling respects MP4 edit-list trims (#48608), oversized items skip the MM processor cache instead of crashing (#53016), encoder cache entries survive until their last use (#54284), and repeated multimodal requests with the SHM cache no longer terminate the engine (#54994).\r\n* **Correctness**: PaliGemma stale image scaling removed (#52692), Mistral3 placeholder grid with processor size overrides (#52874), MiniCPM-o on Transformers v5 (#54501), JinaVL cache order (#53553), mixed CLIP/SigLIP pooling batches (#53165), XD-RoPE models on prefix-cache hits (#53456), Qwen3-Omni audio encoder with non-divisible TP (#50858), GLM-5.2 no longer uses dense MHA (#52512), Moondream3 MoE all-reduce (#54152), deepseek-vl2 config defaults (#51302), HY-V3 compressed-tensors ignore matching (#48682), MiniMax-M3 FP8 query allocation under CUDA graph replay (#51203), and GraniteMoeHybrid quantized expert loading (#54052).\r\n\r\n## Engine Core\r\n* **Model Runner V2**: default for all models (#53183) and pooling models (#48290), CUDA graph memory reservation (#53306), batch-sharded sampling (#50465), prompt embeds (#42963), `extract_hidden_states` (#49811), padded FULL cudagraph dispatch (#53407), DP-sync skipping for drafts (#53694), decoupled draft/target gumbel noise streams (#54282), encoder-only path split out (#53176), and memory released correctly on shutdown and sleep (#53508, #54246, #54162, #53955, #53682).\r\n* **Speculative decoding**: per-request acceptance stats (#48915), adaptive verification with logprobs (#52242), on SM100 sparse MLA (#52783) and DSv4 + SM90 (#52795), varlen trtllm-gen decode (#52157), FlashInfer MLA for DSpark drafting under DCP (#54277), fused GDN MTP for all Qwen head ratios (#52539), widest uniform decode batch captured by default (#50488) with memory-safe graph sizes (#54418), and fixes for short_conv/LFM2 targets (#50272), speculators-format config overrides (#42376), DFlash draft RoPE layout (#54373), draft models with a different GQA ratio (#53002), draft models with a larger hidden size under TP (#52193), DSpark backend inheritance scoped to DeepSeek V4 (#52809), generic `DSparkDraftModel` configs for Qwen3 (#52197), and Gemma4 MTP under CUDA graphs (#53884).\r\n* **Prefix caching & scheduling**: Mamba internal prefill checkpoints (#52789), `prefix_cache_retention_interval` argument (#52216, #55760, #55861), deterministic `NONE_HASH` (#51875), queue admission control (#49445), KV null block reserved when validating `max_model_len` (#47272), spec decode no longer padded up to `max_model_len` (#53962), and negative external block allocation prevented (#52707).\r\n* **RL workflows**: `sharded_rdt` P2P weight sync (#43375), rank-local IPC updates (#52497), sparse checkpoint updates (#50723, #53751), gpt-oss routed expert loading (#52209), stable DeepSeek V4 mHC broadcast buffers across weight sync (#52626), and packed weight transfer stream reuse capping reserved-memory waste (#52951).\r\n* **Determinism**: `trace_decode_token_ids` for deterministic decode replay (#46701), per-arch tuned batch-invariant matmul configs (about 3x decode kernels on RTX 4090D/H20, #53247), Blackwell autotuning with 33.6% E2E latency reduction (#53649), deterministic MoE combine under DP+EP (#45683), and `fuse_allreduce_rms` disabled under batch invariance (#51292).\r\n* **Kernels**: fused embedding kernel (#53677), FA4 re-enabled for head_size=256 on Blackwell (#52980), vectorized sparse MLA mask loads (#52217), masked MHA prefill for GLM-5 head dimensions (#53785), GLM-5.2 sparse MLA Q concatenation fused with head padding (#53878), fused QK-norm + partial MRoPE + gate for Qwen3.6 (#52676), FlashInfer CuTeDSL BF16 low-latency GEMM as opt-in `--linear-backend flashinfer_cutedsl` (#50572), replicated embedding and norm fusion for DSV3 flat models (#48484), standardized fused shared-expert selection (#51695), GPT-OSS MoE topk metadata reuse (#45457), tuned cooperative topk (#53382), and tuned FP8 fused_moe for Qwen3.5 on L40S (+7%, #53819).\r\n* **Robustness**: KV cache layout standardized under a `KVCacheLayout` enum (#51718), JIT warmup provider registry (#50174), `--cpu-offload-params` now reaches vision/audio towers (#53120), attention backend probe failures no longer crash init (#51703), FlashInfer XQA falls back on unsupported head_dim (#53111), FlashInfer prefill LSE normalized before merging to fix prefix-cache logits divergence (#52796), seed preserved when a batch mixes seeded and unseeded requests (#51866), startup thread allocation accounts for local DP workers (#52385), int32 overflow fixes in fused SiLU block quant (#53409) and LoRA kernels (#53034), a shared-memory race in fused groupwise RMSNorm quantization (#54111), BLHNC addressing for FlashInfer sparse MLA (#54465), Mamba state copy race (#50729), and a `start_profile` no-op after auto-stop (#51839).\r\n\r\n## Hardware & Performance\r\n* **NVIDIA**: DeepSeek V3.2 / GLM-5.2 DSA routed to the optimized CUDA path on all GPUs (#52861), PCP for DSv3.2 sparse MLA (#52046), cuBLAS out_dtype router GEMM on all archs including GB10 (#54048), FlashInfer all-reduce tuning on SM103 (#53318, #53606), FA4 hdim256 on SM100 (#52980), SM120 sparse MLA fixes (#51395, #53574), DeepSeek V3.2 fused kernel grids hardened for 65k+ token launches (#52381), FlashMLA sparse decode workspace fix (#53755), MNNVL Lamport corruption fix (#53000), and opt-in Rubin Docker builds for CUDA 13.4/13.5 (#53443).\r\n* **AMD ROCm**: dual-stream decode with hipgraphs (#52033), W4A4 preshuffled asm GEMM by default (+15% throughput on Llama-3.3-70B MXFP4, #53141), ROCr/CLR update fixing graph replay segfaults with up to about 20% TPOT improvement (#53712), fused KDA decode on MI325X (#52293), FULL cudagraphs for AITER MLA spec decode (#51171), FP8 asm MLA prefill for non-divisor head counts (#51040), DCP causal multi-token verification (#51705) and prefix cache hits for Kimi-K3 (#53598), AITER PA gluon decode for MiniMax-M3 (#52849), DeepSeek-V4 fusions for mHC/RMSNorm (#52737), C4A top-k (#52882), C4 compressor GEMMs (#53838) and SWA q/kv norm + FP8 quant (#53540), fused shared experts for block-FP8 (#53097), CPU offload on ROCm 7.13+ (#43018), TheRock 7.14 preview docker (#49925), int4/int8 quantization fixes (#52112, #48998, #51632, #53110), CUDA graphs captured on the current stream (#53818), AITER metadata preserved across graph replay (#53821), and improved ROCm detection under WSL (#38434).\r\n* **Intel XPU**: INC int4 W4A8 linear backend (#50501), AutoRound MXFP8 MoE (#51248), EC connector KV offloading (#49532), and fixes for HunyuanOCR XD-RoPE (#52174), sparse-MLA metadata sync (#52066), Mamba state pointer overflow (#48109), oneCCL warm-up at world size 1 (#52389), and MRoPE (#53201).\r\n* **CPU**: AMX high-performance MLA backend for DeepSeek V2/V3/R1 (#52616) with MLA now running end-to-end (#51471), FP16/BF16 persisted GDN state on AMX (#52191), Int8 MoE through zentorch on AMD Zen (#44834), Voxtral support (#53921), C++ causal_conv1d GDN on non-AMX AVX-512BF16 (#49688), and assorted fixes (#54042).\r\n\r\n## Large Scale Serving\r\n* **Context parallelism**: Kimi-K3 DCP with DSpark (#52188) and partial prefix cache hits (#50493), FlashInfer native CP for MLA decode (#54012), FlashMLA sparse DCP on Hopper with MTP (#46514), NIXL P/D DCP for MLA models (#50611), PCP for DSv3.2 (#52046) and NIXL PCP producers (#52779), `--dcp-q-replicate` with query replication default-on for GLM sparse attention (#50382), DCP fused attention fix for DeepSeek-V3.2 / GLM-5.2 (#50005), sparse MLA metadata and kernel block sizes under DCP (#52377, #51031), PCP PIECEWISE cudagraph fixes (#53869, #53515), and PCP compatibility checks delegated to the PCP manager so plugins can enable GQA+PCP (#53853).\r\n* **Elastic EP**: reduced eager-mode reconfiguration downtime (#51885), AOT cache reuse preserved during scaling (#53378), and scale-below-minimum rejected (#52702).\r\n* **MoE communication**: DeepEP v2 receiver CPU overhead (#51114) and MXFP8 activation scale dispatch (#51398), FlashInfer one-sided All2All refinements (#51924), DeepEP v2 fixes for `--enforce-eager` startup (#51824) and the decode/cudagraph path (#52632), NCCL>=2.31 heap overflow fix (#53008), cross-node MNNVL all-reduce gated by capability (#53253), and MNNVL all-reduce buffers sized for DSpark (#50932).\r\n* **KV connectors**: Mooncake Store decode KV saving via `save_decode_cache` (#52466) and hybrid DCP prefix caching (#53324), externally transferable KV cache group identification (#53779), async KV loads deferred past forward launch (#53333), `kv_transfer_params` for `/inference/v1/generate` (#42644), MoRIIO shared KV region registration after the layout refactor (#53698), and Mamba fixes for Mooncake (#51362, #51358, #53663) and NIXL (#53523).\r\n* **KV offloading**: EC offloading connector driven by CUDA events (#49994), ownership in KV cache events (#52067, #52068), P2P tier request-level offload (#52912) and abort handling (#52571), `/dev/shm` leak on crash fixed (#52596), CPU->GPU loads ordered against compute stream (#50696), `store_threshold` counting fixed (#52227), in-flight primary keys cascaded (#53329), and padded GPU cache storage handled (#54021).\r\n* **Data/pipeline parallel**: PP silent corruption fix (#54962, #49274), DP coordinator wake handling (#51481), device sync on pause (#52914), TCPStore port fixes for Ray (#53666, #50969), DP supervisor inheriting the uvicorn config (#52473), EPD encoder round-robin fix (#52491), and producer-only EC config normalization (#53656).\r\n\r\n## Quantization\r\n* **New backends**: FlashInfer TRT-LLM MXFP8 linear (#52204), b12x FP4 MoE for SM120/SM121 (#52018), AutoRound block-wise FP8 (#47434), Humming MoE with MXFP4 weights + block-FP8 activations (#51332), and Humming for compressed-tensors WNA16 MoE (#48918).\r\n* **Fixes**: weight-only NVFP4 checkpoints routed through W4A16 (#54427), compressed-tensors block FP8 with Marlin (#52966) and int8 grouped WNA16 MoE (#52002), OCP MX mxfp6 activation emulation (#52704), MXFP8 FlashInfer path guarded on availability (#52648), and Humming activation aliasing (#54056).\r\n\r\n## API & Frontend\r\n* **New endpoints & options**: `/v1/messages/render` for the Anthropic Messages API (#45803), `/cohere/v2/chat/render` (#53219), SSE keep-alive comments for idle streams (#51034), per-request spec decode metrics (#48915), video embeds input (#54242), `--max-num-queued-reqs` / `--max-num-queued-tokens` (#49445), and pooling requests can set padding (#51157).\r\n* **Anthropic & Cohere**: `vllm_xargs` forwarded to sampling params (#53308), `stop_sequence` stop reason reported (#45807), Cohere citation/tool 500s fixed (#52175), and stop string limit applied (#53750).\r\n* **OpenAI compatibility**: `logprobs=-1` in Completions (#46175), streamed logprob offsets with echo (#47815), batched chat echo (#52529), all choices returned from `/inference/v1/generate` with n>1 (#52399), `cache_salt` forwarded for content parts (#54315), `stop_token_ids` validated against vocab (#54196), and 4xx for client errors in `/detokenize` (#52622), malformed namespace tools (#53763), malformed base64 audio (#53744), and unknown chat roles in DeepSeek encoders (#53071); tool-call arguments in replayed history are parsed defensively (#48922), empty `FlatLogprobs` slices are handled (#53704), and empty bad-word tokenizations are rejected (#53433).\r\n* **Structured output & parsers**: reasoning-end detection scoped to the current turn (#54089), terminal grammars stop under `min_tokens` (#54218), unsupported pattern+length schemas rejected (#49996), MistralCommonBackend tokenizers (#52720), XGrammar termination in batches (#52805), spurious FSM errors after speculative reasoning end (#53046), shared parser engine adapters (#52830), Gemma4 parenthesized tool calls (#53657) and `enable_thinking` default (#52430), HY-V3 parallel calls in one delta (#53965), GPT-OSS Harmony strict grammar (#52222), Kimi K3 reserved markers excluded from response text (#52889), and unused request-local reasoners skipped (#52573).\r\n* **Rust frontend**: HY3 unified parser with local XGrammar structural tags (#53054), gRPC audio/video inputs (#53760), gRPC LoRA lifecycle control (#52840, #52031, #53756), `--generation-config vllm` (#53044), `truncate_prompt_tokens` (#48584), OpenAI edge-case alignment (#53218), optimized SSE hot path (#51321), pure-Rust `protox` replacing `protoc` (#52892), RL world-size reporting (#53204) and routed expert prompt offsets (#52703), and fixes for GLM-5.2 template parity (#51426), Qwen parser auto-detection (#51169), Kimi K3 `reasoning_effort=\"none\"` (#53043), `n > 1` rejection on `/inference/v1/generate` (#52844), and the `LogprobsTensors` wire schema (#53939).\r\n* **Pooling**: BGE-M3 throughput (+3.13%, #53464) and task validation (#51823), prompts truncated before padding (#54364), parallel arrays truncated with the prompt (#54407, #54509), and batched input throughput in `vllm bench serve` (#53213).\r\n* **CLI & tooling**: `vllm launch` runs the serve argument checks (#52825), `run_batch.py` moved out of the openai folder (#53500) with its upload retry fixed (#50588), `vllm bench` warns on a warm prefix cache for random runs (#53920) and restores multimodal datasets on the `vllm` throughput backend (#52168), and the vLLM recipes tool gained sweep recommendations and alias parsing (#53325, #53946).\r\n\r\n## Security\r\n* `cache_salt` length bounded to 1024 to prevent scheduler CPU exhaustion (#54353).\r\n* Oversized media rejected before full download (#51896); `VLLM_MAX_AUDIO_CLIP_FILESIZE_MB` enforced on all audio paths (#53561).\r\n* Decoder prompt-length validation enforced for processors that skip the check (#46588); PyNvVideoCodec decoder slot limit bypass fixed (#52126).\r\n* `api_key` and `hf_token` redacted from startup logs, compile cache factors, and the Rust frontend launch log (#52523, #53625, #53738).\r\n\r\n## Dependencies\r\n* FlashInfer 0.6.18 (#54313), huggingface-hub 1.28.0 (#52797), tpu-inference v0.28.0 (#54020), NIXL 1.3.2 (#51777).\r\n* InstantTensor added to CUDA dependencies; the loader is not enabled by default (#52801). CuPy constraint relaxed to exclude only 14.1.0 (#44284).\r\n* ROCm base image: ROCr/CLR update (#53712), rocprofiler-sdk 1.3.2 (#53182), TheRock 7.14 preview (#49925), LMCache connector packages (#51208).\r\n* XPU: vllm_xpu_kernels 0.1.14.1 (#54203), UCX install updated (#53817).\r\n* Build fails closed when the selected precompiled CUDA variant is unavailable (#52545).\r\n\r\n## Breaking Changes & Deprecations\r\n* Ten deprecated architectures removed: Arctic, Chameleon, Cheers, Fairseq2Llama, FireRedLID, GritLM, HCXVision, MPT, the `RWForCausalLM` and `StableLMEpochForCausalLM` aliases, and `PrithviGeoSpatialMAE` (superseded by `Terratorch`) (#53608).\r\n* FlexOlmo, Olmo3, Hunyuan V1 and Hunyuan VL are now served through the Transformers modeling backend (#53615).\r\n* PyAV video decoder backend removed; use OpenCV or Torchcodec (#54231).\r\n* `python -m vllm.entrypoints.openai.api_server` is deprecated; use `vllm serve` (#52131).\r\n* Model Runner V2 is the default runner for all models (#53183).\r\n* `prefix_cache_retention_interval` default changed from dense to 0 for SWA/SSM models; the env var is deprecated in favor of the argument (#52216).\r\n* FlashInfer all-reduce enabled by default (#52998).\r\n* `VLLM_TEST_FORCE_FP8_MARLIN` removed in favor of `--linear-backend` / `--moe-backend` (#52182); `VLLM_ROCM_USE_AITER_FP4_ASM_GEMM` removed (#53141); dead `--attention-config.use_prefill_decode_attention` removed (#52557); other long-deprecated parameters cleaned up (#53559).\r\n\r\n## New Contributors\r\n* @030611 made their first contribution in https://github.com/vllm-project/vllm/pull/51823\r\n* @92hyungjun made their first contribution in https://github.com/vllm-project/vllm/pull/47272\r\n* @ActiveSky made their first contribution in https://github.com/vllm-project/vllm/pull/52692\r\n* @adisivaprasad made their first contribution in https://github.com/vllm-project/vllm/pull/53854\r\n* @Agoni-02 made their first contribution in https://github.com/vllm-project/vllm/pull/48850\r\n* @AmitMY made their first contribution in https://github.com/vllm-project/vllm/pull/48608\r\n* @Andy365-365 made their first contribution in https://github.com/vllm-project/vllm/pull/52523\r\n* @andyluo7 made their first contribution in https://github.com/vllm-project/vllm/pull/53821\r\n* @AnkitNakhawa made their first contribution in https://github.com/vllm-project/vllm/pull/52491\r\n* @anmolgupt made their first contribution in https://github.com/vllm-project/vllm/pull/52874\r\n* @bobboli made their first contribution in https://github.com/vllm-project/vllm/pull/51924\r\n* @brianosaurus made their first contribution in https://github.com/vllm-project/vllm/pull/52557\r\n* @CalvinXKY made their first contribution in https://github.com/vllm-project/vllm/pull/50858\r\n* @canlahlah made their first contribution in https://github.com/vllm-project/vllm/pull/53409\r\n* @CherryLemon made their first contribution in https://github.com/vllm-project/vllm/pull/53877\r\n* @CHIPMUNK-T0T made their first contribution in https://github.com/vllm-project/vllm/pull/47625\r\n* @cogniera made their first contribution in https://github.com/vllm-project/vllm/pull/53839\r\n* @cr-zhao made their first contribution in https://github.com/vllm-project/vllm/pull/52385\r\n* @daviswer made their first contribution in https://github.com/vllm-project/vllm/pull/52706\r\n* @DCoEngine made their first contribution in https://github.com/vllm-project/vllm/pull/48682\r\n* @dineshchitlangia made their first contribution in https://github.com/vllm-project/vllm/pull/49688\r\n* @dkrisman made their first contribution in https://github.com/vllm-project/vllm/pull/54380\r\n* @dmvevents made their first contribution in https://github.com/vllm-project/vllm/pull/52632\r\n* @Edge-Explorer made their first contribution in https://github.com/vllm-project/vllm/pull/53435\r\n* @eilamc14 made their first contribution in https://github.com/vllm-project/vllm/pull/47640\r\n* @eligotts made their first contribution in https://github.com/vllm-project/vllm/pull/54315\r\n* @Eoin-Houstoun made their first contribution in https://github.com/vllm-project/vllm/pull/51703\r\n* @floatlibai made their first contribution in https://github.com/vllm-project/vllm/pull/52144\r\n* @fuzzifikation made their first contribution in https://github.com/vllm-project/vllm/pull/51034\r\n* @hagaikwa-redhat made their first contribution in https://github.com/vllm-project/vllm/pull/53230\r\n* @haoyangqian made their first contribution in https://github.com/vllm-project/vllm/pull/52690\r\n* @Hert4 made their first contribution in https://github.com/vllm-project/vllm/pull/51157\r\n* @ima-helikoptaaa made their first contribution in https://github.com/vllm-project/vllm/pull/54427\r\n* @jbyczkow made their first contribution in https://github.com/vllm-project/vllm/pull/52174\r\n* @JC-ut0 made their first contribution in https://github.com/vllm-project/vllm/pull/53071\r\n* @jiahaoliang made their first contribution in https://github.com/vllm-project/vllm/pull/51262\r\n* @JiataiWang made their first contribution in https://github.com/vllm-project/vllm/pull/53553\r\n* @jl9876 made their first contribution in https://github.com/vllm-project/vllm/pull/51248\r\n* @JulianZJN made their first contribution in https://github.com/vllm-project/vllm/pull/53253\r\n* @jungjiyu made their first contribution in https://github.com/vllm-project/vllm/pull/53939\r\n* @kyleliang-nv made their first contribution in https://github.com/vllm-project/vllm/pull/51203\r\n* @LH-and-FPGA made their first contribution in https://github.com/vllm-project/vllm/pull/52648\r\n* @li-ukumar made their first contribution in https://github.com/vllm-project/vllm/pull/52571\r\n* @LioEinaudi made their first contribution in https://github.com/vllm-project/vllm/pull/53247\r\n* @LironKesem made their first contribution in https://github.com/vllm-project/vllm/pull/53921\r\n* @Lossfull made their first contribution in https://github.com/vllm-project/vllm/pull/52948\r\n* @lxy-alexander made their first contribution in https://github.com/vllm-project/vllm/pull/52430\r\n* @lxyxinyi made their first contribution in https://github.com/vllm-project/vllm/pull/50809\r\n* @machero made their first contribution in https://github.com/vllm-project/vllm/pull/53531\r\n* @matthewkotila made their first contribution in https://github.com/vllm-project/vllm/pull/48915\r\n* @mhoqueanik made their first contribution in https://github.com/vllm-project/vllm/pull/49636\r\n* @mhuzaifa3 made their first contribution in https://github.com/vllm-project/vllm/pull/53625\r\n* @minjang made their first contribution in https://github.com/vllm-project/vllm/pull/53530\r\n* @MKQuantum made their first contribution in https://github.com/vllm-project/vllm/pull/51866\r\n* @new-TonyWang made their first contribution in https://github.com/vllm-project/vllm/pull/53704\r\n* @nicholaskh-ai made their first contribution in https://github.com/vllm-project/vllm/pull/53819\r\n* @oliverholworthy made their first contribution in https://github.com/vllm-project/vllm/pull/52185\r\n* @prakharPant made their first contribution in https://github.com/vllm-project/vllm/pull/52743\r\n* @Prudhvivuda made their first contribution in https://github.com/vllm-project/vllm/pull/53016\r\n* @rajathpi made their first contribution in https://github.com/vllm-project/vllm/pull/52622\r\n* @ray24777 made their first contribution in https://github.com/vllm-project/vllm/pull/53120\r\n* @roachsinai made their first contribution in https://github.com/vllm-project/vllm/pull/53528\r\n* @RookieCoder-Camera made their first contribution in https://github.com/vllm-project/vllm/pull/49274\r\n* @sandeep-maddipatla made their first contribution in https://github.com/vllm-project/vllm/pull/51777\r\n* @seonjinn made their first contribution in https://github.com/vllm-project/vllm/pull/52204\r\n* @ShengleiFu made their first contribution in https://github.com/vllm-project/vllm/pull/53650\r\n* @shepark made their first contribution in https://github.com/vllm-project/vllm/pull/51302\r\n* @shijuzhao made their first contribution in https://github.com/vllm-project/vllm/pull/45683\r\n* @shipiyouniao made their first contribution in https://github.com/vllm-project/vllm/pull/51979\r\n* @ShuaiShao93 made their first contribution in https://github.com/vllm-project/vllm/pull/53034\r\n* @simon-veitner-redhat made their first contribution in https://github.com/vllm-project/vllm/pull/52980\r\n* @sseanliu made their first contribution in https://github.com/vllm-project/vllm/pull/52041\r\n* @studioego made their first contribution in https://github.com/vllm-project/vllm/pull/52389\r\n* @tanchao made their first contribution in https://github.com/vllm-project/vllm/pull/50191\r\n* @thanhpt1110 made their first contribution in https://github.com/vllm-project/vllm/pull/52720\r\n* @theamalsebastian made their first contribution in https://github.com/vllm-project/vllm/pull/52588\r\n* @thunguo made their first contribution in https://github.com/vllm-project/vllm/pull/53101\r\n* @tolleybot made their first contribution in https://github.com/vllm-project/vllm/pull/51292\r\n* @tommy-asai-sonarsource made their first contribution in https://github.com/vllm-project/vllm/pull/51395\r\n* @tthakkal made their first contribution in https://github.com/vllm-project/vllm/pull/50501\r\n* @ukannika made their first contribution in https://github.com/vllm-project/vllm/pull/52849\r\n* @VBS2004 made their first contribution in https://github.com/vllm-project/vllm/pull/48922\r\n* @wyettzeng made their first contribution in https://github.com/vllm-project/vllm/pull/52209\r\n* @Xuan-1998 made their first contribution in https://github.com/vllm-project/vllm/pull/53008\r\n* @y0hnn made their first contribution in https://github.com/vllm-project/vllm/pull/52002\r\n* @Yiqin-17 made their first contribution in https://github.com/vllm-project/vllm/pull/52925\r\n* @YukioZzz made their first contribution in https://github.com/vllm-project/vllm/pull/51705\r\n* @ZHIHANCHEN03 made their first contribution in https://github.com/vllm-project/vllm/pull/53432\r\n* @Zhou248 made their first contribution in https://github.com/vllm-project/vllm/pull/52560\r\n* @zllion made their first contribution in https://github.com/vllm-project/vllm/pull/46701\r\n* @zwischenraum made their first contribution in https://github.com/vllm-project/vllm/pull/50272\r\n\r\n## Contributors\r\n@AndreasKaratzas, @khluu, @mgoin, @taneem-ibrahim, @yewentao256, @njhill, @BugenZhao, @hmellor, @GirasoleY, @DarkLight1337, @mayuyuace, @WoosukKwon, @jeejeelee, @noooop, @jperezdealgaba, @LucasWilkinson, @wzhao18, @stefankoncarevic, @gau-nernst, @okorzh-amd, @ZJY0516, @HollowMan6, @connorcarpenter15, @gcanlin, @aoshen02, @linitra24, @mhuzaifa3, @Isotr0py, @Etelis, @MatthewBonanni, @TheEpicDolphin, @zupengwang, @atalman, @NickLucche, @rasmith, @elvircrn, @JasonKeyiL, @Hotragn, @waizuichougou, @pmanczak, @divakar-amd, @he-yufeng, @xuebwang-amd, @qgallouedec, @esmeetu, @ZeldaHuang, @Prudhvivuda, @chaunceyjiang, @zxd1997066, @louie-tsai, @andyxning, @pisceskkk, @tjtanaa, @qli88, @tlrmchlsmth, @LopezCastroRoberto, @micah-wil, @hongxiayang, @BabyDrangoner, @yimdev, @mganczarenko, @SageMoore, @biswapanda, @sfeng33, @russellb, @itayalroy, @bigPYJ1151, @drakosha, @qwerqwerqwe8688-jpg, @akii96, @shen-shanshan, @meiyeh123, @reidliu41, @yma11, @frgossen, @lukealonso, @tanchao, @Fangzhou-Ai, @vllm-agent, @djramic, @andrewbcohere, @zyongye, @gty111, @ShuoleiWang, @ZHIHANCHEN03, @KurodaKanbei, @alexeldeib, @khushali9, @xiaohuguo2023, @hungnnvidia, @thisjiang, @arpera, @ShengleiFu, @zhenwei-intel, @jungjiyu, @yzong-rh, @charlifu, @hclsys, @Ronald1995, @mkhazraee, @YukioZzz, @theamalsebastian, @030611, @cr-zhao, @jbyczkow, @tommy-asai-sonarsource, @jikunshang, @bobboli, @LH-and-FPGA, @SayHelloToWorld, @kkt-cohere, @floatlibai, @rajathpi, @lxy-alexander, @gangula-karthik, @ActiveSky, @fxmarty-amd, @sstamenk, @mpashkovskii, @LiuYinfeng01, @chaojun-zhang, @Oxygen56, @daviswer, @vineethsaivs, @Agoni-02, @Andy365-365, @wangxiyuan, @haoyangqian, @Naveassaf, @eilamc14, @sseanliu, @y0hnn, @dineshchitlangia, @yiliu30, @Lossfull, @lxyxinyi, @vanshbhatia-amd, @jcotant-inferact, @chengy-sysu, @92hyungjun, @xyang16, @AmitMY, @zllion, @AnkitNakhawa, @dmvevents, @lengrongfu, @JC-ut0, @Eoin-Houstoun, @sagearc, @wjabbour, @sandeep-maddipatla, @seonjinn, @Yiqin-17, @kyleliang-nv, @thanhpt1110, @MKQuantum, @matthewkotila, @vhagor, @tthakkal, @ShuaiShao93, @hagaikwa-redhat, @shijuzhao, @elwhyjay, @Gregory-Pereira, @ErenAta16, @morrison-turnansky, @DCoEngine, @SoluMilken, @xianbaoqian, @kliuae, @nascheme, @waynehacking8, @stecasta, @vMaroon, @zwischenraum, @Rohan138, @Zhou248, @brianosaurus, @anmolgupt, @wyettzeng, @hao-aaron, @frank-suwen, @danisereb, @fuzzifikation, @KernelClint, @thunguo, @studioego, @shipiyouniao, @cjackal, @LioEinaudi, @guan404ming, @libinta, @mhoqueanik, @minjang, @Edge-Explorer, @shepark, @JiataiWang, @jiahaoliang, @therealnaveenkamal, @ColinZ22, @tolleybot, @almogtavor, @simon-veitner-redhat, @hyeongyun0916, @nicholaskh-ai, @mawong-amd, @adisivaprasad, @ray24777, @robertgshaw2-redhat, @new-TonyWang, @oliverholworthy, @fanxingran, @lucianommartins, @hallerite, @avininjamay8, @VBS2004, @omerpaz95, @Sunt-ing, @Hert4, @wangshangsam, @cogniera, @jeffreywang88, @Xarbirus, @fynnsu, @jiahanc, @zixi-qi, @prakharPant, @haic0, @maobaolong, @xinyu-intel, @CHIPMUNK-T0T, @hangy-amd, @afriedri, @canlahlah, @Alnusjaponica, @ukannika, @pranavthakur0-0, @zzaebok, @Xuan-1998, @JulianZJN, @roachsinai, @ivanium, @CalvinXKY, @yu-xin-c, @rchalamala, @QwertyJack, @Kaif10, @yudigege86, @yiz-liu, @dkrisman, @simondanielsson, @machero, @CherryLemon, @ima-helikoptaaa, @RookieCoder-Camera, @LironKesem, @peakcrosser7, @eligotts, @foraxe, @zufangzhu, @tianmu-li, @jl9876, @lucifer1004, @liranschour, @maithilijoshi20, @li-ukumar, @andyluo7, @Zhenzhong1, @faaany, @joerowell, @juhi10071998, @ganeshr10, @Priyjain-amd, @SubSir, @luyixiao95, @djw8605, @codex\r\n","reactions":{"url":"https://api.github.com/repos/vllm-project/vllm/releases/385359183/reactions","total_count":17,"+1":7,"-1":0,"laugh":0,"hooray":5,"confused":0,"heart":0,"rocket":5,"eyes":0},"mentions_count":200},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/377027961","assets_url":"https://api.github.com/repos/vllm-project/vllm/releases/377027961/assets","upload_url":"https://uploads.github.com/repos/vllm-project/vllm/releases/377027961/assets{?name,label}","html_url":"https://github.com/vllm-project/vllm/releases/tag/v0.28.0","id":377027961,"author":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"node_id":"RE_kwDOI7xefs4WeP15","tag_name":"v0.28.0","target_commitish":"main","name":"v0.28.0","draft":false,"immutable":false,"prerelease":false,"created_at":"2026-08-24T23:42:42Z","updated_at":"2026-08-26T10:17:05Z","published_at":"2026-08-26T09:46:30Z","assets":[{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/530602995","id":530602995,"node_id":"RA_kwDOI7xefs4foFvz","name":"vllm-0.28.0+cpu-cp312-cp312-macosx_11_0_arm64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":26437437,"digest":"sha256:e8c5a3930367b740914a14420efcc3535da2c2dba5bb23d77221ff81094cc630","download_count":2820,"created_at":"2026-08-26T09:47:59Z","updated_at":"2026-08-26T09:48:01Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.28.0/vllm-0.28.0%2Bcpu-cp312-cp312-macosx_11_0_arm64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/530602992","id":530602992,"node_id":"RA_kwDOI7xefs4foFvw","name":"vllm-0.28.0+cpu-cp38-abi3-manylinux_2_34_aarch64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":66167752,"digest":"sha256:38f37515e58a39eb0f23847387ee3790cb5100d3cdfae40a77d0265f135461d1","download_count":592,"created_at":"2026-08-26T09:47:59Z","updated_at":"2026-08-26T09:48:04Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.28.0/vllm-0.28.0%2Bcpu-cp38-abi3-manylinux_2_34_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/530602994","id":530602994,"node_id":"RA_kwDOI7xefs4foFvy","name":"vllm-0.28.0+cpu-cp38-abi3-manylinux_2_34_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":130133207,"digest":"sha256:71f69f1549ba5832595db2e85a485d067ba13b62c60055340128fbda0842bb97","download_count":2273,"created_at":"2026-08-26T09:47:59Z","updated_at":"2026-08-26T09:48:09Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.28.0/vllm-0.28.0%2Bcpu-cp38-abi3-manylinux_2_34_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/530602993","id":530602993,"node_id":"RA_kwDOI7xefs4foFvx","name":"vllm-0.28.0+cu129-cp38-abi3-manylinux_2_28_aarch64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":521137608,"digest":"sha256:09ae6e121339b3208b628aa17158bf2300dfbee9658fa296e4565b82afec0511","download_count":590,"created_at":"2026-08-26T09:47:59Z","updated_at":"2026-08-26T09:48:32Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.28.0/vllm-0.28.0%2Bcu129-cp38-abi3-manylinux_2_28_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/530602989","id":530602989,"node_id":"RA_kwDOI7xefs4foFvt","name":"vllm-0.28.0+cu129-cp38-abi3-manylinux_2_28_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":548009842,"digest":"sha256:8ec943b66a0c6b4351d0778e99d7bacfca5788dd8eedd49425092bacb61c4397","download_count":5776,"created_at":"2026-08-26T09:47:59Z","updated_at":"2026-08-26T09:48:34Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.28.0/vllm-0.28.0%2Bcu129-cp38-abi3-manylinux_2_28_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/530603038","id":530603038,"node_id":"RA_kwDOI7xefs4foFwe","name":"vllm-0.28.0+xpu-cp38-abi3-manylinux_2_34_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":29994280,"digest":"sha256:14529b4222c3e1df3a900e41a88ece297cbd656f0a1ea16a993b108496c30eaa","download_count":77,"created_at":"2026-08-26T09:48:02Z","updated_at":"2026-08-26T09:48:04Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.28.0/vllm-0.28.0%2Bxpu-cp38-abi3-manylinux_2_34_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/530603096","id":530603096,"node_id":"RA_kwDOI7xefs4foFxY","name":"vllm-0.28.0-cp38-abi3-manylinux_2_28_aarch64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":308873410,"digest":"sha256:817b8181f7f61b4a62dc1d5d9ab39f2bfb60a6cb86c29879a78a147b85756787","download_count":9720,"created_at":"2026-08-26T09:48:05Z","updated_at":"2026-08-26T09:48:23Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.28.0/vllm-0.28.0-cp38-abi3-manylinux_2_28_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/530603106","id":530603106,"node_id":"RA_kwDOI7xefs4foFxi","name":"vllm-0.28.0-cp38-abi3-manylinux_2_28_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":314500103,"digest":"sha256:addb0ffdaafd8155d75e9b3f5ddb3da28fdee9e8a7097ede91f7db2e9e1a3889","download_count":9892,"created_at":"2026-08-26T09:48:05Z","updated_at":"2026-08-26T09:48:23Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.28.0/vllm-0.28.0-cp38-abi3-manylinux_2_28_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/530603202","id":530603202,"node_id":"RA_kwDOI7xefs4foFzC","name":"vllm-0.28.0.tar.gz","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/x-gtar","state":"uploaded","size":39991441,"digest":"sha256:ac96dd0ec5be9c13f2aa4bfb50498c4727110406f6e4980b58a4097e4d18634a","download_count":311,"created_at":"2026-08-26T09:48:10Z","updated_at":"2026-08-26T09:48:13Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.28.0/vllm-0.28.0.tar.gz"}],"tarball_url":"https://api.github.com/repos/vllm-project/vllm/tarball/v0.28.0","zipball_url":"https://api.github.com/repos/vllm-project/vllm/zipball/v0.28.0","body":"# v0.28.0\r\n\r\n## Highlights\r\n\r\nThis release features 584 commits from 270 contributors (76 new)!\r\n\r\n* **Kimi-K3 performance push**: a major optimization effort for Kimi-K3 across the stack — Decode Context Parallel (DCP) support (#50484), fused FlashKDA decode and prefill kernels (#50654, #51311, #52458), SiTU activation support for MegaMoE (#50510), GEMM-RS for sequence parallelism (#52079), combined all-gathers with 1.5~3x kernel-level speedup (#51070), an adaptive speculative token budget delivering ~60% better DSpark TTFT (#51725), and optional shared-expert sharding saving ~17 GiB of memory per GPU (#50912). Kimi-K3 also now runs on ROCm with the V2 model runner (#51653).\r\n* **DeepSeek V4**: sparse MLA now works end-to-end for plain decode, MTP, and DSpark speculative decoding (#51538), joined by AMD Quark NVFP4 support (#47972), reasoning-effort prompts and mappings (#50580), sparse top-k metadata kernel optimizations (#52084, #51967), narrowed eager CUDA graph regions (#51430, #52401), and ROCm enablement on gfx11 and gfx950 (#47017, #52212).\r\n* **Speculative decoding advances**: DFlash2 with local convolution and a candidate selector (#52816), DSpark confidence-scheduled verification (#47808), and async scheduling auto-enabled for draft models (#48341).\r\n* **Model Runner V2 maturation**: E/P/D disaggregation (#38390), weight offloading (#51413), multi-layer MTP KV cache support (#50062), encoder CUDA graphs (#49852), decoder token-wise pooling (#50931) plus Transformers pooling models (#52425), attention-free models (#52374), and `thinking_token_budget` support (#46727).\r\n* **Tiered KV cache offloading**: disk offloading support (#49644), out-of-tree secondary tier managers via `module_path` (#51007), partial secondary-tier load results (#50321), tiering metrics (#48798), and a canonical CPU layout for parallelism-agnostic offload (#48414).\r\n* **Rust frontend & gRPC**: a standalone renderer (#50289), multimodal image inference over gRPC (#50368), explicit data-parallel rank routing (#51178), and RL lifecycle control (#51316), with protobuf schemas now published to Buf (#51276).\r\n* **New defaults**: `max_num_batched_tokens` raised from 8192 to 16384 (#51726), prefix caching enabled by default for Mamba models (#50991), and the Blackwell CUDA graph capture default raised to 1024 (#49390).\r\n* **Breaking changes**: bitsandbytes support migrated to an out-of-tree plugin (#43529); Transformers bumped to 5.15.0 (#51668); the deprecated `calculate_kv_scales` runtime KV scale calculation was removed (#49389); `override_attention_dtype` was removed (#48684).\r\n\r\n## Release Artifacts\r\n\r\n### Python Wheels\r\n\r\n| Platform | Install |\r\n|---|---|\r\n| PyPI (CUDA 13.0) | `pip install vllm` |\r\n| PyPI (CUDA 13.0, uv) | `uv pip install vllm --torch-backend=auto` |\r\n| ROCm | `pip install vllm --extra-index-url https://wheels.vllm.ai/rocm/0.28.0/rocm722` |\r\n\r\n### Docker Images\r\n\r\n| Platform | Docker Image |\r\n|---|---|\r\n| CUDA 13.0 (Default) | `docker pull vllm/vllm-openai:v0.28.0` (`v0.28.0-cu130` also works) |\r\n| CUDA 12.9 | `docker pull vllm/vllm-openai:v0.28.0-cu129` |\r\n| CUDA 13.0 + Ubuntu 24.04 | `docker pull vllm/vllm-openai:v0.28.0-ubuntu2404` |\r\n| CUDA 12.9 + Ubuntu 24.04 | `docker pull vllm/vllm-openai:v0.28.0-cu129-ubuntu2404` |\r\n| ROCm | `docker pull vllm/vllm-openai-rocm:v0.28.0` |\r\n| CPU | `docker pull vllm/vllm-openai-cpu:v0.28.0` |\r\n| XPU | `docker pull vllm/vllm-openai-xpu:v0.28.0` |\r\n\r\n### Other Artifacts\r\n\r\nPre-built release artifacts are available in the **Assets** section at the bottom of this page, including:\r\n- Source distribution tarball\r\n- CUDA 12.9 Python wheels for x86_64 and arm64\r\n- CUDA 13.0 Python wheels for x86_64 and arm64\r\n- CPU Python wheels for x86_64, arm64, and macOS\r\n\r\n## Model Support\r\n* **New models**: Muse Glimmer (#51655), Ling 3.0 Flash with BF16, MTP, and parser support (#51045) plus an FP8 variant (#51265) and hybrid MXFP4 routed experts (#52114), Dots3 NOTE native multimodal support (#51255), and Interns2mobius (#51149).\r\n* **Qwen**: Qwen3.8 enabled on AMD ROCm (#50068), fused CUDA post-conv MTP decode kernel for Qwen3.5 GDN (#51674), GDN gates aligned with speculative tokens (#51812), and Qwen3.5 fixes for text-only checkpoints (#50734, #50355).\r\n* **Transformers modeling backend**: MLA support (#48250), hardware-agnostic model definition (#49458), fully generalized input embedding handling (#51247), logit softcapping (#52173), and a hardened multimodal path (#51408, #51657).\r\n* **LoRA**: vision tower LoRA for Gemma4 (#42662), tower/connector LoRA for Keye (#51780) and Ultravox (#48215).\r\n* **Vision encoders**: ViT full CUDA graph for Kimi-K2.5 (#50929) and Ernie-4.5-VL (#45254, #51461), torch.compile for the Qwen3-VL encoder (#40116), and long-blocking H2D copies avoided in ViT (#51841).\r\n* **MoE**: extended EPLB support for Mistral Large 3 and additional MoE backends (#48355), CuTe DSL skinny GEMM extended to GLM-5.2 (#49791).\r\n* **Speculative decoding coverage**: EAGLE3 support declared on KimiLinear (#52171), Qwen3.6 dSpark acceptance coverage (#51310).\r\n* **Correctness**: MiniMax-M3 NVFP4 inference (#48929) and compressed-tensors FP8 MoE SwiGLU params (#46845), Gemma3n/Gemma4 variable-length audio batch padding (#50958), Gemma 4 compatibility with the upcoming Transformers version (#49797), and a Qwen3-Omni crash on video without an audio track (#48420).\r\n* **Multimodal performance**: fused on-device multimodal preprocess normalization (#50411), faster placeholder and token-match scanning (#50716), and repeated prompt-update scans avoided (#51774).\r\n\r\n## Engine Core\r\n* **Speculative decoding**: DSpark confidence-scheduled verification (#47808), top-k DSpark Markov projection (#49969), DFlash2 with local convolution and a candidate selector (#52816), async scheduling auto-enabled for draft models (#48341), fused MTP trailing all-reduce with local-argmax draft tokens (#49793), and an adaptive budget for speculative scheduled input tokens (#51725).\r\n* **KV cache & scheduling**: per-request scheduling for MLA chunked context (#50613), partial-tail prefix reuse with fine-grained prefix matching (#50507), backend-published KV packing in the KV-cache layout refactor (#51612, #51704), LIFO free-block reuse order restored when prefix caching is off (#51482), and silent request skipping in priority scheduling fixed (#49206).\r\n* **Performance**: continued elimination of GPU<->CPU syncs on the execution path (#51458, #51738, #52369) now guarded by a CI sync check (#43107), new JIT warmup infrastructure with predicate filtering (#49315), the top-k/top-p Triton sampler launched with 8 warps (#51507), detokenization skipped in offline beam search (#50333), Mask Replay (#49577), optimized long-context MLA cache gathers (#51739), and HF revisions resolved to a commit hash once per model load (#49990).\r\n* **Hybrid/Mamba**: prefix caching on by default (#50991), the final part of the Mamba attention module refactor (#44857), 3D-grid tiling of the state-copy Triton kernels (#49436), and Mamba alignment applied before encoder caps (#51603).\r\n* **RL workflows**: stateful trainer send over NCCL and sparse NCCL (#50902), `CuMemAllocator.discard()` for tag-selective GPU memory release (#52514), level-2 sleep/wake/reload fixed with LoRA enabled (#39935), and rewritten weight-transfer docs with standardized examples (#51729).\r\n* **Startup robustness**: file:// rendezvous for single-node executors eliminates startup port races (#50999, #51652), frontend processes are watched during engine startup (#43417), a `get_open_port()` livelock on DP-reserved ports was fixed (#50965), and NVML is no longer re-initialized on every device-capability check (#50393).\r\n\r\n## Hardware & Performance\r\n* **NVIDIA**: FlashInfer XQA decode support on SM12x (#49718), a CuTeDSL fused query kernel on SM100 (#49792), programmatic dependent launch for the DSA decode kernels (#50230), the native DSA decode path for MTP=3 on SM90 (#52164), GB10 fused-MoE FP8 tuning configs (#52502), and B12X dense linear backends (#52016).\r\n* **AMD ROCm**: torch 2.12 / triton 3.7 stack bump (#50607), AITER and FP8 inference enabled on GFX120x (#43615), DeepSeek-V4 on gfx11 (#47017), optimized Triton sparse-MLA decode on gfx950 (#52212), FlyDSL decode-attention kernel for 4-bit TurboQuant KV cache (#47896) and an fp8 MQA logits kernel on gfx942 (#49544), a fused Kimi-K3 KDA decode kernel (#50654), fused bf16→fp32 router GEMM (#50268), pinned memory on supported WSL2 kernels (#50126), and preshuffled sparse indexing for 16-token blocks (#51216).\r\n* **Intel XPU**: a torch linear backend including blockwise GEMM (#49664, #50826), MXFP8 linear weights for the INC DeepSeek V4 model (#48476), async-scheduling PP sampled-token broadcast overlapped with compute (#51650), an XPU wheel added to the release pipeline (#52108), tuned Mamba SSU configs for Arc Pro B70 (#50534), and UVA weight offloading fixes (#51770).\r\n* **CPU**: an MLA backend so DeepSeek-V2/V3 can run on CPU (#49453), a triton-cpu wheel (#52092), GPTQ and AWQ enabled on s390x (#51148) along with tcmalloc (#50841), BF16 MoE routed through zentorch on AMD (#44201), an unquantized MoE backend for Power (VSX) (#51624), unquantized MoE migrated to the modular-kernel experts structure (#50133), and the MXFP4 block scale folded in 2 instructions instead of 4 (#51583).\r\n\r\n## Large Scale Serving\r\n* **E/P/D disaggregation**: Model Runner V2 E/P/D support (#38390), duplicate image preprocessing removed with GPU-side preprocessing (#50390), KV consumers may omit multimodal embeddings (#52697), encoder-instance requests kept alive until their images are encoded (#50275), and EC connector scheduler/worker metadata plumbing (#49579, #49585).\r\n* **KV offloading**: disk offloading for SimpleCPUOffloadConnector (#49644), out-of-tree secondary tier managers via `module_path` (#51007), partial secondary-tier load results (#50321), tiering offloading metrics (#48798), data-parallel topology exposed to offloading backends (#51879), a canonical CPU layout for parallelism-agnostic offload (#48414), and quadratic ARC batch eviction avoided (#50992).\r\n* **Mooncake**: store group semantics (#44956), tenant ID support (#48069), and official wheels in the Docker image (#51067).\r\n* **Connectors**: transfer mode (push/pull) included in the NIXL compatibility hash (#50620), a MoRIIO per-layer READ-completion barrier (#48534), 2P2D wide-EP with mori-ep/mori-io at dp=ep=16 (#45043), and stale remote cleanup in the push connector (#50234).\r\n* **Parallelism**: EPLB balancedness calculation fixed and tested (#51813), dense multinode DP rescope (#49212).\r\n\r\n## Quantization\r\n* **Online quantization**: online MXFP4 support (#49347), online weight scales shared across TP (#49764), precision preserved in online NVFP4 expert packing (#50029), and the online NVFP4 MoE kernel reused across reloads (#50074).\r\n* **NVFP4**: batch-invariant NVFP4 MoE via CUTLASS (#40372), KV 4-over-6 scale search (#45187), CuTeDSL MoE with SwiGLU-OAI and ReLU2 activations (#47106), and out_dtype matched to the model dtype (#48861).\r\n* **New kernels**: block-wise scaled_mm (#49932), DeepSeek-V4 AMD Quark NVFP4 with an emulation kernel (#47972).\r\n* **Fixes**: dynamic INT8 W8A8 MoE config no longer built as W8A16 (#50833) and a TritonExperts crash (#51411), MXFP4 conversion for FlashInfer CUTLASS (#51038), fp32 weight scales and per-expert checkpoint mapping for MXFP4 (#51419), and fused block-scale orientation (#50727).\r\n\r\n## API & Frontend\r\n* **New capabilities**: request priority parsed from an HTTP header (#51089), session ID plumbing into requests (#48048), `count_reasoning_tokens` in the streaming parser engine (#45802), `content_parts` on `/inference/v1/generate` (#51478), `model` optional on all `/derender` request classes (#51463), output token IDs logged at DEBUG level (#52098), and vLLM Recipes connected to native config-based deployment and benchmarking (#51308, #51878).\r\n* **Rust frontend**: a standalone renderer (#50289), gRPC multimodal image inference (#50368), explicit data-parallel rank routing (#51178), RL lifecycle control (#51316), dynamic tools from developer messages (#51144), protobuf schemas published to Buf (#51276), and MiniJinja upgraded to 2.22 (#51235).\r\n* **Anthropic API**: 4xx returned for client-caused errors on `/v1/messages` (#52246), `disable_parallel_tool_use` preserved (#52021), and stop sequences bounded (#51997).\r\n* **Cohere**: upstreamed parser fixes (#51998), stop sequences reported correctly (#51556), and vectorized binary embedding bit-packing (#52277).\r\n* **Structured output**: request stop tokens masked in xgrammar until the grammar terminates (#49227, #50595), NUL bytes rejected in `structured_outputs.regex` (#51796), negative token IDs rejected as out-of-vocabulary (#51795), and `VLLMValidationError` raised from validators (#52394).\r\n* **Robustness**: uvicorn signal handlers disabled instead of racing them (#50916), a consolidated entrypoint exception handler (#52261), 400 instead of 500 on non-object JSON bodies (#51654, #52528), generation inputs bounded before expensive work (#51447), and `cache_salt` now required to be non-empty (#50816).\r\n\r\n## Security\r\n* Fixed a DoS via sample-rate forgery that bypassed the audio decode duration guard (#49948); the audio decode duration limit is now also enforced in NanoNemotronVL (#50221).\r\n* DeepStream classified as a GPU backend with pixel limits enforced (#50755).\r\n* `_load_ov2_processor` guarded with `resolve_trust_remote_code` (#52952).\r\n* Documentation now warns that `--api-key` does not gate all endpoints (#51999).\r\n\r\n## Dependencies\r\n* Transformers 5.15.0 (#51668), huggingface-hub 1.27.0 (#51422), fastsafetensors upgrade (#50827).\r\n* ROCm: torch 2.12, triton 3.7, torchaudio, torchvision (#50607).\r\n* Runtime image upgraded to Ubuntu 24.04, picking up rdma-core > 44 (#51058).\r\n* DeepGEMM pinned to the deepseek-ai nv_dev tip (#52035), DeepEP pinned by full commit hash (#52028), FlashAttention 3 built with the torch stable API (#49599).\r\n\r\n## Breaking Changes & Deprecations\r\n* bitsandbytes support is now an out-of-tree plugin (#43529).\r\n* The deprecated `calculate_kv_scales` runtime KV scale calculation was removed (#49389).\r\n* `override_attention_dtype` was removed (#48684).\r\n* `reasoning_content` output removal is documented as a breaking client change (#50624).\r\n* KV offload tiering metrics renamed from `kv_offload_tiering_block_{queries,hits}` to `..._chunk_...` (#52812).\r\n* MoE legacy code removed (#51078).\r\n\r\n## New Contributors\r\n* @acheamponge made their first contribution in https://github.com/vllm-project/vllm/pull/49353\r\n* @acmore made their first contribution in https://github.com/vllm-project/vllm/pull/51259\r\n* @anhtra3889 made their first contribution in https://github.com/vllm-project/vllm/pull/51002\r\n* @anujbolewar made their first contribution in https://github.com/vllm-project/vllm/pull/50867\r\n* @arthurgao2003 made their first contribution in https://github.com/vllm-project/vllm/pull/48215\r\n* @baodii made their first contribution in https://github.com/vllm-project/vllm/pull/51215\r\n* @bitborne made their first contribution in https://github.com/vllm-project/vllm/pull/44956\r\n* @bohnstingl made their first contribution in https://github.com/vllm-project/vllm/pull/49458\r\n* @ccaadaro made their first contribution in https://github.com/vllm-project/vllm/pull/51583\r\n* @cmiyai made their first contribution in https://github.com/vllm-project/vllm/pull/46870\r\n* @d4l3k made their first contribution in https://github.com/vllm-project/vllm/pull/51097\r\n* @dmai-afk made their first contribution in https://github.com/vllm-project/vllm/pull/51120\r\n* @dmholtz made their first contribution in https://github.com/vllm-project/vllm/pull/50977\r\n* @efschu made their first contribution in https://github.com/vllm-project/vllm/pull/50734\r\n* @fangchenli made their first contribution in https://github.com/vllm-project/vllm/pull/52277\r\n* @fanxingran made their first contribution in https://github.com/vllm-project/vllm/pull/51011\r\n* @fatday made their first contribution in https://github.com/vllm-project/vllm/pull/48171\r\n* @fattchris made their first contribution in https://github.com/vllm-project/vllm/pull/48861\r\n* @fcui-amd made their first contribution in https://github.com/vllm-project/vllm/pull/50126\r\n* @fede-kamel made their first contribution in https://github.com/vllm-project/vllm/pull/50624\r\n* @fxfxfxfxfxfxfxfx made their first contribution in https://github.com/vllm-project/vllm/pull/49139\r\n* @gabriel-peracio made their first contribution in https://github.com/vllm-project/vllm/pull/50183\r\n* @gchinora made their first contribution in https://github.com/vllm-project/vllm/pull/51427\r\n* @guanxingithub made their first contribution in https://github.com/vllm-project/vllm/pull/50589\r\n* @haregali made their first contribution in https://github.com/vllm-project/vllm/pull/50716\r\n* @hsusul made their first contribution in https://github.com/vllm-project/vllm/pull/49613\r\n* @iwannagotobed made their first contribution in https://github.com/vllm-project/vllm/pull/51664\r\n* @jacobzhang22 made their first contribution in https://github.com/vllm-project/vllm/pull/38771\r\n* @jairitAge made their first contribution in https://github.com/vllm-project/vllm/pull/51100\r\n* @jamesETsmith made their first contribution in https://github.com/vllm-project/vllm/pull/51216\r\n* @jimmy-adams made their first contribution in https://github.com/vllm-project/vllm/pull/47972\r\n* @jyan-R made their first contribution in https://github.com/vllm-project/vllm/pull/52311\r\n* @Kaif10 made their first contribution in https://github.com/vllm-project/vllm/pull/52528\r\n* @karen-sy made their first contribution in https://github.com/vllm-project/vllm/pull/48048\r\n* @Lin-z-w made their first contribution in https://github.com/vllm-project/vllm/pull/48069\r\n* @liushujia122 made their first contribution in https://github.com/vllm-project/vllm/pull/51780\r\n* @Luosuu made their first contribution in https://github.com/vllm-project/vllm/pull/48789\r\n* @mispa-ms made their first contribution in https://github.com/vllm-project/vllm/pull/52419\r\n* @mkhazraee made their first contribution in https://github.com/vllm-project/vllm/pull/50321\r\n* @NVShreyas made their first contribution in https://github.com/vllm-project/vllm/pull/50911\r\n* @pavelzak made their first contribution in https://github.com/vllm-project/vllm/pull/52502\r\n* @positive666 made their first contribution in https://github.com/vllm-project/vllm/pull/52329\r\n* @rajfirke made their first contribution in https://github.com/vllm-project/vllm/pull/51573\r\n* @Rapisurazurite made their first contribution in https://github.com/vllm-project/vllm/pull/50950\r\n* @rchalamala made their first contribution in https://github.com/vllm-project/vllm/pull/50487\r\n* @RobbieJ made their first contribution in https://github.com/vllm-project/vllm/pull/49328\r\n* @ruirui6946 made their first contribution in https://github.com/vllm-project/vllm/pull/52098\r\n* @RyanJHamby made their first contribution in https://github.com/vllm-project/vllm/pull/48420\r\n* @samuelkim7 made their first contribution in https://github.com/vllm-project/vllm/pull/50333\r\n* @shanewidanagama made their first contribution in https://github.com/vllm-project/vllm/pull/51901\r\n* @shikamd123 made their first contribution in https://github.com/vllm-project/vllm/pull/45043\r\n* @SilenNaihin made their first contribution in https://github.com/vllm-project/vllm/pull/39935\r\n* @skysnow2001 made their first contribution in https://github.com/vllm-project/vllm/pull/43615\r\n* @Sundaresan-G made their first contribution in https://github.com/vllm-project/vllm/pull/50526\r\n* @syedalijaseem made their first contribution in https://github.com/vllm-project/vllm/pull/47692\r\n* @taking-lying-flat made their first contribution in https://github.com/vllm-project/vllm/pull/51391\r\n* @tandixit95 made their first contribution in https://github.com/vllm-project/vllm/pull/50462\r\n* @Tejas-Raj01 made their first contribution in https://github.com/vllm-project/vllm/pull/49206\r\n* @theminghuang made their first contribution in https://github.com/vllm-project/vllm/pull/51482\r\n* @tobymao made their first contribution in https://github.com/vllm-project/vllm/pull/51318\r\n* @TrainToGPB made their first contribution in https://github.com/vllm-project/vllm/pull/50958\r\n* @tzulingk made their first contribution in https://github.com/vllm-project/vllm/pull/49230\r\n* @UgaTheDev made their first contribution in https://github.com/vllm-project/vllm/pull/51627\r\n* @varoudis made their first contribution in https://github.com/vllm-project/vllm/pull/50404\r\n* @Vegetog made their first contribution in https://github.com/vllm-project/vllm/pull/49876\r\n* @vitamin-chaos made their first contribution in https://github.com/vllm-project/vllm/pull/47106\r\n* @wangxian001 made their first contribution in https://github.com/vllm-project/vllm/pull/50276\r\n* @WillZZZy made their first contribution in https://github.com/vllm-project/vllm/pull/46747\r\n* @xiaopusun made their first contribution in https://github.com/vllm-project/vllm/pull/51495\r\n* @xijiaat made their first contribution in https://github.com/vllm-project/vllm/pull/50693\r\n* @xudonlyu made their first contribution in https://github.com/vllm-project/vllm/pull/51682\r\n* @yifjiang made their first contribution in https://github.com/vllm-project/vllm/pull/51218\r\n* @yu-xin-c made their first contribution in https://github.com/vllm-project/vllm/pull/51652\r\n* @zcxGGmu made their first contribution in https://github.com/vllm-project/vllm/pull/50746\r\n* @ziqifan617 made their first contribution in https://github.com/vllm-project/vllm/pull/51879\r\n* @zobinHuang made their first contribution in https://github.com/vllm-project/vllm/pull/52164\r\n\r\n## Contributors\r\n@yewentao256, @AndreasKaratzas, @njhill, @mgoin, @aoshen02, @khluu, @hmellor, @stefankoncarevic, @taneem-ibrahim, @LucasWilkinson, @chaunceyjiang, @jikunshang, @zufangzhu, @bigPYJ1151, @NickLucche, @askliar, @fxmarty-amd, @Rohan138, @gty111, @zhenwei-intel, @Isotr0py, @jeejeelee, @ZJY0516, @wangxiyuan, @BugenZhao, @yma11, @zyongye, @DarkLight1337, @jperezdealgaba, @zhou9402, @connorcarpenter15, @kliuae, @lucifer1004, @Alex-ai-future, @aarushjain29, @TheEpicDolphin, @chaojun-zhang, @pmanczak, @WoosukKwon, @BabyDrangoner, @noooop, @almogtavor, @hongxiayang, @fuscof-ibm, @gau-nernst, @zxd1997066, @tlrmchlsmth, @music-dino, @zexplorerhj, @S1ro1, @jdebache, @sfeng33, @mayuyuace, @ganeshr10, @benchislett, @tzulingk, @gcanlin, @ivanium, @divakar-amd, @R3hankhan123, @sagearc, @vhagor, @ronensc, @micah-wil, @qyYue1389, @vanshbhatia-amd, @hao-aaron, @chengy-sysu, @elvircrn, @taking-lying-flat, @omerpaz95, @Etelis, @vllmellm, @frank-suwen, @KernelClint, @Fangzhou-Ai, @louie-tsai, @simondanielsson, @maxyanghu, @dmai-afk, @KurodaKanbei, @ziqifan617, @bastefaniak, @ECMGit, @haregali, @cmiyai, @fede-kamel, @drakosha, @vineethsaivs, @zcxGGmu, @TQCB, @skysnow2001, @shenoyvvarun, @karen-sy, @fattchris, @RyanJHamby, @shikamd123, @namgyu-youn, @zzt93, @abmfy, @reidliu41, @Rapisurazurite, @tandixit95, @mganczarenko, @yimdev, @anujbolewar, @LiuYinfeng01, @lk-chen, @NVShreyas, @huangzhilin-hzl, @varoudis, @Yejing-Lai, @mkhazraee, @jzakrzew, @TrainToGPB, @waynehacking8, @zixi-qi, @Sundaresan-G, @mindungil, @bitborne, @Wauplin, @jacobzhang22, @zhewenl, @bnellnm, @pisceskkk, @Lin-z-w, @gabriel-peracio, @SilenNaihin, @baodii, @YunzhuLu, @xwu-intel, @BWAAEEEK, @thisjiang, @maobaolong, @anhtra3889, @JaredforReal, @lvhan028, @xiaolong-intel, @andyxning, @cleonard530, @gnovack, @MatthewBonanni, @wangxian001, @lengrongfu, @Tejas-Raj01, @simon-mo, @vitamin-chaos, @arpera, @jairitAge, @jimmy-adams, @ILikeIneine, @woosebastian, @haic0, @edwinlim0919, @fcui-amd, @jhu960213, @jinzhen-lin, @coltonottley, @walterbm, @meenchen, @matteso1, @djramic, @gchinora, @davidjpyu, @tianmu-li, @xiaopusun, @majunze2001, @Vegetog, @puririshi98, @janeyx99, @RobbieJ, @oonyshch, @thegoldenflow, @Srinivasoo7, @fatday, @acheamponge, @efschu, @rajfirke, @fanxingran, @xudonlyu, @lcskrishna, @xijiaat, @GirasoleY, @d4l3k, @samuelkim7, @tarukumar, @acmore, @theminghuang, @khushali9, @wzhao18, @Priyjain-amd, @yiz-liu, @lkm2835, @dmholtz, @Dao007forever, @liushujia122, @LopezCastroRoberto, @UgaTheDev, @tuukkjs, @aditi-amd, @guan404ming, @yiliu30, @zou3519, @Luosuu, @JoursBleu, @varun-sundar-rabindranath, @mpashkovskii, @yu-xin-c, @WillZZZy, @vrdn-23, @xyang16, @ccrhx4, @tanpinsiang, @russellb, @fxfxfxfxfxfxfxfx, @afriedri, @yifjiang, @Akashcodes732, @HF-001, @ovidiusm, @arthurgao2003, @TomerBN-Nvidia, @hotTea123, @vx120, @bohnstingl, @qwerqwerqwe8688-jpg, @jasonozuzu-cohere, @vineetatiwari27, @ruirui6946, @linitra24, @syedalijaseem, @nickus, @yzong-rh, @s3woz, @jhaotingc, @lukealonso, @Jie-Fang, @kzwrime, @xianbaoqian, @velonica0, @ccaadaro, @yisustc, @fangchenli, @iwannagotobed, @zobinHuang, @rchalamala, @shanjiaz, @jamesETsmith, @stacyroberts, @guanxingithub, @biswapanda, @shanewidanagama, @UranusSeven, @hsusul, @tobymao, @mispa-ms, @jeffreywang88, @SayHelloToWorld, @jyan-R, @oops-oom, @shantipriya-amd, @andakai, @akii96, @shen-shanshan, @Kaif10, @yitingdc, @positive666, @pavelzak, @SubSir, @ywang96\r\n","reactions":{"url":"https://api.github.com/repos/vllm-project/vllm/releases/377027961/reactions","total_count":47,"+1":12,"-1":0,"laugh":0,"hooray":12,"confused":0,"heart":10,"rocket":13,"eyes":0},"mentions_count":200},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/368501615","assets_url":"https://api.github.com/repos/vllm-project/vllm/releases/368501615/assets","upload_url":"https://uploads.github.com/repos/vllm-project/vllm/releases/368501615/assets{?name,label}","html_url":"https://github.com/vllm-project/vllm/releases/tag/v0.27.1","id":368501615,"author":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"node_id":"RE_kwDOI7xefs4V9uNv","tag_name":"v0.27.1","target_commitish":"main","name":"v0.27.1","draft":false,"immutable":false,"prerelease":false,"created_at":"2026-08-11T08:12:11Z","updated_at":"2026-08-11T10:49:10Z","published_at":"2026-08-11T10:47:49Z","assets":[{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/510015744","id":510015744,"node_id":"RA_kwDOI7xefs4eZjkA","name":"vllm-0.27.1+cpu-cp312-cp312-macosx_11_0_arm64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":25808613,"digest":"sha256:a3b537892ebc98e97132101aa2096210842b982e05f976fa337caa3f771f9f0c","download_count":2918,"created_at":"2026-08-11T10:48:39Z","updated_at":"2026-08-11T10:48:42Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.27.1/vllm-0.27.1%2Bcpu-cp312-cp312-macosx_11_0_arm64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/510015743","id":510015743,"node_id":"RA_kwDOI7xefs4eZjj_","name":"vllm-0.27.1+cpu-cp38-abi3-manylinux_2_34_aarch64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":65467207,"digest":"sha256:510021af7d5d953040c3b97a112e50902bad7748b238c5d32675c04d23a29e52","download_count":5652,"created_at":"2026-08-11T10:48:39Z","updated_at":"2026-08-11T10:48:44Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.27.1/vllm-0.27.1%2Bcpu-cp38-abi3-manylinux_2_34_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/510015745","id":510015745,"node_id":"RA_kwDOI7xefs4eZjkB","name":"vllm-0.27.1+cpu-cp38-abi3-manylinux_2_34_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":127738428,"digest":"sha256:36f0e7b2031233ff09e521716723b0e05ab62054c9a9a05d873af43052140f33","download_count":8348,"created_at":"2026-08-11T10:48:39Z","updated_at":"2026-08-11T10:48:47Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.27.1/vllm-0.27.1%2Bcpu-cp38-abi3-manylinux_2_34_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/510015741","id":510015741,"node_id":"RA_kwDOI7xefs4eZjj9","name":"vllm-0.27.1+cu129-cp38-abi3-manylinux_2_28_aarch64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":518782535,"digest":"sha256:d73c897b0bde9e515c30101f694989fb65821a8543e7cad8f95e5440dc6bae81","download_count":872,"created_at":"2026-08-11T10:48:39Z","updated_at":"2026-08-11T10:49:09Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.27.1/vllm-0.27.1%2Bcu129-cp38-abi3-manylinux_2_28_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/510015742","id":510015742,"node_id":"RA_kwDOI7xefs4eZjj-","name":"vllm-0.27.1+cu129-cp38-abi3-manylinux_2_28_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":545751073,"digest":"sha256:bf0d52faa2a51e7a01c6856a7a8a2d1307fd0ff711415d34168a67ffac0fa47b","download_count":13177,"created_at":"2026-08-11T10:48:39Z","updated_at":"2026-08-11T10:49:10Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.27.1/vllm-0.27.1%2Bcu129-cp38-abi3-manylinux_2_28_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/510015798","id":510015798,"node_id":"RA_kwDOI7xefs4eZjk2","name":"vllm-0.27.1-cp38-abi3-manylinux_2_28_aarch64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":307180998,"digest":"sha256:74c1a04548c9016ace8f6c03592edf62517152a0e833f46282556ffbe0796254","download_count":548,"created_at":"2026-08-11T10:48:43Z","updated_at":"2026-08-11T10:49:03Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.27.1/vllm-0.27.1-cp38-abi3-manylinux_2_28_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/510015815","id":510015815,"node_id":"RA_kwDOI7xefs4eZjlH","name":"vllm-0.27.1-cp38-abi3-manylinux_2_28_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":312887766,"digest":"sha256:98e9fc2a1ed8549a733c9d1b242e2002b82367da9e29e37801761438cb3a2670","download_count":3678,"created_at":"2026-08-11T10:48:45Z","updated_at":"2026-08-11T10:49:06Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.27.1/vllm-0.27.1-cp38-abi3-manylinux_2_28_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/510015850","id":510015850,"node_id":"RA_kwDOI7xefs4eZjlq","name":"vllm-0.27.1.tar.gz","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/x-gtar","state":"uploaded","size":39257426,"digest":"sha256:eec2d54d137ac1e59cb4c39226dfee1943eefc8f4788f5821d7300d6acbdb646","download_count":540,"created_at":"2026-08-11T10:48:48Z","updated_at":"2026-08-11T10:48:52Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.27.1/vllm-0.27.1.tar.gz"}],"tarball_url":"https://api.github.com/repos/vllm-project/vllm/tarball/v0.27.1","zipball_url":"https://api.github.com/repos/vllm-project/vllm/zipball/v0.27.1","body":"This is a patch release on top of v0.27.0.\r\n\r\n- Support quantized DSpark Markov heads (#50424)","reactions":{"url":"https://api.github.com/repos/vllm-project/vllm/releases/368501615/reactions","total_count":27,"+1":17,"-1":0,"laugh":0,"hooray":10,"confused":0,"heart":0,"rocket":0,"eyes":0}},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/368180220","assets_url":"https://api.github.com/repos/vllm-project/vllm/releases/368180220/assets","upload_url":"https://uploads.github.com/repos/vllm-project/vllm/releases/368180220/assets{?name,label}","html_url":"https://github.com/vllm-project/vllm/releases/tag/v0.27.0","id":368180220,"author":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"node_id":"RE_kwDOI7xefs4V8fv8","tag_name":"v0.27.0","target_commitish":"main","name":"v0.27.0","draft":false,"immutable":false,"prerelease":false,"created_at":"2026-08-10T10:48:34Z","updated_at":"2026-08-10T21:18:11Z","published_at":"2026-08-10T21:18:11Z","assets":[{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/509268039","id":509268039,"node_id":"RA_kwDOI7xefs4eWtBH","name":"vllm-0.27.0+cpu-cp312-cp312-macosx_11_0_arm64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":25808556,"digest":"sha256:72fc5842b13635ce4f80620fc96adfa60d51bbd7f822cf5b5a4d9e9821fd6690","download_count":355,"created_at":"2026-08-10T21:16:12Z","updated_at":"2026-08-10T21:16:14Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.27.0/vllm-0.27.0%2Bcpu-cp312-cp312-macosx_11_0_arm64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/509268034","id":509268034,"node_id":"RA_kwDOI7xefs4eWtBC","name":"vllm-0.27.0+cpu-cp38-abi3-manylinux_2_34_aarch64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":65467431,"digest":"sha256:9a3e823d214a6ba69fecdf4753623f596007f2f181e4fd275b25426b3b4436b4","download_count":23,"created_at":"2026-08-10T21:16:12Z","updated_at":"2026-08-10T21:16:17Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.27.0/vllm-0.27.0%2Bcpu-cp38-abi3-manylinux_2_34_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/509268037","id":509268037,"node_id":"RA_kwDOI7xefs4eWtBF","name":"vllm-0.27.0+cpu-cp38-abi3-manylinux_2_34_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":127738330,"digest":"sha256:306e4facfb5515de00258d3deacb133eeb2f53f386f7ab2b8b6f49a104e9c703","download_count":1155,"created_at":"2026-08-10T21:16:12Z","updated_at":"2026-08-10T21:16:22Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.27.0/vllm-0.27.0%2Bcpu-cp38-abi3-manylinux_2_34_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/509268035","id":509268035,"node_id":"RA_kwDOI7xefs4eWtBD","name":"vllm-0.27.0+cu129-cp38-abi3-manylinux_2_28_aarch64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":518782486,"digest":"sha256:102b4611cd6010fd7ba060db96a83f4d7bf698b8d60b99cce5b9528c421c0edf","download_count":76,"created_at":"2026-08-10T21:16:12Z","updated_at":"2026-08-10T21:16:45Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.27.0/vllm-0.27.0%2Bcu129-cp38-abi3-manylinux_2_28_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/509268038","id":509268038,"node_id":"RA_kwDOI7xefs4eWtBG","name":"vllm-0.27.0+cu129-cp38-abi3-manylinux_2_28_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":545751056,"digest":"sha256:c1cfee107e76944797b1982502cdf49e22d98b739285ab3971e9d589f841fc1c","download_count":591,"created_at":"2026-08-10T21:16:12Z","updated_at":"2026-08-10T21:16:48Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.27.0/vllm-0.27.0%2Bcu129-cp38-abi3-manylinux_2_28_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/509268075","id":509268075,"node_id":"RA_kwDOI7xefs4eWtBr","name":"vllm-0.27.0-cp38-abi3-manylinux_2_28_aarch64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":307180947,"digest":"sha256:d38f7f738a6c340fa86abaa56287ab126d0c841f9d129fb6eb21056c3a83e40b","download_count":18,"created_at":"2026-08-10T21:16:15Z","updated_at":"2026-08-10T21:16:35Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.27.0/vllm-0.27.0-cp38-abi3-manylinux_2_28_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/509268120","id":509268120,"node_id":"RA_kwDOI7xefs4eWtCY","name":"vllm-0.27.0-cp38-abi3-manylinux_2_28_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":312887666,"digest":"sha256:02d8265e71bab1cf50f93026211c9a75562a7f0a72b3a32ec5e0e8bcdd62ec75","download_count":133,"created_at":"2026-08-10T21:16:18Z","updated_at":"2026-08-10T21:16:38Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.27.0/vllm-0.27.0-cp38-abi3-manylinux_2_28_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/509268159","id":509268159,"node_id":"RA_kwDOI7xefs4eWtC_","name":"vllm-0.27.0.tar.gz","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/x-gtar","state":"uploaded","size":39253011,"digest":"sha256:862671641625d6cfb7aaab98a5386126738b59ca50e5d01ca539a4e32f63ffcb","download_count":54,"created_at":"2026-08-10T21:16:23Z","updated_at":"2026-08-10T21:16:27Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.27.0/vllm-0.27.0.tar.gz"}],"tarball_url":"https://api.github.com/repos/vllm-project/vllm/tarball/v0.27.0","zipball_url":"https://api.github.com/repos/vllm-project/vllm/zipball/v0.27.0","body":"# vLLM v0.27.0 Release Notes\r\n\r\n## Highlights\r\n\r\nThis release features 561 commits from 242 contributors (64 new)!\r\n\r\n* **Kimi K3 support** with a full stack landing in one release: core model files and kernels (#50089, #50000), Python (#50093) and Rust (#50104) frontends, AttnRes kernels (#50090), DeepGEMM support (#50458), compressed-tensors quantized checkpoints (#50500), DSpark AR fusion (#50242), and an option to shard the shared expert instead of replicating it (#50656).\r\n* **More new models**: Qwen3.5 text-only dense and MoE models (#50210) with EVS video token pruning (#48912), K-EXAONE-2.0-750B-A37B (#50524), VaultGemma via the Transformers modeling backend (#49803), and jina-embeddings-v5-text-nano (#50688).\r\n* **PyTorch 2.13.0 upgrade** along with torchvision 0.28.0 and Triton 3.7.1 (#48155) — this is a breaking environment change; XPU (#48677) and CPU (#50412) followed to torch 2.13 as well.\r\n* **FlashAttention 4 integration deepens on SM100**: FP8 KV cache support (#42569) and headdim-256 support (#42669), backed by a new JIT warmup infrastructure (#47451) and runner-owned Triton kernel warmup (#49903) that remove first-request compilation stalls.\r\n* **DeepSeek-V4 performance push**: sequence parallelism (#46789), ~2x kernel improvement by skipping empty c128 launches (#48957), 3.4% E2E TTFT from skipping unneeded topk/router (#49486), 3.9% E2E TTFT from workspace reuse (#49236), 1.88x kernel from removing a redundant full kernel (#50298), adaptive topk width (1.0% E2E, #50004), 448 MiB GPU memory saved in the PP buffer (#50312), a compact MXFP4 indexer KV cache (#48993), and removal of sparse-MLA q-head padding on FlashInfer >= 0.6.14 (#48047).\r\n* **Model Runner V2 expands to non-generative workloads**: encoder-only attention (#49331), sequence pooling for embedding/classification (#48791), encoder token classification (#50293) and token embedding (#50574), BGE-M3 pooling (#50661), multimodal on CPU (#50073), a multi-layer MTP speculator (#48892), and PCP now selects MRV2 (#50034).\r\n* **Resilient large-scale serving**: a (simplified) fault tolerance framework for DP+EP external load-balancer deployments (#44428) and async preparation for elastic EP scaling (#47288).\r\n* **Disaggregation for hybrid models**: NIXL P/D for hybrid MLA+SSM models (#49762), heterogeneous P/D block sizes for hybrid models (#49612), and MoRIIO heterogeneous TP<->DP prefill/decode read routing (#46116).\r\n* **Rust frontend grows a gRPC control plane**: engine-aware health reporting (#48992), abort control (#49255), server and model discovery (#49491), KV event source discovery (#50033), plus `vllm-bench` integrated into the `vllm` CLI (#48930).\r\n* **Early next-gen hardware enablement**: `sm_107` target for NVIDIA Rubin (#49387) with NVLink all-reduce paths on SM107 (#49647), and ROCm gfx1250 architecture enabled (#46516).\r\n\r\n### Model Support\r\n* Kimi K3: new model (#50000) with model files and kernels (#50089), Python frontend (#50093), Rust frontend (#50104), AttnRes kernels (#50090), DeepGEMM support (#50458), DSpark AR fusion (#50242), and optional shared-expert sharding (#50656).\r\n* New models: Qwen3.5 text-only dense and MoE (#50210), K-EXAONE-2.0-750B-A37B (#50524), VaultGemma via Transformers backend (#49803), jina-embeddings-v5-text-nano with EuroBERT encoder backbone (#50688).\r\n* Inkling: llm-compressor NVFP4 weights (#49258) and compressed-tensors dynamic FP8 (#48876).\r\n* Multimodal: VidCom2 video token pruning (#47750), EVS for Qwen3.5 (#48912), ViT CUDA graph for Gemma-4 (#46837), Cosmos3 FP8 ModelOpt/Diffusers remapping (#48952), MiniMax-M3 MSA speculative decode verification (#50032) and default video processor (#50305), DeepSeek-OCR-2 TTFT optimization (#49531), longer max audio duration for MOSS-TD (#49403).\r\n* Diffusion models: top_k and top_p sampling for DiffusionGemma (#45429).\r\n* Transformers modeling backend: audio model support (#39330), improved `fx` tracer (#49957), fused residual-add + RMSNorm compilation pass (#48757), and fixes for MLA padding + grouped topk routing (#49982), MQA with TP (#49987), and Qwen3-VL M-RoPE (#49292).\r\n\r\n### Engine Core\r\n* Warmup: new JIT warmup infrastructure (#47451), runner-owned Triton kernels warmed before the first request (#49903), and proper renderer warmup (#50408).\r\n* Attention: FlashAttention 4 SM100 FP8 KV cache (#42569) and headdim-256 (#42669); query replication for MLA decode under DCP for DeepSeek-V2/R1 and Kimi-K2.5 (#45964); masked MHA for sparse MLA prefills (#48770); skip sparse indexer scoring for short dense prefills (#48407); FlexAttention epilogue hook (#45841) and encoder block-mask compile explosion avoided (#50339); attention backends stay eligible for text-only serving of prefix-LM models (#48796); merge-attention context count as a runtime argument (#48739); unified multi-path encoder CUDA graph support (#49934); encoder cache extension hooks (#48218).\r\n* Model Runner V2: encoder-only attention (#49331), sequence pooling for embedding and classification (#48791), encoder token classification (#50293), encoder token embedding (#50574), BGE-M3 pooling (#50661), multimodal on CPU (#50073), multi-layer MTP speculator (#48892), PCP selects MRV2 (#50034), attention metadata always built at capture time (#49364), encoder cache profiling (#47985), skipped no-op FP32 logits materialization (#47711), chunked rejection sampler to avoid OOM (#48630), fewer GPU<->CPU syncs in hybrid Mamba (#49736).\r\n* KV offloading: generic P2P secondary tier with peer lookup/serving (#48021), per-request tier filtering with TierFilter/TierMatcher (#48123), self-describing KV events with TieringOffloadingSpec (#48679), pluggable eviction policies via CachePolicyFactory (#49114), deduplicated replicated MLA KV in the shared CPU region (#48906), single-copy MLA layout for CPUOffloadingSpec (#50301), CPUOffloadingSpec moved onto SharedOffloadRegion (#50094), TP-independent compact secondary identity (#49858), batched C store/load for filesystem offload (#49152), reliable partial-tail offload for sub-block prompts (#49502), per-layer canonical KV page mappings for parallelism-agnostic offload (#48408).\r\n* Mamba/hybrid: ReplaySSM caching for faster Mamba2 standard decode (#48018), fused align-mode DS-conv state migration with num_accepted_tokens > 1 (#49291), FlashInfer Mamba SSU algorithm selection (#50157), fixed `/wake_up` crash on hybrid models (#41602).\r\n* Spec decode: DSpark Markov head replicated across TP ranks (#49731), `sample_from_anchor` loaded from speculators config (#48639), earliest-completing stop string selected (#49391).\r\n* Structured outputs: grammar advanced across the reasoning boundary with spec decode (#44993).\r\n* Preprocessing performance: derender CPU work offloaded to the renderer thread pool (#49396), raw-prompt preprocessing off the event loop in AsyncLLM (#49608), multimodal preprocessing isolated on its own executor (#49524), MM embeds loading deferred off the event loop (#49477), parallel preprocessing within a request for online pooling (#49153), videos hashed by source bytes (#49607), original image mode preserved in ImageIO (#49159).\r\n* RL: weight version tagging for RL rollouts (#49040), stateful trainer-send IPC (#48981), vLLM config set during weight reload (#45989), router replay output from the FlashInfer monolithic MoE kernel (#44214).\r\n* Memory & robustness: CuMem slept-L1 fragmentation accounting (#49208), cgroup memory limits respected on all platforms (#49966), fail fast when /dev/shm is too small (#48879), zero-copy tensor pickling in shm_broadcast (#48442), LRU hash-split skipped in free_blocks when prefix caching is off (#48017), location-derived path vars excluded from torch.compile cache factors (#47573), `CustomOp.forward_native` compiled for ReLU^2 (#50244), HF config used for HF tokenizers (#49907), batch-invariant RMSNorm via pinned block size (#48391).\r\n\r\n### Hardware & Performance\r\n* DeepSeek-V4: sequence parallelism (#46789), ~2x kernel skipping empty c128 launches (#48957), 3.4% E2E TTFT skipping topk/router in decode (#49486), 3.9% E2E TTFT workspace reuse (#49236), 1.88x kernel removing a redundant full kernel (#50298), adaptive topk width 1.0% E2E (#50004), 448 MiB GPU memory saved (#50312), compact MXFP4 indexer KV cache (#48993), sparse-MLA q-head padding removed for FlashInfer >= 0.6.14 (#48047).\r\n* Kernels: RMSNorm uncontiguous support with 1.2–3.1x kernel improvement (#49750), MoE `reduce_scatter` regression fix restoring 5% E2E throughput (#48763), non-grouped bias-less topk routing dispatched to the fused path (#49618), tuned LL BF16 router GEMM (#48774) with warmup skipped for non-MoE models (#49659), Triton tensor-descriptor path for fused MoE via `VLLM_TRITON_USE_TD` (#42436), cudagraph/DP padding skipped in topk (#48979), coalesced HBM access in the Marlin INT4-FP8 AWQ preprocess kernel (#47268).\r\n* NVIDIA next-gen: `sm_107` for Rubin (#49387), NVLink all-reduce paths on SM107 (#49647), fixed CUDA arch detection producing kernel-less builds on SM121 (#49904).\r\n* ROCm: gfx1250 architecture enabled (#46516), AITER FP8 ViT encoder attention (#49937), fused shared expert for Quark DeepSeek-V4 checkpoints (#48044), Quark GLM-5.2 checkpoint inference fixes (#48886), DSv3.2 per-decode FillFunctor launches eliminated in the sparse-MLA hot loop (#44527), B-preshuffled attention FP8 projections for DSv4 (#46720), TML Inkling enabled (#48841), tuned selective_state_update float16 config for MI325X (#50006), GPT-J-style MRoPE fixed and optimized (#49906), quickreduce accuracy fix in cudagraph mode (#46913), cached fp32 upcast of static e8m0 weight scales (#47773), batch DMA for CPU KV cache loads (#49843), elastic EP scaling accuracy fix (#47206).\r\n* XPU: QK Norm + RoPE fusion pass (#49394), FP8 o_proj with fp8_bmm and load-time scale transpose (#48334), DeepSeek-V4 fuse_index_q SYCL kernel path (#45991), TD operand loads for batched MoE GEMM (#46340), RMSNorm kernels unified with vllm_c (#46981).\r\n* CPU: INT8 fused MoE kernel for Arm CPUs (#48637), s390x inference optimization with oneDNN INT8 GEMM (#50219), GDN conv path optimized for speculative decoding (#48577), granite-4 enabled (#47641), FAST_EXP for Power (#49571), CPU kernels bumped to the latest version (#50387), macOS build fixes (#49021, #50915).\r\n\r\n### Large Scale Serving & Distributed\r\n* Fault tolerance framework (simplified) for DP+EP external LB deployments (#44428).\r\n* Elastic EP: async preparation (#47288) and non-contiguous weight transfer fix (#50641).\r\n* P/D disaggregation: NIXL P/D for hybrid MLA+SSM models (#49762), NIXL heterogeneous P/D block sizes for hybrid models (#49612), MoRIIO heterogeneous TP<->DP prefill/decode read routing (#46116), optional lookup disable on PD decode (#50498), prefill token ids reused on the decode chat path (#48145), detokenization streaming derender (#47301), NixlPush skips an extra handshake step in D->P (#49345).\r\n* Fixes: P/D preemption race (#50297), KV lease deadlines rebased onto the worker clock (#50326), NIXL hybrid MLA+mamba heterogeneous TP (#49297), internal LB load-balancing (#49204).\r\n* Mooncake: vectorized `prepare_value` on the KV load path (#48531), full external hits re-derived on stored boundaries (#49481).\r\n* Encoder-cache connectors: `has_pending_push_work` (#49582).\r\n* Communicators: process-checkpoint lifecycle hooks, starting with FlashInfer (#46877).\r\n\r\n### Quantization\r\n* New capabilities: FP4 Qutlass integration for compressed-tensors (#43229), CuTeDSL MoE for ReLU2 NVFP4 (#49580), MXFP8 linear support in INC (#47514), AutoRound W4A16 MoE and MXFP4 linear/MoE on XPU (#47124), KV quant mode for TurboQuant (#50533), ModelOpt FP8 emulation on SM80 (#50019), `--linear-backend` honored for ModelOpt W4A16 (#50273).\r\n* Checkpoints: compressed-tensors support for DeepSeek-V4 (#41276) and Kimi-K3 (#50500); `find_matched_target` prioritizes fused-name matches (#49483).\r\n* MoE refactor: FusedMoE renamed to FusedMoEFactory (#44941), MoeWNA16 migrated to the MK oracle scheme (#44120), Quark w8a8-int8 (#46765) and MXFP4 `aiter`/`emulation` backends (#49348, #48949) moved to kernel abstractions, CT WNA16 Marlin/MoE methods merged (#44570), Quark W4A8 (INT4-FP8) MoE CI coverage (#48050).\r\n\r\n### API & Frontend\r\n* Rust frontend: gRPC control plane with engine-aware health reporting (#48992), abort control RPC (#49255), server and model discovery (#49491), and KV event source discovery (#50033); `vllm-bench` integrated into `vllm-rs` and the `vllm` CLI (#48930) with opt-in Rust delegation for `vllm bench serve` (#50081); zero-copy multimodal tensor slicing (#48781), multimodal tensors in auxiliary frames (#49341), `--limit-mm-per-prompt` (#49604), ordinary-text tokenizer encoding (#49992).\r\n* APIs: Cohere chat v2 API support (#47189), `cache_salt` in the Anthropic Messages API (#49498), strict tool calling and constrained decoding for GPT-OSS Harmony (#45560), unified engine-based Mistral parser for reasoning and tool calls (#48947), `stream_interval` exposed as a per-request sampling param (#49754), additional sampling parameters for the translation API (#45839), `diarized_json` for MOSS-Transcribe-Diarize (#48543), cumulative speech-to-text chunk timestamps (#41131).\r\n* Multimodal: mm hash algorithm selection via CLI (#49686), configurable PyNvVideoCodec decoder concurrency (#49753), RFC 2397 parameters accepted in base64 data URLs (#48973).\r\n* UX & validation: standardized request error handling with a VLLMError hierarchy (#49665), reduced startup log noise (#50590), incompatible nested runtime overrides rejected (#49247), improved data-parallel launch validation (#49124), DCP topology validation (#49777), 400 instead of 500 for non-numeric logprobs (#49144), bare Inkling text preserved in Python and Rust parsers (#50403).\r\n* Benchmarks: probe requests in `vllm bench serve` (#49611).\r\n\r\n### Dependencies\r\n* PyTorch 2.13.0, torchvision 0.28.0, Triton 3.7.1 (#48155); torch 2.13 for XPU (#48677) and CPU (#50412).\r\n* Transformers 5.14.1 (#49223), FlashInfer 0.6.15 (#48914) then 0.6.16.post3 (#50892), AITER 0.1.16.post5 (#48683) then 0.1.19 (#49361), NCCL 2.30.7 enabling DeepEPv2 in the vllm/vllm-openai image (#45321), tpu-inference v0.25.0 (#49431) then v0.26.0 (#50522), Helion 1.4.0 (#50307), NIXL and UCX upgraded on ROCm (#49251).\r\n* Build: vllm-flash-attn bumped to a C++20-compatible commit for torch-nightly (#49326), ABI-stable FA2 build pin (#50474).\r\n\r\n### Deprecations & Removals\r\n* Models removed: Plamo2 (#49729), Ouro (#49786).\r\n* Removed the no-longer-supported `max_num_partial_prefills` and `max_long_partial_prefills` arguments (#49244).\r\n\r\n## New Contributors\r\n\r\n* @afriedri made their first contribution in https://github.com/vllm-project/vllm/pull/49621\r\n* @amd-sourjya made their first contribution in https://github.com/vllm-project/vllm/pull/48050\r\n* @andreatassi made their first contribution in https://github.com/vllm-project/vllm/pull/48879\r\n* @andrewbcohere made their first contribution in https://github.com/vllm-project/vllm/pull/47189\r\n* @avininjamay8 made their first contribution in https://github.com/vllm-project/vllm/pull/47764\r\n* @bastefaniak made their first contribution in https://github.com/vllm-project/vllm/pull/49190\r\n* @boe20211 made their first contribution in https://github.com/vllm-project/vllm/pull/49431\r\n* @bugkeep made their first contribution in https://github.com/vllm-project/vllm/pull/41357\r\n* @cagrikymk made their first contribution in https://github.com/vllm-project/vllm/pull/46720\r\n* @chun-wan made their first contribution in https://github.com/vllm-project/vllm/pull/44972\r\n* @ColinZ22 made their first contribution in https://github.com/vllm-project/vllm/pull/48044\r\n* @dsocek made their first contribution in https://github.com/vllm-project/vllm/pull/47791\r\n* @euisuh made their first contribution in https://github.com/vllm-project/vllm/pull/49654\r\n* @evantakahashi made their first contribution in https://github.com/vllm-project/vllm/pull/47953\r\n* @FeathBow made their first contribution in https://github.com/vllm-project/vllm/pull/49111\r\n* @fxmarty made their first contribution in https://github.com/vllm-project/vllm/pull/49732\r\n* @hotTea123 made their first contribution in https://github.com/vllm-project/vllm/pull/48218\r\n* @Ibrahim2595 made their first contribution in https://github.com/vllm-project/vllm/pull/45432\r\n* @jcotant-inferact made their first contribution in https://github.com/vllm-project/vllm/pull/49474\r\n* @jiacao-amd made their first contribution in https://github.com/vllm-project/vllm/pull/47773\r\n* @Johnny-Liou made their first contribution in https://github.com/vllm-project/vllm/pull/48018\r\n* @johnnyychiu made their first contribution in https://github.com/vllm-project/vllm/pull/45532\r\n* @jongukc made their first contribution in https://github.com/vllm-project/vllm/pull/49440\r\n* @kevglynn made their first contribution in https://github.com/vllm-project/vllm/pull/41602\r\n* @krishnateja95 made their first contribution in https://github.com/vllm-project/vllm/pull/48876\r\n* @Lafunamor made their first contribution in https://github.com/vllm-project/vllm/pull/40289\r\n* @latent-9 made their first contribution in https://github.com/vllm-project/vllm/pull/50491\r\n* @LG-0927 made their first contribution in https://github.com/vllm-project/vllm/pull/50571\r\n* @liminfei-amd made their first contribution in https://github.com/vllm-project/vllm/pull/48739\r\n* @lkk12014402 made their first contribution in https://github.com/vllm-project/vllm/pull/47124\r\n* @loulanyue made their first contribution in https://github.com/vllm-project/vllm/pull/50704\r\n* @markyangcc made their first contribution in https://github.com/vllm-project/vllm/pull/49021\r\n* @meiyeh123 made their first contribution in https://github.com/vllm-project/vllm/pull/50522\r\n* @Mi-Jiazhi made their first contribution in https://github.com/vllm-project/vllm/pull/49691\r\n* @microslaw made their first contribution in https://github.com/vllm-project/vllm/pull/47298\r\n* @molly-ting made their first contribution in https://github.com/vllm-project/vllm/pull/49660\r\n* @neweyes made their first contribution in https://github.com/vllm-project/vllm/pull/49659\r\n* @nickus made their first contribution in https://github.com/vllm-project/vllm/pull/48366\r\n* @nikhilkulkarni1755 made their first contribution in https://github.com/vllm-project/vllm/pull/44239\r\n* @nvbfalk made their first contribution in https://github.com/vllm-project/vllm/pull/47750\r\n* @omkar-droid made their first contribution in https://github.com/vllm-project/vllm/pull/50688\r\n* @oops-oom made their first contribution in https://github.com/vllm-project/vllm/pull/49985\r\n* @ormandj made their first contribution in https://github.com/vllm-project/vllm/pull/48317\r\n* @PerkzZheng made their first contribution in https://github.com/vllm-project/vllm/pull/50210\r\n* @philippesic made their first contribution in https://github.com/vllm-project/vllm/pull/49114\r\n* @qtris123 made their first contribution in https://github.com/vllm-project/vllm/pull/48796\r\n* @RyanClark2k made their first contribution in https://github.com/vllm-project/vllm/pull/48438\r\n* @samlaf made their first contribution in https://github.com/vllm-project/vllm/pull/50200\r\n* @ShuoleiWang made their first contribution in https://github.com/vllm-project/vllm/pull/49040\r\n* @siddhant-bharti made their first contribution in https://github.com/vllm-project/vllm/pull/50065\r\n* @Spycsh made their first contribution in https://github.com/vllm-project/vllm/pull/47871\r\n* @stacyroberts made their first contribution in https://github.com/vllm-project/vllm/pull/47207\r\n* @stefan-kaestle made their first contribution in https://github.com/vllm-project/vllm/pull/49177\r\n* @thomas-fahrner-parasail made their first contribution in https://github.com/vllm-project/vllm/pull/48973\r\n* @TobyB1702 made their first contribution in https://github.com/vllm-project/vllm/pull/41131\r\n* @vecheruk-amd made their first contribution in https://github.com/vllm-project/vllm/pull/38293\r\n* @wangqia0309 made their first contribution in https://github.com/vllm-project/vllm/pull/48563\r\n* @wkutak made their first contribution in https://github.com/vllm-project/vllm/pull/48952\r\n* @wskr00 made their first contribution in https://github.com/vllm-project/vllm/pull/49073\r\n* @xuanyu-mistral made their first contribution in https://github.com/vllm-project/vllm/pull/44214\r\n* @yamt made their first contribution in https://github.com/vllm-project/vllm/pull/50547\r\n* @yuan-alex made their first contribution in https://github.com/vllm-project/vllm/pull/48917\r\n* @yudigege86 made their first contribution in https://github.com/vllm-project/vllm/pull/50476\r\n* @zaristei made their first contribution in https://github.com/vllm-project/vllm/pull/49647\r\n\r\n## Contributors\r\n\r\n@AndreasKaratzas, @njhill, @hmellor, @mgoin, @BugenZhao, @yewentao256, @khluu, @taneem-ibrahim, @guan404ming, @NickLucche, @stefankoncarevic, @Change72, @zhenwei-intel, @reidliu41, @oonyshch, @Isotr0py, @MatthewBonanni, @andyxning, @WoosukKwon, @tlrmchlsmth, @ZJY0516, @fxmarty-amd, @ivanium, @aoshen02, @xwu-intel, @connorcarpenter15, @lengrongfu, @bnellnm, @mikekg, @kylesayrs, @aarushjain29, @fadara01, @netanel-haber, @Etelis, @LopezCastroRoberto, @jikunshang, @chaunceyjiang, @umut-polat, @divakar-amd, @gau-nernst, @ColinZ22, @jcotant-inferact, @R3hankhan123, @chaojun-zhang, @jongukc, @zxd1997066, @bigPYJ1151, @yzong-rh, @hickeyma, @yushangdi, @majunze2001, @Akashcodes732, @zufangzhu, @sagearc, @FeathBow, @xiaolong-intel, @coltonottley, @okorzh-amd, @sungsooha, @cleonard530, @microslaw, @vllm-agent, @LucasWilkinson, @eicherseiji, @Fangzhou-Ai, @peizhang56, @ZeldaHuang, @noooop, @atalman, @wzhao18, @matteso1, @GirasoleY, @Dao007forever, @chaeminlim-mb, @music-dino, @amd-sourjya, @BadrBasowid, @Johnny-Liou, @Rohan138, @brandonpelfrey, @harjothkhara, @tianmu-li, @fxmarty, @oops-oom, @TheEpicDolphin, @yma11, @itayalroy, @shen-shanshan, @JaredforReal, @wskr00, @janeyx99, @mayuyuace, @omkar-droid, @mganczarenko, @Spycsh, @hclsys, @EdalatiAli, @wangqia0309, @tc-mb, @wkutak, @charlifu, @fynnsu, @xuanyu-mistral, @ormandj, @flutist, @simon-mo, @yuan-alex, @esmeetu, @stefan-kaestle, @bastefaniak, @markyangcc, @ilmarkov, @jjmiao1, @stecasta, @evantakahashi, @ariG23498, @sfeng33, @rasmith, @gnovack, @boe20211, @bbrowning, @S1ro1, @davidjpyu, @nikhilkulkarni1755, @vllmellm, @yuyue0225sc, @euisuh, @zhou9402, @li-jinpeng, @hotTea123, @mgazz, @thomas-fahrner-parasail, @elvircrn, @djramic, @adobrzyn, @qtris123, @harshaljanjani, @mosya415, @gcanlin, @fangyuchu, @anthonsu, @Palaiologos1453, @Achyuthan-S, @athrael-soju, @LiuLi1998, @liranschour, @Alex-ai-future, @dsocek, @galletas1712, @nickus, @edwinlim0919, @walterbm, @thegoldenflow, @Zhenzhong1, @haoyangli0109, @tjtanaa, @ronensc, @neweyes, @liminfei-amd, @garrygale, @jesse996, @TobyB1702, @puririshi98, @andreatassi, @avininjamay8, @frida-andersson, @nvbfalk, @varun-sundar-rabindranath, @mindungil, @ayush1399, @afriedri, @ShuoleiWang, @Liangliang-Ma, @xin3he, @liangel-02, @Ibrahim2595, @qianlihuang, @jiacao-amd, @brian-dellabetta, @labAxiaoming, @johnnyychiu, @siddhant-bharti, @jdebache, @mrn3088, @kevglynn, @krishnateja95, @danielafrimi, @zqzten, @philippesic, @afierka-intel, @PerkzZheng, @omerpaz95, @cinnamonica02, @fallintoplace, @yintong-lu, @chanh, @roikoren755, @zaristei, @bugkeep, @vecheruk-amd, @molly-ting, @lkk12014402, @LiuYinfeng01, @jperezdealgaba, @stacyroberts, @juliendenize, @hao-aaron, @cagrikymk, @zou3519, @RyanClark2k, @samlaf, @jpvillam-amd, @mawong-amd, @vanshbhatia-amd, @rjrock, @majian4work, @woosebastian, @Mi-Jiazhi, @jeejeelee, @latent-9, @oguzhankir, @LG-0927, @ylangtsou, @meiyeh123, @skavulya, @Lafunamor, @Amir-19, @andrewbcohere, @yudigege86, @aeon-x, @wentian-byte, @almogtavor, @hongxiayang, @loulanyue, @jasonlizhengjian, @chun-wan, @wjabbour, @yamt, @amitz-nv, @JianDan0212, @lkm2835, @lk-chen\r\n","reactions":{"url":"https://api.github.com/repos/vllm-project/vllm/releases/368180220/reactions","total_count":40,"+1":15,"-1":0,"laugh":0,"hooray":10,"confused":0,"heart":7,"rocket":8,"eyes":0},"mentions_count":200},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/359735987","assets_url":"https://api.github.com/repos/vllm-project/vllm/releases/359735987/assets","upload_url":"https://uploads.github.com/repos/vllm-project/vllm/releases/359735987/assets{?name,label}","html_url":"https://github.com/vllm-project/vllm/releases/tag/v0.26.0","id":359735987,"author":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"node_id":"RE_kwDOI7xefs4VcSKz","tag_name":"v0.26.0","target_commitish":"main","name":"v0.26.0","draft":false,"immutable":false,"prerelease":false,"created_at":"2026-07-27T00:57:50Z","updated_at":"2026-07-27T06:52:36Z","published_at":"2026-07-27T01:06:58Z","assets":[{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/491125700","id":491125700,"node_id":"RA_kwDOI7xefs4dRfvE","name":"vllm-0.26.0+cpu-cp312-cp312-macosx_11_0_arm64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":20732570,"digest":"sha256:c2419fbb091f17915a1e41a952df35d04ecf5373d31c4499bf4a84777e1e8e5a","download_count":2752,"created_at":"2026-07-27T06:52:34Z","updated_at":"2026-07-27T06:52:36Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.26.0/vllm-0.26.0%2Bcpu-cp312-cp312-macosx_11_0_arm64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/489372025","id":489372025,"node_id":"RA_kwDOI7xefs4dKzl5","name":"vllm-0.26.0+cpu-cp38-abi3-manylinux_2_34_aarch64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":57447846,"digest":"sha256:aa9500f2026e6b343e4361bdbbe3de10bb357eef32b0815ae9f0c31947d8427d","download_count":7280,"created_at":"2026-07-25T10:38:57Z","updated_at":"2026-07-25T10:39:01Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.26.0/vllm-0.26.0%2Bcpu-cp38-abi3-manylinux_2_34_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/489372027","id":489372027,"node_id":"RA_kwDOI7xefs4dKzl7","name":"vllm-0.26.0+cpu-cp38-abi3-manylinux_2_34_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":117126821,"digest":"sha256:07de050c09b992c3215615ec11c4bb09eb384760d4e5bbb835496a846b64873d","download_count":9589,"created_at":"2026-07-25T10:38:57Z","updated_at":"2026-07-25T10:39:04Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.26.0/vllm-0.26.0%2Bcpu-cp38-abi3-manylinux_2_34_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/489372028","id":489372028,"node_id":"RA_kwDOI7xefs4dKzl8","name":"vllm-0.26.0+cu129-cp38-abi3-manylinux_2_28_aarch64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":495047593,"digest":"sha256:a21f3a78adffce8f3e88b08ad1145ff5d7d6e0618c652db3e0b914c700cd9b64","download_count":38854,"created_at":"2026-07-25T10:38:57Z","updated_at":"2026-07-25T10:39:29Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.26.0/vllm-0.26.0%2Bcu129-cp38-abi3-manylinux_2_28_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/489372026","id":489372026,"node_id":"RA_kwDOI7xefs4dKzl6","name":"vllm-0.26.0+cu129-cp38-abi3-manylinux_2_28_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":518935080,"digest":"sha256:6ce4ca30616f0a35810391015622b197a7b8b267ed27f8716f0789db79ff578b","download_count":60787,"created_at":"2026-07-25T10:38:57Z","updated_at":"2026-07-25T10:39:26Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.26.0/vllm-0.26.0%2Bcu129-cp38-abi3-manylinux_2_28_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/489372024","id":489372024,"node_id":"RA_kwDOI7xefs4dKzl4","name":"vllm-0.26.0-cp38-abi3-manylinux_2_28_aarch64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":298269785,"digest":"sha256:52a4c3e55c2c80cc8793e52ccc244457ceade25b0ad7caa1c15e5002a95a1b2c","download_count":4948,"created_at":"2026-07-25T10:38:57Z","updated_at":"2026-07-25T10:39:15Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.26.0/vllm-0.26.0-cp38-abi3-manylinux_2_28_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/489372072","id":489372072,"node_id":"RA_kwDOI7xefs4dKzmo","name":"vllm-0.26.0-cp38-abi3-manylinux_2_28_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":303698761,"digest":"sha256:adb1e4c9b46d0dfdb094121ae5aad670a42412dd813ed4e5db069ed6a15006de","download_count":5849,"created_at":"2026-07-25T10:39:02Z","updated_at":"2026-07-25T10:39:16Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.26.0/vllm-0.26.0-cp38-abi3-manylinux_2_28_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/489372088","id":489372088,"node_id":"RA_kwDOI7xefs4dKzm4","name":"vllm-0.26.0.tar.gz","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/x-gtar","state":"uploaded","size":38353572,"digest":"sha256:23e9fa19d7e20ce7dcc1c074d41503e2116d23f19e688f5d5ea91b741f958502","download_count":646,"created_at":"2026-07-25T10:39:05Z","updated_at":"2026-07-25T10:39:08Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.26.0/vllm-0.26.0.tar.gz"}],"tarball_url":"https://api.github.com/repos/vllm-project/vllm/tarball/v0.26.0","zipball_url":"https://api.github.com/repos/vllm-project/vllm/zipball/v0.26.0","body":"# vLLM v0.26.0 Release Notes\r\n\r\n## Highlights\r\n\r\nThis release features 411 commits from 212 contributors (61 new)!\r\n\r\n* **New Inkling model family** with a full support stack: base modeling (#48799), piecewise CUDA graph support (#48822), Hopper FA4 relative attention (#48858), MTP=1 speculative decoding (#48869), LoRA (#48884), and standard ModelOpt NVFP4 quantization (#48990).\r\n* **DeepSeek-V4 performance push** across vendors: a specialized routing kernel (2.94% E2E TPOT, #48660), `fused_topk_bias` (1.5–2x kernel, #47463), and redundant repeat/copy removal (1.8% E2E TPOT, #48137), plus ROCm two-stage compressor for HCA prefill (#47718), sparse decode/prefill optimizations (#48519, #48788, #46275), and DSpark speculative decoding on AMD (#47419) and XPU (#47677).\r\n* **fp32 `lm_head` for generation models via `head_dtype`** (#48390), extended to the LoRA path (#48525) and given a ROCm `torch.mm` fast path (#48688), improving accuracy for generation heads.\r\n* **Flexible attention backends**: the attention backend can now be selected per KV-cache group (#48012), and sliding-window support is now an explicit backend capability (#48011) — improving support for hybrid models.\r\n* **KV offloading & tiered secondary storage** matured substantially: offloading metrics (#45958, #47666, #47679), tier-owned event handling (#46544, #47923), object-store secondary tier with workload identity (#47063, #47274, #48150), DP-replica-aware tiering (#47987), and encoder-cache (EC) connectors including CPU offloading (#42433, #47423).\r\n* **Rust frontend** gained multimodal video (#47959) and audio (#48554), a Seed-OSS tool parser (#47741), and a native `vllm-bench` port (#48107).\r\n* **Transformers 5.13.0** (#47867) with more models migrated to the Transformers modeling backend: Olmo/Olmo2 (#48100), MistralLarge3 (#48153), and HunyuanVL (#47872).\r\n\r\n### Model Support\r\n* New models: Inkling family (#48799, #48822, #48858, #48869, #48884, #48990), BertForMaskedLM (#48463), RobertaForTokenClassification / XLMRobertaForTokenClassification (#47991), LongCat-Flash-Lite n-gram embedding (#47857), Cosmos3 Edge Reasoner (#48291) and Cosmos3-Super registration (#48211), TranslateGemma-12b-it (#41599).\r\n* Transformers backend migrations: Olmo/Olmo2 (#48100), MistralLarge3 to AutoWeightsLoader (#48153), HunyuanVL native transformers processor for transformers 5.13 (#47872).\r\n* GLM5.2: migrate MoE sequence-parallel support to the non-torch-compiled path (#47881).\r\n* LoRA: FlashInfer MoE LoRA for BF16 models (#48632), LoRA for tower/connector in LlavaNextVideo (#48594), fp32 `lm_head` on the LoRA path (#48525), optimized `TrtLlmLoRAExperts` (#48759).\r\n* Multimodal: automatic fallback to ViT data parallelism when TP is unavailable (#49046).\r\n* Fixes: correct pooling scores for chunked prefill under `torch.compile` (#48901).\r\n\r\n### Engine Core\r\n* fp32 `lm_head` for generation models via `head_dtype` (#48390); lower memory for capturing large CUDA graph sizes (#48483); opt-in persistence and reuse of the memory-profiling result across boots (#47388); improved InstantTensor loading (#46868).\r\n* Attention: select a different attention backend per KV-cache group (#48012); sliding-window as an explicit backend capability (#48011); KV-cache layout refactor packing K/V into the content dim across backends (#44455); MRV2 virtual-batch PCP for MLA (#46570).\r\n* Speculative decoding: runtime draft weight update (#46725), hybrid (SWA + full attention) DFlash drafters (#47914), SWA support for qwen-eagle3 (#47568), Gemma4-12B DSpark draft model (#47216), DSv4 DSpark on AMD (#47419), separate `kv_cache_dtype` for `speculative_config` (#48787).\r\n* KV offloading: basic offloading metrics (#45958), split CPU cache usage into read/write gauges (#47666) and tiering-lookup-delay into sync/async histograms (#47679), tier-owned event handling and BlockStored events (#46544, #47923), object-store secondary tier with workload identity (#47063, #47274, #48150), DP-replica-aware tiering (#47987), `blocks_per_chunk` config for heterogeneous KV groups (#48878), P2P default host/port env vars (#47636).\r\n* Caching: partial prefix-cache hit for hybrid models (#46384), selective hybrid cache retention (#47782), report prefix-cache-reused blocks in full report mode (#45261).\r\n* Reasoning: optimize TPOT for thinking budget when used with speculative decoding (#46662).\r\n* RLHF: stateful trainer-send abstractions (#48042).\r\n* Fixes: host memory leak from undrained `new_block_ids` (#44490), DSv3.2 + MTP + sequence-parallel accuracy (#48036).\r\n\r\n### Hardware & Performance\r\n* DeepSeek-V4: specialized routing kernel (2.94% E2E TPOT, #48660), `fused_topk_bias` 1.5–2x (#47463), redundant repeat/copy removal (1.8% TPOT, #48137).\r\n* MoE router GEMMs: BF16x3 router GEMM (#47973), FP32 router GEMV (#48335), generic CuteDSL LL BF16 router GEMM (#42562); TRTLLM BF16 MoE modular kernel (#45182); write FlashInfer combine into final output (#47156).\r\n* Qwen: fuse more RMSNorm + all-reduce in Qwen3.5 (#46998), replace MoE all-reduce with reduce-scatter (#47006), Qwen3.5 H20 optimization (#48350), expand Triton warmup coverage (#47546).\r\n* MLA: dense MHA path for short sparse-MLA sequences (#47327); MiniMax-M3 long-context decode indexer on sm100 (#48582).\r\n* Kernels: CUDA kernel for ReLUSquaredActivation / relu^2 (#39058), Helion kernel lazy registration (#48264), vectorize `_copy_mamba_state_block` to uint64 (#48110), stop upcasting logits to fp32 in the sampler (#48641).\r\n* ROCm: fp32 `head_dtype` `torch.mm` fast path (#48688), DSv4 two-stage compressor kernel (#47718), sparse decode/prefill optimizations (#48519, #48788, #46275), DSv3.2 sparse MLA KV-split heuristic (#46832) and MTP CUDA-graph mode (#45149), MXFP8 GEMM for MiniMax-M3 (#46117), AITER sparse paged attention + spec decode for MiniMax-M3 (#47287, #47984), MiniMax-M2 fused QK-norm + all-reduce via AITER (#44849), HybridW4A16 linear kernel (#40977), Qwen3-30B-A3B QK-Norm+RoPE+KV runtime fusion (#42749).\r\n* XPU: batch-invariant kernels (#41934), HND KV layout support (#47975), DSpark spec decode for DSv4 (#47677), nightly/release image publishing (#47880, #48126).\r\n* CPU: DFlash speculative decoding for GDN models on CPU (#46090), s390x NUMA topology (#40714), native macOS arm64 CPU wheel builds (#48289); POWER VSX math function optimization (#47321) and IBM Power docker builds using prebuilt wheels (#46017).\r\n* Distributed fusion: FlashInfer MNNVL all-reduce RMS quant fusion (#48064).\r\n* Build/autotune: arm64 Blackwell SM10x/SM110 image builds (#48041); skip CuTeDSL fp4_gemm autotuning by default (#48268).\r\n\r\n### Large Scale Serving & Distributed\r\n* Decode Context Parallel (DCP): hybrid attention support (#40996), DCP + Eagle for Tokenspeed MLA backends (#48180).\r\n* PD disaggregation: NIXL pipeline-parallel prefill in push mode (#45880).\r\n* Encoder-cache connectors: EC transfer params (#42433) and CPU-offloading EC connector (#47423).\r\n\r\n### Quantization\r\n* Humming w[2-7]a[4,8] weight-only inference with compressed-tensors (#46390); int4 quantization for the emulation MoE backend (#48451); INT2 XPU weight-only quant linear (#47521).\r\n* NVFP4/MXFP4: `nvfp4_per_token` online MoE quantization (#48538), CuTe-DSL FlashInfer MXFP4 quantization (#48417); bounded peak memory when repacking FP4 MoE weights for Marlin (#47851) and for NVFP4 MoE weight loading (#46276).\r\n* MLA: `kv_cache_dtype_skip_layers` support (#47309).\r\n* ROCm: HybridW4A16 linear kernel (#40977).\r\n\r\n### API & Frontend\r\n* Rust frontend: multimodal video (#47959) and audio (#48554), Seed-OSS tool parser (#47741), native `vllm-bench` port (#48107), `continue_final_message` handling with renderer sentinel (#47844).\r\n* OpenAI compatibility: `bad_words` in `/v1/completions` (#46793), expose `logprob_token_ids` on Python OpenAI endpoints (#43463), `include_reasoning` param for non-Harmony models (#44301), populate `num_cache_creation_tokens` on Messages responses (#48535).\r\n* Endpoint plugins framework (#47454); `/abort_requests` on the RLHF dev API router (#47173); Deepstream video decoding backend (#42424); overlap preprocessing and computation for pooling models in offline inference (#47699).\r\n* UX: human-readable integers for more CLI args (#47608), CuTeDSL compilation progress bar (#48881), expanded GPU profiler config scope/annotations (#37524), log worker exit code when a process dies unexpectedly (#38641).\r\n* Stability/correctness: handle grammar compilation failures without crashing the engine (#47312), fix logprobs token-string collision from SentencePiece spaces (#48674).\r\n\r\n### Security\r\n* Replace diskcache to eliminate pickle deserialization (#44549).\r\n* Fix a concurrent sparse-invariant race that bypassed CVE remediation (#48583).\r\n* Add resource-bounds validation to derender endpoints (#47260); sanitize server file paths from validation error responses (#46415); bound the completion prompt list to prevent unbounded engine fan-out (#47845); guard lm-format-enforcer regex compilation with a timeout (#47595).\r\n\r\n### Dependencies\r\n* Transformers 5.13.0 (#47867), FlashInfer 0.6.14 (#47669), NIXL 1.3.1 (#47559), tpu-inference v0.24.0 (#47835), nvidia-cutlass-dsl 4.6.0 (#47442), vllm_xpu_kernels v0.1.11.1 (#48942).\r\n* FlashAttention 3 pinned to the torch stable-ABI commit (#47995); ABI-stable FlashMLA build (#48174).\r\n\r\n### Deprecations & Removals\r\n* Models removed: TeleChat (#47989), Persimmon and Fuyu (#48096).\r\n\r\n## New Contributors\r\n\r\n* @adhi29 made their first contribution in https://github.com/vllm-project/vllm/pull/48262\r\n* @adsridhar made their first contribution in https://github.com/vllm-project/vllm/pull/48291\r\n* @alexxu-roblox made their first contribution in https://github.com/vllm-project/vllm/pull/48025\r\n* @amd-ethany made their first contribution in https://github.com/vllm-project/vllm/pull/46117\r\n* @AndyDai-nv made their first contribution in https://github.com/vllm-project/vllm/pull/48507\r\n* @aoright made their first contribution in https://github.com/vllm-project/vllm/pull/47797\r\n* @ap9272 made their first contribution in https://github.com/vllm-project/vllm/pull/43896\r\n* @avalliappan-nvidia made their first contribution in https://github.com/vllm-project/vllm/pull/47460\r\n* @brijrajk made their first contribution in https://github.com/vllm-project/vllm/pull/44349\r\n* @Debasish-87 made their first contribution in https://github.com/vllm-project/vllm/pull/48878\r\n* @devalshahamd made their first contribution in https://github.com/vllm-project/vllm/pull/37524\r\n* @DiegoCao made their first contribution in https://github.com/vllm-project/vllm/pull/47216\r\n* @drakosha made their first contribution in https://github.com/vllm-project/vllm/pull/46972\r\n* @edwinlim0919 made their first contribution in https://github.com/vllm-project/vllm/pull/47495\r\n* @ErenAta16 made their first contribution in https://github.com/vllm-project/vllm/pull/48333\r\n* @Functionhx made their first contribution in https://github.com/vllm-project/vllm/pull/48153\r\n* @gangula-karthik made their first contribution in https://github.com/vllm-project/vllm/pull/48594\r\n* @Gavin-Morris-04 made their first contribution in https://github.com/vllm-project/vllm/pull/48293\r\n* @giuseppegrossi made their first contribution in https://github.com/vllm-project/vllm/pull/48159\r\n* @GongLei-HW made their first contribution in https://github.com/vllm-project/vllm/pull/45261\r\n* @guoriyue made their first contribution in https://github.com/vllm-project/vllm/pull/48211\r\n* @hugo-cen made their first contribution in https://github.com/vllm-project/vllm/pull/48330\r\n* @ibondarenko1 made their first contribution in https://github.com/vllm-project/vllm/pull/43117\r\n* @iyastreb made their first contribution in https://github.com/vllm-project/vllm/pull/48209\r\n* @jacklin78911-collab made their first contribution in https://github.com/vllm-project/vllm/pull/47744\r\n* @jhu960213 made their first contribution in https://github.com/vllm-project/vllm/pull/42749\r\n* @joanvelja made their first contribution in https://github.com/vllm-project/vllm/pull/44371\r\n* @KKothuri made their first contribution in https://github.com/vllm-project/vllm/pull/48390\r\n* @kl527 made their first contribution in https://github.com/vllm-project/vllm/pull/47464\r\n* @krishy91 made their first contribution in https://github.com/vllm-project/vllm/pull/47991\r\n* @langzhao-netizen made their first contribution in https://github.com/vllm-project/vllm/pull/43463\r\n* @lishunyang12 made their first contribution in https://github.com/vllm-project/vllm/pull/49015\r\n* @mahadrehmann made their first contribution in https://github.com/vllm-project/vllm/pull/48098\r\n* @ManaEstras made their first contribution in https://github.com/vllm-project/vllm/pull/47872\r\n* @mosya415 made their first contribution in https://github.com/vllm-project/vllm/pull/48846\r\n* @mwoodson made their first contribution in https://github.com/vllm-project/vllm/pull/48523\r\n* @MynameFelix made their first contribution in https://github.com/vllm-project/vllm/pull/41811\r\n* @nicklasfrahm made their first contribution in https://github.com/vllm-project/vllm/pull/45313\r\n* @passtoor-agi made their first contribution in https://github.com/vllm-project/vllm/pull/48849\r\n* @pierDipi made their first contribution in https://github.com/vllm-project/vllm/pull/47063\r\n* @ruikangliu made their first contribution in https://github.com/vllm-project/vllm/pull/47309\r\n* @Sahil170595 made their first contribution in https://github.com/vllm-project/vllm/pull/45207\r\n* @samnordmann made their first contribution in https://github.com/vllm-project/vllm/pull/47156\r\n* @shawntsai made their first contribution in https://github.com/vllm-project/vllm/pull/47801\r\n* @sungbin1015 made their first contribution in https://github.com/vllm-project/vllm/pull/46793\r\n* @tanish-malekar made their first contribution in https://github.com/vllm-project/vllm/pull/39058\r\n* @tomerg-nvidia made their first contribution in https://github.com/vllm-project/vllm/pull/47021\r\n* @tsvikas made their first contribution in https://github.com/vllm-project/vllm/pull/47296\r\n* @vanshbhatia-amd made their first contribution in https://github.com/vllm-project/vllm/pull/47767\r\n* @ViranjanPagar made their first contribution in https://github.com/vllm-project/vllm/pull/42424\r\n* @vivek8123 made their first contribution in https://github.com/vllm-project/vllm/pull/46017\r\n* @vx120 made their first contribution in https://github.com/vllm-project/vllm/pull/46725\r\n* @wangxingda made their first contribution in https://github.com/vllm-project/vllm/pull/48829\r\n* @wendadawen made their first contribution in https://github.com/vllm-project/vllm/pull/46213\r\n* @wenpengw-nv made their first contribution in https://github.com/vllm-project/vllm/pull/44863\r\n* @woosebastian made their first contribution in https://github.com/vllm-project/vllm/pull/48901\r\n* @Yancey0623 made their first contribution in https://github.com/vllm-project/vllm/pull/40996\r\n* @yushangdi made their first contribution in https://github.com/vllm-project/vllm/pull/48868\r\n* @yuvalluria made their first contribution in https://github.com/vllm-project/vllm/pull/46396\r\n* @zihaomu made their first contribution in https://github.com/vllm-project/vllm/pull/47404\r\n* @zqzten made their first contribution in https://github.com/vllm-project/vllm/pull/44303\r\n\r\n## Contributors\r\n\r\n@mgoin, @yewentao256, @NickLucche, @njhill, @LucasWilkinson, @micah-wil, @khluu, @AndreasKaratzas, @WoosukKwon, @aoshen02, @BugenZhao, @vanshbhatia-amd, @jperezdealgaba, @jeejeelee, @hmellor, @tlrmchlsmth, @MatthewBonanni, @Change72, @chaojun-zhang, @gau-nernst, @gnovack, @ZJY0516, @reidliu41, @Yejing-Lai, @taneem-ibrahim, @Srinivasoo7, @Sunt-ing, @AmeenP, @zhenwei-intel, @LopezCastroRoberto, @matteso1, @yzong-rh, @stefankoncarevic, @Isotr0py, @djramic, @charlifu, @peizhang56, @giuseppegrossi, @drakosha, @NickCao, @benchislett, @Rohan138, @hickeyma, @muhammadfawaz1, @rasmith, @zixi-qi, @music-dino, @BWAAEEEK, @xianbaoqian, @wendyliu235, @atalman, @joerowell, @ErenAta16, @AlejandroParedesLT, @gcanlin, @liranschour, @omerpaz95, @Alex-ai-future, @tanpinsiang, @Fangzhou-Ai, @KKothuri, @zxd1997066, @akii96, @nemanjaudovic, @Etelis, @afierka-intel, @DaoyuanLi2816, @HDCharles, @tahsintunan, @xiaohongchen1991, @sagearc, @mikekg, @kliuae, @qli88, @arpera, @yushangdi, @edwinlim0919, @tjtanaa, @ariG23498, @lucifer1004, @netanel-haber, @kl527, @Rukhaiya2004, @ronensc, @guan404ming, @shaunkotek, @liulanze, @pierDipi, @eldarkurtic, @simon-mo, @robinguo23, @CienetStingLin, @RishabhSaini, @Sahil170595, @jasonlizhengjian, @jacklin78911-collab, @walterbm, @amd-ethany, @zqzten, @nicklasfrahm, @hongxiayang, @alexeldeib, @ManaEstras, @sungbin1015, @aoright, @Saddss, @vivek8123, @voipmonitor, @shawntsai, @almayne, @ilmarkov, @cleonard530, @kjiang249, @chaunceyjiang, @bigPYJ1151, @tsvikas, @deng451e, @tvirolai-amd, @zhewenl, @zihaomu, @ap9272, @staugust, @Yancey0623, @GongLei-HW, @albertoperdomo2, @guoriyue, @ViranjanPagar, @Functionhx, @XuZhou26, @MynameFelix, @larryli2-amd, @ashwing, @thisisjimmyfb, @robertgshaw2-redhat, @mayuyuace, @ibondarenko1, @zhejiangxiaomai, @vx120, @hugo-cen, @tanish-malekar, @zzt93, @guybd, @R3hankhan123, @mmangkad, @omera-nv, @yma11, @Gavin-Morris-04, @pavanimajety, @shanjiaz, @wenpengw-nv, @atalhens, @langzhao-netizen, @emricksini-h, @Zhenzhong1, @DanBlanaru, @mgehre-amd, @mwoodson, @wangxiyuan, @adsridhar, @hnt2601, @gangula-karthik, @tomerg-nvidia, @adhi29, @rishitdholakia13, @divakar-amd, @chaeminlim-mb, @joanvelja, @russellb, @janeyx99, @aarushjain29, @wjabbour, @mahadrehmann, @krishy91, @tzielinski-habana, @avalliappan-nvidia, @ruikangliu, @majian4work, @maxyanghu, @brijrajk, @lengrongfu, @Josephasafg, @elvircrn, @xiaguan, @ricky-chaoju, @iyastreb, @ovidiusm, @tuukkjs, @noooop, @samnordmann, @AndyDai-nv, @xiao-llm, @DiegoCao, @yuvalluria, @jhu960213, @woosebastian, @Debasish-87, @esmeetu, @hao-aaron, @zhangj1an, @wendadawen, @juliendenize, @passtoor-agi, @mosya415, @labAxiaoming, @devalshahamd, @wangxingda, @xuebwang-amd, @fuscof-ibm, @alexxu-roblox, @frida-andersson, @lishunyang12, @izhuhaoran\r\n","reactions":{"url":"https://api.github.com/repos/vllm-project/vllm/releases/359735987/reactions","total_count":36,"+1":20,"-1":0,"laugh":0,"hooray":12,"confused":0,"heart":4,"rocket":0,"eyes":0},"mentions_count":200},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/353668444","assets_url":"https://api.github.com/repos/vllm-project/vllm/releases/353668444/assets","upload_url":"https://uploads.github.com/repos/vllm-project/vllm/releases/353668444/assets{?name,label}","html_url":"https://github.com/vllm-project/vllm/releases/tag/v0.25.1","id":353668444,"author":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"node_id":"RE_kwDOI7xefs4VFI1c","tag_name":"v0.25.1","target_commitish":"main","name":"v0.25.1","draft":false,"immutable":false,"prerelease":false,"created_at":"2026-07-12T23:40:12Z","updated_at":"2026-07-14T08:58:07Z","published_at":"2026-07-14T08:51:20Z","assets":[{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/476490889","id":476490889,"node_id":"RA_kwDOI7xefs4cZqyJ","name":"vllm-0.25.1+cpu-cp38-abi3-manylinux_2_34_aarch64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":55427774,"digest":"sha256:17555e9624fded9436edcc61ba911976a89a0688937132e1dd66354a74d09726","download_count":3350,"created_at":"2026-07-14T08:57:32Z","updated_at":"2026-07-14T08:57:36Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.25.1/vllm-0.25.1%2Bcpu-cp38-abi3-manylinux_2_34_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/476490892","id":476490892,"node_id":"RA_kwDOI7xefs4cZqyM","name":"vllm-0.25.1+cpu-cp38-abi3-manylinux_2_34_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":114797087,"digest":"sha256:3c2131248fe7bb34f46b298f7e0a44fe3a5893b6b4243bc3ba1a60a6f90ca3de","download_count":6478,"created_at":"2026-07-14T08:57:32Z","updated_at":"2026-07-14T08:57:40Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.25.1/vllm-0.25.1%2Bcpu-cp38-abi3-manylinux_2_34_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/476490887","id":476490887,"node_id":"RA_kwDOI7xefs4cZqyH","name":"vllm-0.25.1+cu129-cp38-abi3-manylinux_2_28_aarch64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":409647950,"digest":"sha256:bdffbe35b2c1ab8f2a9dcc337b657261d9b192c92c217e5a2f98a8835fe78daa","download_count":492,"created_at":"2026-07-14T08:57:32Z","updated_at":"2026-07-14T08:58:07Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.25.1/vllm-0.25.1%2Bcu129-cp38-abi3-manylinux_2_28_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/476490891","id":476490891,"node_id":"RA_kwDOI7xefs4cZqyL","name":"vllm-0.25.1+cu129-cp38-abi3-manylinux_2_28_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":432371078,"digest":"sha256:9e206f370c934a2d4b6b1f05d3d09708d344e05d80260189ef19f60755709431","download_count":29585,"created_at":"2026-07-14T08:57:32Z","updated_at":"2026-07-14T08:57:55Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.25.1/vllm-0.25.1%2Bcu129-cp38-abi3-manylinux_2_28_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/476490890","id":476490890,"node_id":"RA_kwDOI7xefs4cZqyK","name":"vllm-0.25.1-cp38-abi3-manylinux_2_28_aarch64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":244036150,"digest":"sha256:902be760af4c5ebfad8af5b8ea07a53ae14e5a6c839c8ab56da30581abb75ad2","download_count":123641,"created_at":"2026-07-14T08:57:32Z","updated_at":"2026-07-14T08:57:48Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.25.1/vllm-0.25.1-cp38-abi3-manylinux_2_28_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/476490973","id":476490973,"node_id":"RA_kwDOI7xefs4cZqzd","name":"vllm-0.25.1-cp38-abi3-manylinux_2_28_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":250100306,"digest":"sha256:16fc7a28df1576eb6f7ca0455026551b8f9adb674c19c66059359ef3e964bd1e","download_count":124494,"created_at":"2026-07-14T08:57:37Z","updated_at":"2026-07-14T08:57:51Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.25.1/vllm-0.25.1-cp38-abi3-manylinux_2_28_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/476491033","id":476491033,"node_id":"RA_kwDOI7xefs4cZq0Z","name":"vllm-0.25.1.tar.gz","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/x-gtar","state":"uploaded","size":37731826,"digest":"sha256:ddbdec3f1c0f21afa70b7eb6ddf3faa29d26b1302ee4b5d5e00ec3af41b0c2e4","download_count":2425,"created_at":"2026-07-14T08:57:41Z","updated_at":"2026-07-14T08:57:44Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.25.1/vllm-0.25.1.tar.gz"}],"tarball_url":"https://api.github.com/repos/vllm-project/vllm/tarball/v0.25.1","zipball_url":"https://api.github.com/repos/vllm-project/vllm/zipball/v0.25.1","body":"# vLLM v0.25.1\r\n\r\n## Highlights\r\n\r\nThis release features 2 commits from 2 contributors (1 new)!\r\n\r\nv0.25.1 is a patch release containing two targeted bug fixes on top of v0.25.0.\r\n\r\n### Bug Fixes\r\n* **Avoid blocking model launching when no system FFmpeg is available for TorchCodec** (#47888). Previously `import torchcodec` raised a `RuntimeError` at import time when system FFmpeg was missing, which blocked startup (e.g. `vllm serve Qwen/Qwen3-VL-2B-Instruct`) even when TorchCodec was not in use. The error is now deferred to runtime so it only surfaces if TorchCodec is actually needed.\r\n* **Guard mixed-dtype allreduce RMSNorm quant fusions** (#48330). The fused FlashInfer allreduce + RMSNorm + static-quantization patterns could match graphs where the activation and RMSNorm weight dtypes differ (e.g. a BF16 residual stream with an FP32 Gemma/Qwen-style RMSNorm weight in NVFP4 models), corrupting the hidden state and producing garbage output such as repeated `!!!!!` tokens. A dtype-match guard now routes incompatible mixed-dtype graphs to the safe path, while same-dtype models retain the full allreduce + RMSNorm + quant fusion.\r\n\r\n## Contributors\r\n\r\n@Isotr0py, @hugo-cen\r\n\r\n## New Contributors\r\n\r\n* @hugo-cen made their first contribution in https://github.com/vllm-project/vllm/pull/48330","reactions":{"url":"https://api.github.com/repos/vllm-project/vllm/releases/353668444/reactions","total_count":21,"+1":11,"-1":0,"laugh":0,"hooray":10,"confused":0,"heart":0,"rocket":0,"eyes":0},"mentions_count":2},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/351977001","assets_url":"https://api.github.com/repos/vllm-project/vllm/releases/351977001/assets","upload_url":"https://uploads.github.com/repos/vllm-project/vllm/releases/351977001/assets{?name,label}","html_url":"https://github.com/vllm-project/vllm/releases/tag/v0.25.0","id":351977001,"author":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"node_id":"RE_kwDOI7xefs4U-r4p","tag_name":"v0.25.0","target_commitish":"main","name":"v0.25.0","draft":false,"immutable":false,"prerelease":false,"created_at":"2026-07-11T10:25:04Z","updated_at":"2026-07-11T20:06:44Z","published_at":"2026-07-11T20:06:44Z","assets":[{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/473745550","id":473745550,"node_id":"RA_kwDOI7xefs4cPMiO","name":"vllm-0.25.0+cpu-cp38-abi3-manylinux_2_34_aarch64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":55427468,"digest":"sha256:bf3ccb8ad9acc46f259c410a263e31e8ed1820a9086b124ce099953181f2b88e","download_count":102,"created_at":"2026-07-11T20:00:24Z","updated_at":"2026-07-11T20:00:29Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.25.0/vllm-0.25.0%2Bcpu-cp38-abi3-manylinux_2_34_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/473745548","id":473745548,"node_id":"RA_kwDOI7xefs4cPMiM","name":"vllm-0.25.0+cpu-cp38-abi3-manylinux_2_34_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":114796579,"digest":"sha256:3af0d8a493539d7a27efd6c8291593672b48f13e163545f97cf93794e12b13cd","download_count":1121,"created_at":"2026-07-11T20:00:24Z","updated_at":"2026-07-11T20:00:32Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.25.0/vllm-0.25.0%2Bcpu-cp38-abi3-manylinux_2_34_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/473745551","id":473745551,"node_id":"RA_kwDOI7xefs4cPMiP","name":"vllm-0.25.0+cu129-cp38-abi3-manylinux_2_28_aarch64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":409647640,"digest":"sha256:31201233810c42d7954764123731be38693d615e57b6db460a98b70d33a72345","download_count":168,"created_at":"2026-07-11T20:00:24Z","updated_at":"2026-07-11T20:00:48Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.25.0/vllm-0.25.0%2Bcu129-cp38-abi3-manylinux_2_28_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/473745549","id":473745549,"node_id":"RA_kwDOI7xefs4cPMiN","name":"vllm-0.25.0+cu129-cp38-abi3-manylinux_2_28_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":432370721,"digest":"sha256:16670fbbad1483ae8d06794c0dd3c4e6b583f4bd1a61612a73b91af5848fdeb1","download_count":3834,"created_at":"2026-07-11T20:00:24Z","updated_at":"2026-07-11T20:00:51Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.25.0/vllm-0.25.0%2Bcu129-cp38-abi3-manylinux_2_28_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/473745545","id":473745545,"node_id":"RA_kwDOI7xefs4cPMiJ","name":"vllm-0.25.0-cp38-abi3-manylinux_2_28_aarch64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":244035821,"digest":"sha256:a36add91a63e4f8543ffd85766fcc604d1f5e48ddae81b9f159bfa8ef25cdc56","download_count":448,"created_at":"2026-07-11T20:00:24Z","updated_at":"2026-07-11T20:00:43Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.25.0/vllm-0.25.0-cp38-abi3-manylinux_2_28_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/473745611","id":473745611,"node_id":"RA_kwDOI7xefs4cPMjL","name":"vllm-0.25.0-cp38-abi3-manylinux_2_28_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":250099898,"digest":"sha256:c82e96b01af2142ef398ca92f1e9c5f4ffc119f2cc09fe68ea28865323532f0c","download_count":661,"created_at":"2026-07-11T20:00:30Z","updated_at":"2026-07-11T20:00:45Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.25.0/vllm-0.25.0-cp38-abi3-manylinux_2_28_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/473745642","id":473745642,"node_id":"RA_kwDOI7xefs4cPMjq","name":"vllm-0.25.0.tar.gz","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/x-gtar","state":"uploaded","size":37732784,"digest":"sha256:7e04e2b37164de8c4012f27f75af6c4768039b32610865e9b6fb8c49c34a84aa","download_count":102,"created_at":"2026-07-11T20:00:33Z","updated_at":"2026-07-11T20:00:36Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.25.0/vllm-0.25.0.tar.gz"}],"tarball_url":"https://api.github.com/repos/vllm-project/vllm/tarball/v0.25.0","zipball_url":"https://api.github.com/repos/vllm-project/vllm/zipball/v0.25.0","body":"# vLLM v0.25.0 Release Notes\r\n\r\n## Highlights\r\n\r\nThis release features 558 commits from 232 contributors (64 new)!\r\n\r\n* **Model Runner V2 is now the default for all dense models** (#44443). Building on quantized-model support from the previous release, MRv2 is now the standard execution path, with new support for EVS (#46535), realtime embeddings (#46762), prefix caching for Mamba hybrid models (#42406), multimodal-prefix bidirectional attention (#46942), and dynamic speculative decoding compatible with full CUDA graphs (#45953).\r\n* **PagedAttention has been removed** (#47361). The legacy attention implementation is deleted now that V1/MRv2 backends are the standard path.\r\n* **The Transformers modeling backend is now as fast as native vLLM** (#47187), and gained FP8 MoE support (#46820), CUDA graph + embed scaling fixes (#48010), and migration of GPTBigCode/Starcoder2 (#30966) and RoBERTa (#47452).\r\n* **New models**: LLaVA-OneVision-2 (#44785), Unlimited OCR (#46564, #47102), MOSS-Transcribe-Diarize (#47729), openai/privacy-filter (#41026), and Hy3 (#47192). GLM-5 / DeepSeek-V3.2 landed in the model zoo (#46808) with GLM-5.2 tuning, and MiniMax-M3 gained pipeline parallelism (#45810) and NVFP4 support (#46756).\r\n* **New Streaming Parser Engine** (#46610) — a unified tool-call/reasoning parsing framework, with a new Kimi k2.5/k2.6/k2.7 parser and ports of seed_oss (#46314) and DeepSeek V4 (#45877). The Rust frontend continues to mature with HTTPS/mTLS (#45890), a DP supervisor (#47076), and profiler control routes (#46306).\r\n* **Universal speculative decoding for heterogeneous vocabularies (TLI)** (#38174), plus new DSpark (#46995) and DFlash (#46770, #46853) drafters.\r\n\r\n### Model Support\r\n* New models: LLaVA-OneVision-2 (#44785), Unlimited OCR (#46564) with a Triton R-SWA backend (#47102), MOSS-Transcribe-Diarize (#47729), openai/privacy-filter (#41026), Hy3 with token-suffix and JSON Schema array support (#47192).\r\n* GLM-5 family: GLM-5 / DeepSeek-V3.2 added to the model zoo (#46808), GLM-5.2 FP32 gate (#47410), GLM MTP post-final-norm fix (#47448), GLM4V startup fix (#47155).\r\n* MiniMax-M3: pipeline parallelism (#45810), streaming reasoning parsing (#45718), and `tok_sparse_select` from MSA replacing Triton kernels (#47502).\r\n* Transformers backend: now as fast as native vLLM (#47187), FP8 MoE fix (#46820), embed scaling + CUDA graph fix (#48010), GPTBigCode/Starcoder2 (#30966) and RoBERTa (#47452) migration, M-RoPE `mm_token_type_ids` fix (#46552), tied-embedding `lm_head.bias` fix (#46835).\r\n* Voxtral: migrated to mistral-common 1.11.5 audio API (#46705) and realtime token-feedback hang fix (#44461).\r\n* Gemma family: Gemma4 sliding-window/FA4 attention fixes (#47217, #47332), Gemma4 MTP quant_config fix (#47091); DiffusionGemma tensor parallelism (#45719) and HF stability-window semantics (#45965).\r\n* Other fixes: MiniCPM-V 4.6 language-backbone LoRA (#46740) and placeholder grid fix (#45918), pooled Whisper sliding-window sizing (#47071, #47437), Mamba/Mamba2 checkpoint-without-`architectures` crash fix (#46037), DeepSeek-V2 hidden-size and aux-hidden-state fixes (#46986, #46973).\r\n\r\n### Engine Core\r\n* Model Runner V2: default for all dense models (#44443); EVS (#46535), realtime embeddings (#46762), Mamba hybrid prefix caching (#42406), multimodal-prefix bidirectional attention (#46942), cross-attention warmup/block-table fixes (#46753, #47308), Mamba2 crash fix (#47428), scheduling slot accounting (#46974), model-ref cleanup on shutdown (#47483), bounded memory for large-logprobs requests (#46746).\r\n* Speculative decoding: universal spec decode for heterogeneous vocabularies (TLI) (#38174); DSpark drafter + speculators checkpoint support (#46995, #47093); DFlash backend selection (#46770), per-layer RMSNorm fusion (#46761), CPU support (#44029), SWA+DFlash for MiMo (#46104), Laguna XS.2.1 drafter (#46853); MTP for Bailing hybrid models (#44880); block verification for rejection sampling (#46781); reduced TP communication for draft tokens (#46448).\r\n* Sleep mode: pluggable sleep-mode backend abstraction (RFC #34303, #44074) with communicator-agnostic capability flags (#47243).\r\n* Attention: FlashAttention block-size restriction removed for hybrid models (#36701), `FLASH_ATTN_MLA_SPARSE` Hopper sparse-MLA backend (#46189), DCP + FP8 KV cache in MLA decode (#44044), XQA decode kernels (#43232).\r\n* KV offloading: tiering metric plumbing (#45959), request lifecycle fix (#46284), batched lookup in C (#46713), `LookupResult` enum (#46363).\r\n* Misc: `VLLM_GPU_SYNC_CHECK` env var (#44800), VRAM semaphore infrastructure (#44465), skip detokenization in online beam search (#46422), several int32-overflow fixes in sampler/attention kernels (#46560, #47383, #47671).\r\n\r\n### Hardware & Performance\r\n* GLM-5.2 / DeepSeek: `fused_indexer_q_rope_quant` Triton kernel (1.9–3.3% E2E throughput) (#46862), reduce-scatter MoE all-reduce (3.1–3.2% E2E) (#46635), op fusion for GLM5/DSV3.2 (#46876), `token_to_req_indices` cache for DSv4 (5–6x kernel speedup) (#47474), better DSv4 MXFP8 kernel (#47229), redundant-op removal (#47198, #46651).\r\n* NVIDIA/Blackwell: FlashInfer fused all-reduce tuned for world_size=16 on GB300 (#46392), restored NVFP4 swizzled-scale zero-init to recover Blackwell decode throughput (#45739), CuTeDSL/FA4-MLA warmup infrastructure (#46182), skip cooperative top-K on SM120 (#47164), B12x backend for non-gated MoEs (#43328).\r\n* Kernels: Helion `fused_qk_norm_rope` (#44010) and `silu_and_mul_per_block_quant` (#43994), Triton MLA logits workspace (#46819), swap-AB optimization for fused MoE (#36559), vectorized fp32 `moe_sum` supporting any top-k (#46643), blocking CUDA events to avoid busy-polling the driver lock (#47081).\r\n* AMD/ROCm: moved to torch 2.11 stable ABI (#47128); AITER FlashAttention MLA prefill backend `ROCM_AITER_FA` (#45033); fused shared-expert for GLM-4.5/6/7 (#44313) and MiniMax-M3 (#46474, #46545); AITER MoE optimization for DeepSeek-V4 (#46122); AITER custom all-reduce in CudaCommunicator (#46065); INT3 quantization for quickreduce (#45666).\r\n* Intel XPU: W8A8 FP8 linear kernel with multi-granularity quant (#43645), pipeline-parallel accuracy fix (#47253), uniform-batch CUDA graph for FA2 (#46555), route mm_prefix models to Triton attention (#47688), C++ `get_memory_info` (#47134).\r\n* CPU: accelerated unquantized MoE for AArch64 (#46353), macOS/Apple Silicon hang fix via OpenMP (#46769) and broken-install fix (#47457), compressed-tensor w8a8 int8 MoE (#42920), Mamba ShortConv (#35059), chunked prefill + prefix caching for Qwen3.5 (#46202), faster gelu via tanh AOR (#44639).\r\n* RISC-V: RVV path for W4A8 INT4 GEMM (#45269), BF16 on VLEN=256 hardware (#45243), reduced LMUL pressure in INT4 LUT dequant (#47538). POWER: fp16 support on PowerPC (#46135).\r\n* Platform: accelerator-agnostic `get_memory_info` (#44825).\r\n\r\n### Large Scale Serving & Distributed\r\n* Sequence parallelism without requiring DP, 1.9–5.0% E2E throughput improvement (#47070).\r\n* Distributed: NCCL symmetric memory extended to AllGather and ReduceScatter (#46703), FlashInfer all-reduce defaults to MNNVL on single node (#47219, #47589), fault-tolerance backend to detect all2all peer faults and prevent corrupted output (#43637).\r\n* Data parallel: throttle prefills based on local prefill work (#46532), rotate load-balancer tie-break to avoid engine bias (#47420), DP supervisor via the Rust frontend (#47076), DP MTP hang fix (#40589).\r\n* PD disaggregation: secondary-tier implementation (#42285), Mooncake connector GDN (Qwen3.5) + MLA (DeepSeek-V4-Flash) support (#46807), NIXL Mamba1 support (#45019), MultiConnector `kv_transfer_params` merging (#46777), usage field exposed for disaggregated serving (#42748).\r\n* DCP: FlashInfer MLA support (#43729), FLASHINFER_MLA_SPARSE support (#46076), LSE log-base fixes (#47079); Mooncake parallelized KV load (#45971) and DCP>1 lookup fix (#46855).\r\n* ROCm: stabilized high-throughput DBO for DP+EP (#46990), EPLB for Quark OCP MXFP4 MoE (#47220).\r\n\r\n### Quantization\r\n* 2/3/5/6/7-bit pack-quantized weight-only inference (Humming) (#46389), Triton INT4 per-token-head KV cache quantization (#40835).\r\n* NVFP4: fused weight dequantization with compute in the MoE MLP Triton kernel (#44667), NVFP4 KV cache with skip-layers sliding window (#42890), MiniMax-M3 ModelOpt NVFP4 support (#46756).\r\n* FP8: weights padding for per-block online quantization (#44763); deprecated the old FP8 online MoE quantization class (#44514).\r\n* Marlin: thread-tile padding extended to MoE (WNA16 + FP8/MXFP8) (#45703), int8 grouped WNA16 MoE (#47154); FlashInfer MXINT4 MoE for gated SiLU (#46518).\r\n* Fixes: W8A8 int-quant scheme-selection regression (#46860), tied quantized embeddings for ModelOpt Gemma4 (#45544), NVFP4+MTP crash on Qwen3Next (#46316), ModelOpt mixed-precision for sparse configs (#47318), CPU w4a8_int8 MoE path (#46739), actionable error on group-size/TP mismatch (#46230).\r\n\r\n### API & Frontend\r\n* Streaming Parser Engine (#46610): unified tool-call/reasoning parsing with a new Kimi k2.5/k2.6/k2.7 parser; ported seed_oss (#46314) and DeepSeek V4 (#45877).\r\n* OpenAI compatibility: Responses API namespace tools (#47024), per-request timing `metrics` field on Chat/Completions responses (#46768), token offsets on render endpoints (#44226), `return_loss_mask` for training-data generation (#46846), HTTP 422 for unprocessable image URLs (#47165).\r\n* gpt-oss / Harmony: dedicated Harmony renderer (#46800), `process_eos()` flush (#46437), raw-output recovery on non-terminal parse (#47062, #47379).\r\n* Rust frontend: static HTTPS and mTLS for HTTP and gRPC (#45890), DP supervisor (#47076), profiler control routes (#46306), `repetition_detection` sampling param (#46684), unified/combined parser interface (#46583), reduced multimodal tensor copies (#47581), plus many parser and validation fixes.\r\n* Video: TorchCodec added as a video decoding backend (#46609).\r\n* CLI/UX: TTFT and TPS printing in `vllm chat` (#46775), `model_class_overrides` for development/debugging (#47148).\r\n* Tooling/validation: many tool-parser fixes (Kimi K2 IDs #46344, PoolsideV1 #46486/#47311, non-ASCII arguments #46308, `thinking_token_budget` re-entry #43757); rejection of invalid config values (#44070, #44002, #46612) and degenerate `structured_outputs` that crash EngineCore (#45346).\r\n\r\n### Security\r\n* Prevent image decompression-bomb OOM denial of service (#47010).\r\n* Prevent an infinite loop in `split_audio` with NaN audio samples (#46463).\r\n* Bound tokenizer work when an explicit `truncation_side` is set (#47007).\r\n* Block request-level GPU video backend selection (#47259).\r\n* Document the gRPC interface as insecure, for private use only (#45903).\r\n\r\n### Dependencies\r\n* FlashInfer 0.6.13 (#46683), tpu-inference v0.23.0 (#46568), aiter 0.1.16.post2 (#46692), vllm_xpu_kernels v0.1.10.1 (#46607), huggingface-hub v1.22.0 (#47551).\r\n* DeepGEMM updated to enable SM120 support (#47304), FlashAttention 3 built against the torch stable API (#46644), Rust frontend TLS switched from rustls to native-tls/OpenSSL (#46696).\r\n\r\n### Deprecations & Removals\r\n* **PagedAttention deleted** (#47361).\r\n* Models removed: Baichuan (#46362), Aquila (#46605), Grok (#46706), Tarsier / Tarsier2 (#47143), AyaVision / MusicFlamingo (#47263), Mantis (#46806).\r\n* Deprecated the old FP8 online MoE quantization class (#44514); legacy `api_server.py` moved to the examples directory (#46783); `gptq_marlin` removed from supported ROCm quant schemes (#46655).\r\n\r\n## New Contributors\r\n\r\n* @aaarkai made their first contribution in https://github.com/vllm-project/vllm/pull/44610\r\n* @Acaciasama made their first contribution in https://github.com/vllm-project/vllm/pull/45850\r\n* @ACEEE-1222 made their first contribution in https://github.com/vllm-project/vllm/pull/47716\r\n* @adamkbaranowski made their first contribution in https://github.com/vllm-project/vllm/pull/46853\r\n* @AgenticSpark made their first contribution in https://github.com/vllm-project/vllm/pull/46071\r\n* @AIvashov made their first contribution in https://github.com/vllm-project/vllm/pull/42748\r\n* @akinsella made their first contribution in https://github.com/vllm-project/vllm/pull/47165\r\n* @aldenlobo made their first contribution in https://github.com/vllm-project/vllm/pull/45961\r\n* @alex101-ops made their first contribution in https://github.com/vllm-project/vllm/pull/44880\r\n* @aman0603 made their first contribution in https://github.com/vllm-project/vllm/pull/46945\r\n* @Aneureka made their first contribution in https://github.com/vllm-project/vllm/pull/46838\r\n* @ArsalanShakil made their first contribution in https://github.com/vllm-project/vllm/pull/46236\r\n* @ayush1399 made their first contribution in https://github.com/vllm-project/vllm/pull/47091\r\n* @blasrodri made their first contribution in https://github.com/vllm-project/vllm/pull/46827\r\n* @calvarado2004 made their first contribution in https://github.com/vllm-project/vllm/pull/46177\r\n* @chengzheng345 made their first contribution in https://github.com/vllm-project/vllm/pull/44785\r\n* @cpersson-amd made their first contribution in https://github.com/vllm-project/vllm/pull/47519\r\n* @cyq1017 made their first contribution in https://github.com/vllm-project/vllm/pull/46101\r\n* @davispuh made their first contribution in https://github.com/vllm-project/vllm/pull/35232\r\n* @decarpentierg made their first contribution in https://github.com/vllm-project/vllm/pull/46552\r\n* @eparshut made their first contribution in https://github.com/vllm-project/vllm/pull/47467\r\n* @fenghourun made their first contribution in https://github.com/vllm-project/vllm/pull/47384\r\n* @fjosw made their first contribution in https://github.com/vllm-project/vllm/pull/41026\r\n* @guybd made their first contribution in https://github.com/vllm-project/vllm/pull/44029\r\n* @harsha20032020 made their first contribution in https://github.com/vllm-project/vllm/pull/44720\r\n* @hclsys made their first contribution in https://github.com/vllm-project/vllm/pull/44070\r\n* @hhhhhhhhhhhhhhhhho made their first contribution in https://github.com/vllm-project/vllm/pull/46467\r\n* @hillelda made their first contribution in https://github.com/vllm-project/vllm/pull/46069\r\n* @I3eg1nner made their first contribution in https://github.com/vllm-project/vllm/pull/47532\r\n* @imargulis made their first contribution in https://github.com/vllm-project/vllm/pull/46301\r\n* @ItsMatti4 made their first contribution in https://github.com/vllm-project/vllm/pull/45263\r\n* @jesco-absolut made their first contribution in https://github.com/vllm-project/vllm/pull/47589\r\n* @jessiewei7 made their first contribution in https://github.com/vllm-project/vllm/pull/46560\r\n* @jialoop-git made their first contribution in https://github.com/vllm-project/vllm/pull/45159\r\n* @JohnLangford made their first contribution in https://github.com/vllm-project/vllm/pull/46835\r\n* @Jyothirmaikottu made their first contribution in https://github.com/vllm-project/vllm/pull/47250\r\n* @kalyanamdewri made their first contribution in https://github.com/vllm-project/vllm/pull/47517\r\n* @Laurent-Zhang made their first contribution in https://github.com/vllm-project/vllm/pull/47429\r\n* @lcheng321 made their first contribution in https://github.com/vllm-project/vllm/pull/45715\r\n* @LiJzd made their first contribution in https://github.com/vllm-project/vllm/pull/45813\r\n* @lslusarczyk made their first contribution in https://github.com/vllm-project/vllm/pull/43092\r\n* @Lynn-hh made their first contribution in https://github.com/vllm-project/vllm/pull/46543\r\n* @Meihan-chen made their first contribution in https://github.com/vllm-project/vllm/pull/44483\r\n* @nagisa-kunhah made their first contribution in https://github.com/vllm-project/vllm/pull/44124\r\n* @NathanielMcVicar made their first contribution in https://github.com/vllm-project/vllm/pull/45965\r\n* @NicolasHug made their first contribution in https://github.com/vllm-project/vllm/pull/46609\r\n* @omirosh made their first contribution in https://github.com/vllm-project/vllm/pull/44313\r\n* @orestis-z made their first contribution in https://github.com/vllm-project/vllm/pull/46488\r\n* @pranavthakur0-0 made their first contribution in https://github.com/vllm-project/vllm/pull/46306\r\n* @Priyjain-amd made their first contribution in https://github.com/vllm-project/vllm/pull/45818\r\n* @skajre made their first contribution in https://github.com/vllm-project/vllm/pull/46818\r\n* @soaringk made their first contribution in https://github.com/vllm-project/vllm/pull/45810\r\n* @spandantiwari made their first contribution in https://github.com/vllm-project/vllm/pull/46260\r\n* @sriganesh123 made their first contribution in https://github.com/vllm-project/vllm/pull/35076\r\n* @tarjan1 made their first contribution in https://github.com/vllm-project/vllm/pull/45657\r\n* @thisjiang made their first contribution in https://github.com/vllm-project/vllm/pull/45924\r\n* @umarkovi-amd made their first contribution in https://github.com/vllm-project/vllm/pull/46381\r\n* @VectorPeak made their first contribution in https://github.com/vllm-project/vllm/pull/47099\r\n* @wan-danfeng made their first contribution in https://github.com/vllm-project/vllm/pull/38174\r\n* @yangyang-cs95 made their first contribution in https://github.com/vllm-project/vllm/pull/46684\r\n* @yuyue0225sc made their first contribution in https://github.com/vllm-project/vllm/pull/44297\r\n* @zhongjing123 made their first contribution in https://github.com/vllm-project/vllm/pull/47024\r\n* @zhou9402 made their first contribution in https://github.com/vllm-project/vllm/pull/47448\r\n* @ZichenYuan made their first contribution in https://github.com/vllm-project/vllm/pull/46452\r\n\r\n## Contributors\r\n\r\nThank you to all the contributors who made this release possible!\r\n\r\n@AndreasKaratzas, @njhill, @BugenZhao, @hmellor, @yewentao256, @WoosukKwon, @Sunt-ing, @micah-wil, @mgoin, @reidliu41, @peizhang56, @mawong-amd, @TheEpicDolphin, @jeejeelee, @taneem-ibrahim, @chaunceyjiang, @chaojun-zhang, @divakar-amd, @fxmarty-amd, @LopezCastroRoberto, @wzhao18, @mayuyuace, @jperezdealgaba, @noooop, @yzong-rh, @jikunshang, @zxd1997066, @bigPYJ1151, @yma11, @hickeyma, @benchislett, @xianbaoqian, @andakai, @NickLucche, @ivanium, @joerowell, @EazyReal, @mganczarenko, @majunze2001, @hongxiayang, @WindChimeRan, @Rohan138, @tjtanaa, @bbrowning, @thisjiang, @Fangzhou-Ai, @blasrodri, @Isotr0py, @zhenwei-intel, @zyongye, @frida-andersson, @muhammadfawaz1, @lcheng321, @spandantiwari, @Palaiologos1453, @soaringk, @Lynn-hh, @fadara01, @djramic, @Liangliang-Ma, @ronensc, @aarushjain29, @HDCharles, @qianlihuang, @AgenticSpark, @charlifu, @cleonard530, @shen-shanshan, @xaguilar-amd, @xiaohongchen1991, @varun-sundar-rabindranath, @gau-nernst, @tahsintunan, @GirasoleY, @hclsys, @Yejing-Lai, @LucasWilkinson, @matteso1, @akii96, @atalman, @lucianommartins, @I3eg1nner, @rahulssv-ibm, @ZichenYuan, @tanpinsiang, @hillelda, @Srinivasoo7, @Etelis, @Rukhaiya2004, @Oxygen56, @Priyjain-amd, @GuyStone, @nholmber, @CienetStingLin, @xinyu-intel, @JartX, @esmeetu, @hhhhhhhhhhhhhhhhho, @harsha20032020, @walterbm, @Acaciasama, @jessiewei7, @ashwin-phadke, @shivampr, @cyq1017, @kjiang249, @orestis-z, @xyang16, @tianmu-li, @mgehre-amd, @aaarkai, @guybd, @wcynb1023, @Josephasafg, @qyYue1389, @russellb, @haoyangli0109, @sfeng33, @mikekg, @EanWang211123, @ovidiusm, @ItsMatti4, @hyeongyun0916, @qli88, @juliendenize, @calvarado2004, @tdoublep, @brandonpelfrey, @davispuh, @weizhoublue, @jasonozuzu-cohere, @wentian-byte, @skajre, @gty111, @omirosh, @decarpentierg, @fjosw, @ilmarkov, @yuwenzho, @JisoLya, @JohnLangford, @aldenlobo, @bnellnm, @jasonlizhengjian, @zufangzhu, @izhuhaoran, @MatthewBonanni, @deng451e, @ashwing, @sriganesh123, @linitra24, @liranschour, @umarkovi-amd, @aman0603, @adobrzyn, @jwzheng96, @eicherseiji, @ArsalanShakil, @tc-mb, @imargulis, @fangyuchu, @puririshi98, @JeanPaulShapo, @VectorPeak, @tarjan1, @qiching, @Achyuthan-S, @ZJY0516, @lucifer1004, @cinnamonica02, @jmamou, @almayne, @hao-aaron, @Jyothirmaikottu, @andylolu2, @AIvashov, @stevenkuang-tencent, @lcskrishna, @Aneureka, @wan-danfeng, @chengzheng345, @pranavthakur0-0, @zRzRzRzRzRzRzR, @DanBlanaru, @adamkbaranowski, @wendyliu235, @eparshut, @yangyang-cs95, @kalyanamdewri, @maxdebayser, @fenghourun, @tpopp, @okorzh-amd, @labAxiaoming, @sychen52, @ekagra-ranjan, @gausah01, @yuyue0225sc, @cpersson-amd, @lslusarczyk, @alex101-ops, @Zhenzhong1, @velonica0, @zhongjing123, @zhou9402, @llsj14, @majian4work, @akinsella, @BadrBasowid, @afierka-intel, @ayush1399, @LiJzd, @jesco-absolut, @Laurent-Zhang, @Kevin-XiongC, @NathanielMcVicar, @askliar, @ACEEE-1222, @jinzhen-lin, @SherryC41, @simondanielsson, @nv-nedelman-1, @yisustc, @kylesayrs, @jialoop-git, @NicolasHug, @guan404ming, @HumphreySun98, @danielafrimi, @gcanlin, @robertgshaw2-redhat\r\n","reactions":{"url":"https://api.github.com/repos/vllm-project/vllm/releases/351977001/reactions","total_count":42,"+1":0,"-1":0,"laugh":0,"hooray":24,"confused":0,"heart":12,"rocket":6,"eyes":0},"mentions_count":200},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/346469299","assets_url":"https://api.github.com/repos/vllm-project/vllm/releases/346469299/assets","upload_url":"https://uploads.github.com/repos/vllm-project/vllm/releases/346469299/assets{?name,label}","html_url":"https://github.com/vllm-project/vllm/releases/tag/v0.24.0","id":346469299,"author":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"node_id":"RE_kwDOI7xefs4UprOz","tag_name":"v0.24.0","target_commitish":"main","name":"v0.24.0","draft":false,"immutable":false,"prerelease":false,"created_at":"2026-06-28T07:04:08Z","updated_at":"2026-06-30T01:19:02Z","published_at":"2026-06-29T19:41:59Z","assets":[{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/461680015","id":461680015,"node_id":"RA_kwDOI7xefs4bhK2P","name":"vllm-0.24.0+cpu-cp38-abi3-manylinux_2_34_aarch64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":51703820,"digest":"sha256:d91ef4a5616f6270ff2d9b6c400bfd6b04f74eff943f8aaee0e7ffabbd42994f","download_count":165,"created_at":"2026-06-30T01:16:50Z","updated_at":"2026-06-30T01:16:55Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.24.0/vllm-0.24.0%2Bcpu-cp38-abi3-manylinux_2_34_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/461680016","id":461680016,"node_id":"RA_kwDOI7xefs4bhK2Q","name":"vllm-0.24.0+cpu-cp38-abi3-manylinux_2_34_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":112627111,"digest":"sha256:de070be9a46668418e6618993daa41d09218cf091bbf74b2681eb7fab674cb5f","download_count":5949,"created_at":"2026-06-30T01:16:50Z","updated_at":"2026-06-30T01:16:59Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.24.0/vllm-0.24.0%2Bcpu-cp38-abi3-manylinux_2_34_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/461680017","id":461680017,"node_id":"RA_kwDOI7xefs4bhK2R","name":"vllm-0.24.0+cu129-cp38-abi3-manylinux_2_28_aarch64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":472109373,"digest":"sha256:e85edf971a25582993ef3bfd26b0d6ebf498d5ca4ae4944fa8824a734b81ac58","download_count":140357,"created_at":"2026-06-30T01:16:50Z","updated_at":"2026-06-30T01:17:18Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.24.0/vllm-0.24.0%2Bcu129-cp38-abi3-manylinux_2_28_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/461680013","id":461680013,"node_id":"RA_kwDOI7xefs4bhK2N","name":"vllm-0.24.0+cu129-cp38-abi3-manylinux_2_28_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":504590984,"digest":"sha256:597949743f2a00c0539d9e2ff0f67b32c608a378e973f99e7529fd6fa9445f70","download_count":172230,"created_at":"2026-06-30T01:16:50Z","updated_at":"2026-06-30T01:17:17Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.24.0/vllm-0.24.0%2Bcu129-cp38-abi3-manylinux_2_28_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/461680014","id":461680014,"node_id":"RA_kwDOI7xefs4bhK2O","name":"vllm-0.24.0-cp38-abi3-manylinux_2_28_aarch64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":271361241,"digest":"sha256:700db71c3cf14697d42583521f38b12fac38db1e7a8ad062e8e4d63a5dadebd5","download_count":2608,"created_at":"2026-06-30T01:16:50Z","updated_at":"2026-06-30T01:17:09Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.24.0/vllm-0.24.0-cp38-abi3-manylinux_2_28_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/461680067","id":461680067,"node_id":"RA_kwDOI7xefs4bhK3D","name":"vllm-0.24.0-cp38-abi3-manylinux_2_28_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":279209310,"digest":"sha256:2d2831aeba311292250df0132dbc4d8e9f42c654536eaec48e6fe58acb1822cf","download_count":31362,"created_at":"2026-06-30T01:16:55Z","updated_at":"2026-06-30T01:17:09Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.24.0/vllm-0.24.0-cp38-abi3-manylinux_2_28_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/461680125","id":461680125,"node_id":"RA_kwDOI7xefs4bhK39","name":"vllm-0.24.0.tar.gz","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/x-gtar","state":"uploaded","size":37236989,"digest":"sha256:0862453adc1f3339f1a0c9dca1179c34d6ed6e118f87b6e5bddd120af614ac66","download_count":3435,"created_at":"2026-06-30T01:17:00Z","updated_at":"2026-06-30T01:17:03Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.24.0/vllm-0.24.0.tar.gz"}],"tarball_url":"https://api.github.com/repos/vllm-project/vllm/tarball/v0.24.0","zipball_url":"https://api.github.com/repos/vllm-project/vllm/zipball/v0.24.0","body":"# vLLM v0.24.0 Release Notes\r\n\r\n## Highlights\r\n\r\nThis release features 571 commits from 256 contributors (77 new)!\r\n\r\n* **MiniMax-M3**: Added support for the new **MiniMax-M3** model (#45381), with a fast follow-on of BF16/FP8 indexer via MSA (#45892), MXFP4 support (#45896), FP8 sparse GQA (#45744), and extensive AMD/ROCm tuning — mxfp8 MoE/linear on gfx950 (#45725), fp8_per_channel for bf16 weights on MI300X (#45854), FP8 KV-cache fix (#45720), and packed-modules mapping (#45794). A MiniMax-M2 perf regression was also fixed (#45935).\r\n* **DeepSeek-V4 keeps maturing**: Following its debut, DeepSeek-V4 received another large optimization pass — a FlashInfer sparse index cache (2–4% TTFT) (#45863), prefill chunk-planning optimization (4% E2E throughput) (#45061), a cluster-cooperative topK kernel for low-latency (#43008), contiguous per-block KV allocations (#44577), TEP=16 for the block-FP8 shared expert (#46001), and native DSA indexer decode for `next_n > 2` on SM100 (#45322). It is now enabled on **SM120** alongside GLM-5.1 (#43477), with XPU (#44144, #44517, #45240) and ROCm (#44899, #45103, #45681) attention/MoE paths added.\r\n* **Model Runner V2 (MRv2) continues to expand**: MRv2 now **supports quantized models by default** (#44446), enables **GraniteMoE by default** (#45461), and gained migration of Qwen + DeepSeek-V2 MoE models (#42667), DFlash speculative decoding (#44586), and more accurate FP32 Gumbel sampling (#45996).\r\n* **Streaming Parser Engine**: A new streaming parser engine unifies tool-call/reasoning parsing across models, with parsers for Qwen3 (#45413), MiniMax-M2 (#45701), GLM-4.7/5.1/5.2 (#45915), and Nemotron V3 (#45755).\r\n* **Diffusion LLMs**: Added **DiffusionGemma** (#45163), including a CPU path (#45690) and structured-output guardrails for diffusion decoders (#45468).\r\n* **WideEP / DeepEP v2**: Integrated **DeepEP v2** for expert parallelism (#41183), with follow-on robustness fixes (#46404, #46432).\r\n* **Rust frontend matures further**: Added API-key authentication (#44321), CORS (#45753), `/tokenize` + `/detokenize` (#44222), `/pause` `/resume` `/is_paused` (#44499), `/abort_requests` (#44382), `/get_world_size` (#44801), `thinking_token_budget` (#46137), a Python bridge for Rust tool parsers (#44624), and many new parsers and validation paths.\r\n* **Device selection change**: vLLM no longer sets `CUDA_VISIBLE_DEVICES` internally; a new `device_ids` argument is provided instead (#45026). On ROCm, a deprecation window for `CUDA_VISIBLE_DEVICES` has begun (#46636).\r\n\r\n### Model Support\r\n* **New models**: MiniMax-M3 (#45381), DiffusionGemma (#45163) + Gemma Diffusion on CPU (#45690), Hierarchical Reasoning Model — Text / HrmTextForCausalLM (#43098), OpenMOSS (#44124).\r\n* **Gemma 4**: Unified FlashAttention (FA4) across all layers + `mm_prefix` support (#42175); many parser/serving fixes — forced-JSON skip for required/named tool choice (#45795), parsing with thinking disabled (#45832), streaming reasoning-state init (#45852), reasoning rendering on assistant turns (#45867), offline-parser truncation/token-leak fix (#45553); legacy Gemma4 parsers replaced with an engine-based implementation (#45588).\r\n* **DeepSeek-V4**: OOM fix (#44914), MTP projection prefixing (#44821), supported KV-cache dtypes (#44892).\r\n* **Qwen / multimodal**: Qwen3-VL video loader (#44412), Qwen2-VL/Qwen2.5-VL processor-mapped video loader (#45555), Qwen3-VL multi-video processing optimization (#46026) and multi-video crash fix (#46305), Qwen3-Omni VIT cu_seqlens device fix (#44264), fused qk-rmsnorm-rope-gate for Qwen3.5 (#44176), Qwen3.5 EP weight-loading fix (#45002).\r\n* **ViT full CUDA graph**: GLM-4.1V (#40576), DeepSeek-OCR dual-path (#43586), Kimi-VL (#41992), mllama4 (#40660), Lfm2VL encoder (#44930).\r\n* **Other model fixes**: Llama4 weight loading (#45047) and streamed loading to avoid host-OOM (#44645), MiMo v2.x QKV TP sharding + FP4 (#45200), ColQwen3.5 retrieval correctness (#46108), EXAONE-4.5 vision encoder (#45073), MiDashengLM TP>1 audio-encoder crash (#44408), MiniCPM-o/V device-placement and image-size fixes (#43844, #42332, #44980, #45244), Cohere2 MoE weight loading + parser (#44747, #44907), Nemotron V3 reasoning-as-content (#39091), ColBERT AutoWeightsLoader + query/document embedding io processor (#44999, #45210).\r\n* **Kernels**: GLM-5 TRT-LLM ragged MLA prefill dimensions (#43525), GLM-5 router GEMM (#46385).\r\n\r\n### Engine Core\r\n* **Model Runner V2**: Quantized models by default (#44446), GraniteMoE default (#45461), Qwen/DSv2 MoE migration (#42667), DFlash (#44586), simplified async output handling (#45442), attention-group split on `num_heads_q` (#45564), LoRA warmup fix (#35536), more accurate FP32 Gumbel sampling (#45996), `min_tokens` off-by-one fix in the V2 GPU sampler (#46243), plus assorted model/config compatibility fixes (#45868).\r\n* **Speculative decoding**: Dynamic SD (#32374); DFlash with FlashInfer (#43081), mixed KV page sizes (#45181), and Qwen3Next targets (#45319); EAGLE3 support for Qwen3 (#43132); reduced TP communication for large-vocab drafts (#39419); race fix in async accepted counts (#45100); EAGLE multimodal encoder cache fixes (#46315).\r\n* **KV cache & scheduler**: KV-cache watermark to reduce preemptions (#44594), two-phase allocation for cross-group prefix-cache hits (#44409), Marconi-style admission policy for hybrid cache (#37898), prefix-cache retention for Mamba/linear attention (#45845), DS Mamba tail-copy for MTP align mode (#45473), reduced scheduler copy overhead (#45840).\r\n* **Attention**: Re-enabled cross-layer KV cache layout for MLA via stride-aware kernels (#45111), MLA prefill FA4 fp8 output (#43050), FlexAttention custom mask mods made fully cudagraphable (#45232), triton diff-kv backend for MiMo (#41797), FlashMLA sparse accuracy fix (#36616).\r\n* **Weight loading & core**: fastsafetensors `ParallelLoader` for weight loading (#40183), release of cached device memory under pressure on UMA GPUs (#45179), structured outputs for beam search (#35022), `device_ids` arg / no internal `CUDA_VISIBLE_DEVICES` (#45026), graceful fallback when `numactl --membind` is blocked (#45438), config-class registration before tokenizer init (#40299), async scheduling with prompt embeds for multimodal models (#45673).\r\n\r\n### Large Scale Serving & Distributed\r\n* **Expert parallel**: DeepEP v2 integration (#41183) with token-bound and topk-index fixes (#46404, #46432); NIXL EP — DBO with NIXL EP (#45275), top-k index dtype query (#45298), NVFP4 post-receive quantization skip (#45606), elastic-EP communicator (#45013); reject NCCL-based EPLB with async EPLB (#44978).\r\n* **KV connectors / disaggregated serving**: KV push from prefill to decode via NIXL (#35264); per-region KV transfer classification for mixed full-attn + MLA groups (#44583); Mooncake pipeline-parallel PD support (#44528), async lookup (#45659), compact chunk-hash zero-copy lookup (#45969), SWA-block skipping (#45444); P/D fixes with DP supervisor (#46628) and DSV4 disaggregation (#45831); removed `P2pNcclConnector` (#44854).\r\n* **KV offloading**: Multi-tier async batched lookup (#44193), packed HMA KV-cache layout (#46205, gated #46252), parallel-agnostic fs-tier cache (#44733), offloading-manager stats (#35669) and labeled/CPU-usage metrics (#45957, #45737), self-describing KV events (#43468), non-blocking idle flush (#45595), and numerous correctness/race fixes (#44784, #45823, #46231, #46278).\r\n* **Distributed core**: Prefill step cadence for better non-PD DP balancing (#44558), KV-event map encoding (#42892), one-shot fused all-reduce PDL NaN fix (#45448).\r\n\r\n### Hardware & Performance\r\n* **NVIDIA / kernels**: SM90 CUTLASS FP8 mm odd-M support via swap_ab (180–290% kernel speedup) (#44572), tuned `fused_moe` FP8 for Qwen3-Next-80B on H100 (+25%) (#44830), native DSA indexer decode on SM100 (#45322), cluster-cooperative topK for DeepSeek low-latency (#43008), PDL support for DeepGEMM (#46006), FlashInfer cutedsl NVFP4 GEMM (#42235) and cute-dsl MXFP8 linear kernel (#46393), new Helion kernels for FP8/RMSNorm quant (#36902, #33790, #36895, #34432).\r\n* **torch stable ABI**: Continued (and completed) migration of kernels to the libtorch stable ABI — MoE [10c/n] (#44565), Marlin [11a/n] (#45176), Machete [11b/n] (#45304), final `_C` library migration [12/n] (#45415).\r\n* **AMD ROCm**: Torch 2.11 (#45362); fused AR + RMSNorm + per-group FP8 quant (#42864), fused softplus-sqrt-topk MoE router under AITER (#44945), DSv4 flash-decode split-K kernel (#44899) and inverse-RoPE fusion (#45103), W4A16 FlyDSL MoE (#44400), A8W4 MoE CDNA4 swizzle gate for gpt-oss (#44804); deprecation window begun for `CUDA_VISIBLE_DEVICES` on ROCm (#46636).\r\n* **Intel XPU**: Sequence-parallel support (#38608), torch-xpu 2.12 (#42262), vllm-xpu-kernels v0.1.10 (#40367), W4A16 int4 group_size=32 MoE (#45136), DeepSeek-V4 attention/MoE paths (#44144, #44517, #45240), top-p sampling correctness fix (#44470).\r\n* **CPU & other architectures**: 2.5× faster ASR CPU preprocessing via multi-threading (#44612), CPU W4A16 INT4 MoE (#43409), cgroup memory-limit-aware KV cache sizing (#45086), RISC-V oneDNN W8A8 INT8 (#44478) and RVV micro-GEMM for WNA16 (#44324), pinned memory for WSL2 (#41496), ZenCPU runtime logging (#42726).\r\n* **TPU**: tpu-inference upgraded to v0.22.1 (#45793).\r\n* **Misc perf**: `VLLM_TRITON_FORCE_FIRST_CONFIG` to skip Triton autotuning (#42425), Triton recompile detection (#45631), fused multi-group block-table staged writes (#44944).\r\n\r\n### Quantization\r\n* **Online & mixed-precision**: Online FP8 per-token-per-channel (PTPC) quantization (#44132); `modelopt_mixed` support extended to Ampere/SM80-86 (#45306) and Turing/SM75 (#45375).\r\n* **FP4 / MXFP**: FlashInfer cutedsl NVFP4 GEMM backend (#42235) and cute-dsl MXFP8 linear kernel (#46393), MXFP4 W4A4 MoE CUTLASS E8M0 scale fix (#43557), SwiGLU clamp wired for NVFP4 MoE on non-Blackwell (#45836), `flashinfer_cutlass` allowed as a clamped NVFP4 MoE backend (#46492), NVFP4/OCP MX MoE emulation fix (#46254), FP8 MoE re-enabled on NVIDIA Thor (#46339).\r\n* **GGUF / compressed-tensors / AWQ**: GGUF quantization migrated to a plugin (#39612), compressed-tensors WNA16 MoE actorder fix (#41161) and KV-cache-scheme rejection (#45312), AWQ format on XPU (#43404) and AWQ dequantize fix on Intel XPU (#42727).\r\n* **Kernels & correctness**: QuantizedActivation linear-kernel contract (#44260), consolidated Marlin thread-tile padding (#45295), FP8 weight layout canonicalized to (K, N) (#44735), corrupt-output fix for MoE FP8 with LoRAs loaded (#42120), symmetric-quant regression fix in GPTQ/CT MoE (#45656), `fp8_e5m2` KV cache allowed for non-fp8 checkpoints (#45040).\r\n\r\n### API & Frontend\r\n* **Tool calling & parsing**: Strict mode for tool calling in Chat Completions (#45003) and Responses API (#45396); new Streaming Parser Engine (#45413) with Qwen3, MiniMax-M2 (#45701), GLM-4.7/5.1/5.2 (#45915), Nemotron V3 (#45755) parsers; unified Parser consolidation in chat serving (#45548); numerous parser correctness fixes (#46047, #46091, #46159, #45763, #46351, #43984).\r\n* **OpenAI / Responses**: Real `/v1/embeddings` support for messages + `chat_template_kwargs` (#45173), multimodal token counts in `usage.prompt_tokens_details` (#45458), omit empty `tool_calls` from chat responses (#44105), Responses API streaming `function_call` id fix (#44608), Harmony refactor of streaming/non-streaming paths (#45171, #45104).\r\n* **Anthropic Messages API**: Cache-usage reporting in `/v1/messages` (#40912), mid-conversation system-message handling (#46025), inline system-message position preserved for prefix caching (#44602), `tool_use` argument-dropping fix (#45287).\r\n* **Rust frontend**: API-key auth (#44321), CORS (#45753), `/tokenize` + `/detokenize` (#44222), `/pause` `/resume` `/is_paused` (#44499), `/abort_requests` (#44382), `/get_world_size` (#44801), `thinking_token_budget` (#46137), `parallel_tool_calls=false` (#44760), continuous usage stats (#43965), model metadata in `/v1/models` (#45950), Python bridge for Rust tool parsers (#44624), dedicated runtime for HTTP/ZMQ (#46051), and many validation/correctness fixes.\r\n* **Metrics**: `vllm:tool_call_parser_invocations_total` (#44448), group-aware KV cache capacity in `vllm:cache_config_info` (#42206), MLA attention metrics for DeepSeek MFU estimation (#39457).\r\n* **Pooling / embeddings**: Validation for Cohere `/v2/embed` input exclusivity (#45640), non-negative rerank `top_n` (#46119), matryoshka embedding dimension bounds (#46313).\r\n* **Benchmarks**: BFCL tool-calling dataset for `vllm bench serve` (#42457), multi-turn benchmark api_key/custom headers (#44516), tokenizer-mismatch auto-correction (#44708).\r\n\r\n### Security\r\n\r\nThis release ships another coordinated security-hardening batch (much of it from security researcher @jperezdealgaba).\r\n\r\n* **Denial of service**: Audio decompression bomb in the speech-to-text endpoint (#44970), remote DoS via invalid recovered-token reinjection in speculative decoding (#44744), DoS via `prompt_embeds` on M-RoPE models (#45252), regex-compilation timeout guard in structured outputs (#45118), audio upload size limit before full materialization (#45510), audio decode duration limit in the chat-completions path (#45908).\r\n* **Information disclosure**: int32 truncation in the GGUF dequantize kernels (#44971).\r\n* **Input validation & hardening**: Image EXIF orientation and tRNS transparency handling (#44974), rejection of non-finite `temperature`/`repetition_penalty` (#45116), `sanitize_message` applied to Anthropic and STT error paths (#45119).\r\n* **Dependencies**: Upgrade Starlette to ≥ 1.0.1 to fix CVE-2026-48710 (#45675).\r\n\r\n### Dependencies\r\n* Torch 2.11 on ROCm (#45362), torch-xpu 2.12 (#42262), tpu-inference v0.22.1 (#45793), NIXL v0.10.1 for XPU (#40287), Starlette ≥ 1.0.1 (#45675).\r\n* `mistral_common` is now optional via deferred import (#45305); CUDA Dockerfiles upgraded from GCC 10 to GCC 12 for C++20 (#44923); spinloop extension skipped on Python < 3.11 (#44783).\r\n\r\n### Deprecations & Removals\r\n* **Removed models**: ERNIE (obsolete) (#45127), Xverse (#45638), Dots1 (#45637), Bamba (#45990), Mono-InternVL (#45129), InternLM registry alias (#45128).\r\n* **Deprecated**: First-generation Qwen and QwenVL models (#45131), Transformers v4 support (#45161), `CUDA_VISIBLE_DEVICES` on ROCm (#46636); general deprecations for v0.23/v0.24 (#44992).\r\n\r\n## New Contributors\r\n\r\n* @abcd1927 made their first contribution in https://github.com/vllm-project/vllm/pull/43098\r\n* @Achyuthan-S made their first contribution in https://github.com/vllm-project/vllm/pull/44795\r\n* @Alex-ai-future made their first contribution in https://github.com/vllm-project/vllm/pull/45905\r\n* @alexbi29 made their first contribution in https://github.com/vllm-project/vllm/pull/45763\r\n* @amanchugh89 made their first contribution in https://github.com/vllm-project/vllm/pull/45840\r\n* @ankrovv made their first contribution in https://github.com/vllm-project/vllm/pull/44608\r\n* @anony-mous-e made their first contribution in https://github.com/vllm-project/vllm/pull/45412\r\n* @appleparan made their first contribution in https://github.com/vllm-project/vllm/pull/45073\r\n* @ashishpatel26 made their first contribution in https://github.com/vllm-project/vllm/pull/43984\r\n* @Bot1822 made their first contribution in https://github.com/vllm-project/vllm/pull/44053\r\n* @ByteFlowing1337 made their first contribution in https://github.com/vllm-project/vllm/pull/45988\r\n* @Change72 made their first contribution in https://github.com/vllm-project/vllm/pull/43756\r\n* @coder3101 made their first contribution in https://github.com/vllm-project/vllm/pull/44801\r\n* @cquil11 made their first contribution in https://github.com/vllm-project/vllm/pull/45720\r\n* @dmaniloff made their first contribution in https://github.com/vllm-project/vllm/pull/40470\r\n* @factnn made their first contribution in https://github.com/vllm-project/vllm/pull/44955\r\n* @FAUST-BENCHOU made their first contribution in https://github.com/vllm-project/vllm/pull/44760\r\n* @felix0080 made their first contribution in https://github.com/vllm-project/vllm/pull/44602\r\n* @gitbisector made their first contribution in https://github.com/vllm-project/vllm/pull/40183\r\n* @gq112 made their first contribution in https://github.com/vllm-project/vllm/pull/43081\r\n* @guan404ming made their first contribution in https://github.com/vllm-project/vllm/pull/35022\r\n* @HanHan009527 made their first contribution in https://github.com/vllm-project/vllm/pull/44528\r\n* @hello-args made their first contribution in https://github.com/vllm-project/vllm/pull/44109\r\n* @HumphreySun98 made their first contribution in https://github.com/vllm-project/vllm/pull/45466\r\n* @j-i-l made their first contribution in https://github.com/vllm-project/vllm/pull/45319\r\n* @JasonLi314 made their first contribution in https://github.com/vllm-project/vllm/pull/45255\r\n* @jeffye-dev made their first contribution in https://github.com/vllm-project/vllm/pull/43595\r\n* @jimmy-evo made their first contribution in https://github.com/vllm-project/vllm/pull/44516\r\n* @jjppp made their first contribution in https://github.com/vllm-project/vllm/pull/45217\r\n* @JOSH1024 made their first contribution in https://github.com/vllm-project/vllm/pull/44784\r\n* @junkang1991 made their first contribution in https://github.com/vllm-project/vllm/pull/46039\r\n* @KaletoAI made their first contribution in https://github.com/vllm-project/vllm/pull/43495\r\n* @kliukovkin made their first contribution in https://github.com/vllm-project/vllm/pull/43724\r\n* @littlecircle0730 made their first contribution in https://github.com/vllm-project/vllm/pull/44750\r\n* @llx-08 made their first contribution in https://github.com/vllm-project/vllm/pull/45357\r\n* @m4r1k made their first contribution in https://github.com/vllm-project/vllm/pull/45795\r\n* @martin-kukla made their first contribution in https://github.com/vllm-project/vllm/pull/45417\r\n* @MichaelCao0 made their first contribution in https://github.com/vllm-project/vllm/pull/46398\r\n* @mrn3088 made their first contribution in https://github.com/vllm-project/vllm/pull/45383\r\n* @nataliepjlin made their first contribution in https://github.com/vllm-project/vllm/pull/45218\r\n* @nehmathe2 made their first contribution in https://github.com/vllm-project/vllm/pull/44912\r\n* @nikhilesh-csa made their first contribution in https://github.com/vllm-project/vllm/pull/45852\r\n* @nv-nedelman-1 made their first contribution in https://github.com/vllm-project/vllm/pull/42120\r\n* @Oseltamivir made their first contribution in https://github.com/vllm-project/vllm/pull/45879\r\n* @parthash0804 made their first contribution in https://github.com/vllm-project/vllm/pull/43844\r\n* @pjdurden made their first contribution in https://github.com/vllm-project/vllm/pull/44942\r\n* @pst2154 made their first contribution in https://github.com/vllm-project/vllm/pull/45181\r\n* @Saddss made their first contribution in https://github.com/vllm-project/vllm/pull/44409\r\n* @sahilsGit made their first contribution in https://github.com/vllm-project/vllm/pull/44499\r\n* @sasindharan made their first contribution in https://github.com/vllm-project/vllm/pull/44383\r\n* @shantipriya-amd made their first contribution in https://github.com/vllm-project/vllm/pull/39498\r\n* @Sirius29 made their first contribution in https://github.com/vllm-project/vllm/pull/46026\r\n* @srajabos made their first contribution in https://github.com/vllm-project/vllm/pull/44665\r\n* @sridhar-3009 made their first contribution in https://github.com/vllm-project/vllm/pull/44055\r\n* @stefankoncarevic made their first contribution in https://github.com/vllm-project/vllm/pull/45706\r\n* @sunnweiwei made their first contribution in https://github.com/vllm-project/vllm/pull/45100\r\n* @TanNgocDo made their first contribution in https://github.com/vllm-project/vllm/pull/44222\r\n* @thisisjimmyfb made their first contribution in https://github.com/vllm-project/vllm/pull/41496\r\n* @tykow made their first contribution in https://github.com/vllm-project/vllm/pull/44663\r\n* @V-3604 made their first contribution in https://github.com/vllm-project/vllm/pull/43362\r\n* @vincentzed made their first contribution in https://github.com/vllm-project/vllm/pull/44930\r\n* @vraiti made their first contribution in https://github.com/vllm-project/vllm/pull/42331\r\n* @wangjiaxin99 made their first contribution in https://github.com/vllm-project/vllm/pull/45794\r\n* @waynehacking8 made their first contribution in https://github.com/vllm-project/vllm/pull/45376\r\n* @x41lakazam made their first contribution in https://github.com/vllm-project/vllm/pull/43300\r\n* @xiaguan made their first contribution in https://github.com/vllm-project/vllm/pull/45286\r\n* @xiaohuguo2023 made their first contribution in https://github.com/vllm-project/vllm/pull/44804\r\n* @xin3he made their first contribution in https://github.com/vllm-project/vllm/pull/43557\r\n* @xx-thomas made their first contribution in https://github.com/vllm-project/vllm/pull/45210\r\n* @yangdian96 made their first contribution in https://github.com/vllm-project/vllm/pull/44173\r\n* @YellowFoxH4XOR made their first contribution in https://github.com/vllm-project/vllm/pull/45057\r\n* @yzhan1 made their first contribution in https://github.com/vllm-project/vllm/pull/44552\r\n* @Zedong-Liu made their first contribution in https://github.com/vllm-project/vllm/pull/45361\r\n* @ZewenShen-Cohere made their first contribution in https://github.com/vllm-project/vllm/pull/41161\r\n* @zhangshuoming990105 made their first contribution in https://github.com/vllm-project/vllm/pull/40912\r\n* @ZiguanWang made their first contribution in https://github.com/vllm-project/vllm/pull/43981\r\n* @zlxi02 made their first contribution in https://github.com/vllm-project/vllm/pull/44595\r\n\r\n## Contributors\r\n\r\nThank you to everyone who made this release possible!\r\n\r\n@yewentao256, @Sunt-ing, @jperezdealgaba, @AndreasKaratzas, @BugenZhao, @sfeng33, @njhill, @micah-wil, @bbrowning, @mgoin, @jeejeelee, @hmellor, @tlrmchlsmth, @xianbaoqian, @mmangkad, @jikunshang, @Dao007forever, @zhenwei-intel, @noooop, @Isotr0py, @ivanium, @reidliu41, @varun-sundar-rabindranath, @chaunceyjiang, @WoosukKwon, @mawong-amd, @zxd1997066, @chaojun-zhang, @NickLucche, @bigPYJ1151, @ZJY0516, @charlifu, @yzong-rh, @divakar-amd, @khluu, @cleonard530, @wseaton, @xiaohongchen1991, @ywang96, @taneem-ibrahim, @mikekg, @itayalroy, @Alex-ai-future, @sahilsGit, @bnellnm, @littlecircle0730, @majian4work, @ricky-chaoju, @ronensc, @Fangzhou-Ai, @lucianommartins, @Srinivasoo7, @zyongye, @Rohan138, @Etelis, @wentian-byte, @ekagra-ranjan, @LucasWilkinson, @tahsintunan, @waynehacking8, @gau-nernst, @tuukkjs, @stefankoncarevic, @Palaiologos1453, @lucifer1004, @jmamou, @liulanze, @Terrencezzj, @Change72, @LopezCastroRoberto, @he-yufeng, @benchislett, @juliendenize, @s3woz, @panpan0000, @ilmarkov, @zixi-qi, @wcynb1023, @fynnsu, @ZhanqiuHu, @yuwenzho, @tdoublep, @MatthewBonanni, @hickeyma, @majunze2001, @mrn3088, @Yejing-Lai, @vllmellm, @Saddss, @DarkLight1337, @hongxiayang, @m4r1k, @qli88, @jonathanc-n, @felix0080, @djramic, @aoshen02, @fxmarty-amd, @simon-mo, @llsj14, @akii96, @walterbm, @dmaniloff, @zlxi02, @grYe99, @jeffye-dev, @parthash0804, @qyYue1389, @sagearc, @maeehart, @TanNgocDo, @cinnamonica02, @zucchini-nlp, @tykow, @mganczarenko, @yangdian96, @jimmy-evo, @YellowFoxH4XOR, @yzhan1, @shenoyvvarun, @yufufi, @laviier, @xiaohuguo2023, @EanWang211123, @JartX, @shantipriya-amd, @askliar, @hallerite, @appleparan, @effi-ofer, @angelayi, @TheCodeWrangler, @DanBlanaru, @ankrovv, @velonica0, @pjdurden, @cyyever, @wjinxu, @kliukovkin, @x41lakazam, @Jasen2201, @r-barnes, @tc-mb, @nataliepjlin, @KaletoAI, @WineChord, @fangyuchu, @vraiti, @nascheme, @jjppp, @sasindharan, @xiaguan, @snadampal, @chfeng-cs, @thillai-c, @guan404ming, @sridhar-3009, @vincentzed, @j-i-l, @rjrock, @abinggo, @anony-mous-e, @Achyuthan-S, @Harry-Chen, @mfylcek, @amd-asalykov, @noa-neria, @maobaolong, @TheEpicDolphin, @FAUST-BENCHOU, @martin-kukla, @xin3he, @ZiguanWang, @youkaichao, @factnn, @llx-08, @xx-thomas, @gitbisector, @Bortlesboat, @thisisjimmyfb, @JOSH1024, @wendyliu235, @wangxiyuan, @shen-shanshan, @HanHan009527, @amd-lalithnc, @netanel-haber, @fuscof-ibm, @AjAnubolu, @carlyou, @abcd1927, @CienetStingLin, @kouroshHakha, @alexbi29, @jesse996, @sungsooha, @andakai, @cquil11, @nehmathe2, @liangel-02, @hello-args, @j9smith, @nikhilesh-csa, @ruocco, @oguzhankir, @yiliu30, @xaguilar-amd, @amirkl94, @danisereb, @wangjiaxin99, @shanjiaz, @Oseltamivir, @alexeldeib, @wzhao18, @coder3101, @lyd1992, @markmc, @ashishpatel26, @HumphreySun98, @ByteFlowing1337, @nv-nedelman-1, @JaredforReal, @sammshen, @okorzh-amd, @muhammadfawaz1, @vadiklyutiy, @JasonLi314, @SumanthRH, @Sirius29, @tjtanaa, @zhangshuoming990105, @amanchugh89, @umut-polat, @srajabos, @junkang1991, @pst2154, @WindChimeRan, @Zedong-Liu, @gq112, @sunnweiwei, @athrael-soju, @EazyReal, @Liangliang-Ma, @jinzhen-lin, @V-3604, @aarushjain29, @ZewenShen-Cohere, @Bot1822, @BowenBao, @MichaelCao0, @tanpinsiang, @QwertyJack, @nagisa-kunhah, @Meihan-chen, @robertgshaw2-redhat\r\n","reactions":{"url":"https://api.github.com/repos/vllm-project/vllm/releases/346469299/reactions","total_count":60,"+1":32,"-1":0,"laugh":1,"hooray":8,"confused":0,"heart":7,"rocket":12,"eyes":0},"mentions_count":200},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/338844829","assets_url":"https://api.github.com/repos/vllm-project/vllm/releases/338844829/assets","upload_url":"https://uploads.github.com/repos/vllm-project/vllm/releases/338844829/assets{?name,label}","html_url":"https://github.com/vllm-project/vllm/releases/tag/v0.23.0","id":338844829,"author":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"node_id":"RE_kwDOI7xefs4UMlyd","tag_name":"v0.23.0","target_commitish":"main","name":"v0.23.0","draft":false,"immutable":false,"prerelease":false,"created_at":"2026-06-15T03:35:17Z","updated_at":"2026-06-15T05:27:20Z","published_at":"2026-06-15T05:27:20Z","assets":[{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/446465076","id":446465076,"node_id":"RA_kwDOI7xefs4anIQ0","name":"vllm-0.23.0+cpu-cp38-abi3-manylinux_2_34_aarch64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":50461326,"digest":"sha256:3144907b4d3e91ed52a52bf1f8682ffa6100cb2cc489aa25343e400ea10145ef","download_count":1482,"created_at":"2026-06-13T09:15:41Z","updated_at":"2026-06-13T09:15:57Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.23.0/vllm-0.23.0%2Bcpu-cp38-abi3-manylinux_2_34_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/446465075","id":446465075,"node_id":"RA_kwDOI7xefs4anIQz","name":"vllm-0.23.0+cpu-cp38-abi3-manylinux_2_34_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":110644171,"digest":"sha256:571a4ac24629d6100479cdf20d67b4c5cd1dd9bab1b485cd4f3bd631c961187d","download_count":94499,"created_at":"2026-06-13T09:15:41Z","updated_at":"2026-06-13T09:16:09Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.23.0/vllm-0.23.0%2Bcpu-cp38-abi3-manylinux_2_34_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/446465081","id":446465081,"node_id":"RA_kwDOI7xefs4anIQ5","name":"vllm-0.23.0+cu129-cp38-abi3-manylinux_2_28_aarch64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":467263165,"digest":"sha256:17ac76bbfad014d5b584a9499bca22d3f871d24db93cbb1eb27a87deef0219aa","download_count":36476,"created_at":"2026-06-13T09:15:41Z","updated_at":"2026-06-13T09:16:32Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.23.0/vllm-0.23.0%2Bcu129-cp38-abi3-manylinux_2_28_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/446465080","id":446465080,"node_id":"RA_kwDOI7xefs4anIQ4","name":"vllm-0.23.0+cu129-cp38-abi3-manylinux_2_28_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":500028428,"digest":"sha256:8bc2203995d061e6b988916b71b9dee8a5970f5fdc5f37d4445a877a2fab2cc1","download_count":89479,"created_at":"2026-06-13T09:15:41Z","updated_at":"2026-06-13T09:16:34Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.23.0/vllm-0.23.0%2Bcu129-cp38-abi3-manylinux_2_28_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/446465078","id":446465078,"node_id":"RA_kwDOI7xefs4anIQ2","name":"vllm-0.23.0-cp38-abi3-manylinux_2_28_aarch64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":265953967,"digest":"sha256:6a1a534f81f0b62f53d73faa68c73dfae540292ace7f97baf30dbac94fe90f2c","download_count":19022,"created_at":"2026-06-13T09:15:41Z","updated_at":"2026-06-13T09:16:24Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.23.0/vllm-0.23.0-cp38-abi3-manylinux_2_28_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/446465221","id":446465221,"node_id":"RA_kwDOI7xefs4anITF","name":"vllm-0.23.0-cp38-abi3-manylinux_2_28_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":274070208,"digest":"sha256:71eae985c79ddaa999328cc56d206a1e9b785e079fc6da9e2359ec56ef1c842a","download_count":23255,"created_at":"2026-06-13T09:15:58Z","updated_at":"2026-06-13T09:16:23Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.23.0/vllm-0.23.0-cp38-abi3-manylinux_2_28_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/446465349","id":446465349,"node_id":"RA_kwDOI7xefs4anIVF","name":"vllm-0.23.0.tar.gz","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/x-gtar","state":"uploaded","size":36624042,"digest":"sha256:760269db3d9611e12e524681df1bca0977d5d2f5fcb4481cc34d33efc4ae7ff5","download_count":4394,"created_at":"2026-06-13T09:16:10Z","updated_at":"2026-06-13T09:16:13Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.23.0/vllm-0.23.0.tar.gz"}],"tarball_url":"https://api.github.com/repos/vllm-project/vllm/tarball/v0.23.0","zipball_url":"https://api.github.com/repos/vllm-project/vllm/zipball/v0.23.0","body":"# vLLM v0.23.0 Release Notes\r\n\r\nPlease note that Minimax M3 is not yet supported in this version. Please follow [vLLM recipe](https://recipes.vllm.ai/MiniMaxAI/MiniMax-M3) for usage guides for M3.\r\n\r\n## Highlights\r\n\r\nThis release features 408 commits from 200 contributors (63 new)!\r\n\r\n* **DeepSeek-V4 matures across backends**: Following its introduction in v0.22.0, DeepSeek-V4 received another large hardening and optimization pass. Its sparse MLA metadata is now decoupled from DeepSeek-V3.2 (#44699), it gained a TRTLLM-gen attention kernel (#43827), EPLB support for the Mega-MoE (#43339), selective prefix-cache retention for sliding-window KV cache (#43447), and an index-share feature for DSA MTP (#44420). The model was also detached from `torch.compile` (#43746, #43891), its attention and RoPE paths were refactored (#44569, #44262, #43926), and an XPU attention decode path was added (#42953).\r\n* **Model Runner V2 expands to more dense models**: MRv2 is now selected by default for **Llama and Mistral dense models** (#43458) in addition to Qwen3. It gained a FlashInfer sampler (#42472), breakable CUDA graphs (#44050), pipeline-parallel bubble elimination (#42187), kernel block-size support for hybrid models (#38831), and Gemma 4 MTP (#43241).\r\n* **Rust frontend grows up**: The experimental Rust frontend added a streaming `generate` endpoint (#43779), dynamic LoRA endpoints (#43778), `/version` (#43854) and `/server_info` (#43942) endpoints, a server-router extension hook (#43774), request-ID headers (#43883), and many new tool parsers (InternLM2 #43481, hy_v3 #43872, Phi-4-mini #44213, Gemma4 #43850).\r\n* **Gemma 4**: Added encoder-free **Gemma 4 Unified** support (#44429) and Gemma 4 MTP (#43241), plus numerous accuracy and startup fixes.\r\n* **Transformers v5 compatibility**: vLLM now targets Transformers v5, with vendored MiniCPM-V/O processors (#44282) and compatibility fixes for Sarvam (#38804) and Voxtral (#44559).\r\n* **Multi-tier KV cache offloading**: The offloading framework gained an **object-store secondary tier** (#41968), HMA enabled by default for capable connectors (#41847), tiering support for HMA models (#44287), and a per-request offloading policy via the `on_new_request` lifecycle hook (#43205).\r\n* **Unified parser**: Reasoning and tool-call parsing are now unified behind a single `Parser.parse()` interface (#44267), with the Responses parser migrated to it (#42977).\r\n\r\n### Model Support\r\n* **New models**: Step-3.7-Flash (#43859), Cosmos3 Reasoner (#43356), Gemma 4 Unified encoder-free (#44429), JetBrains Mellum v2 (#43992), Granite Speech Plus (#43519), Cohere Mini Code (#44707).\r\n* **Gemma 4**: Encoder-free Unified support (#44429), MTP (#43241), native ViT linear layers (#43798), vision-embedder excluded from quantization (#44571), and fixes for MTP under TP>1 (#43909), block-table mismatch under concurrency (#43982), transformers-processor startup crash (#44232), and CPU init (#44615).\r\n* **Transformers v5**: Vendor MiniCPM-V/O processors (#44282), Sarvam compat (#38804), Voxtral `fetch_audio` for transformers≥5.10 (#44559).\r\n* **Model fixes & enhancements**: Qwen3-VL/Qwen3-omni-thinker deepstack accuracy under `torch.compile` (#43617), EVS for Qwen3-VL (#44205), GLM-5.1 PP loading (#42944), GLM-4.1V processor logits (#43575), GLM-4.6V video loader (#44417), OlmoHybrid init (#43846), HyperCLOVAX remote-code removal (#43860), Bailing-MoE rotary factor (#43770), Step3 PP residual KeyError (#37622), MiniCPM-V-4.6 video (#44509), MiniCPM-O audio unpadding (#38053), MiniCPM-V batched preprocessing (#44609), FunASR-Nano init (#44215), Cohere routing method (#44021), Kimi-K2.5 FlashInfer ViT metadata (#44493).\r\n* **Multimodal**: Auto-select registered video loader for VLMs (#44126), O(log n) multimodal item handling per step (#44212), local image encoding in benchmarks (#43843), interleaved custom image benchmark datasets (#43636).\r\n* **Pooling/Classification**: Proper exceptions for pooling UX (#44593), `extra_repr()` for pooler classes (#44805), LoRA-adapter-name pooling fix (#44410), resettled generative scoring entrypoint (#44153), expanded pooler unit tests (#43818, #44471).\r\n* **Refactor**: AutoWeightsLoader for InternLM2 (#38278).\r\n\r\n### Engine Core\r\n* **Model Runner V2**: Default for Llama and Mistral dense models (#43458), FlashInfer sampler (#42472), breakable CUDA graphs (#44050), removed Eagle's dedicated CUDA graph pool (#44078), pipeline-parallel bubble elimination (#42187), kernel block size for hybrid models (#38831), zeroing of freshly allocated KV blocks for hybrid + FP8 KV cache (#43990), actual batch `max_seq_len` for attention metadata (#43991), rejection-sampling acceptance-rate fix (#40651), KVConnector + PP cleanup (#43732), speculator-prefill warmup/capture (#44253).\r\n* **Speculative decoding (DFlash)**: Causal DFlash (#43445), proper lookahead-slot allocation (#43733), prefix-cache corruption fix (#42971); independent drafter attention-backend selection (#39930), attention-group split by `num_heads_q` for drafts (#43543), EAGLE/MTP lookahead caching in the SWA prefix-cache mask (#44082).\r\n* **Attention & hybrid/Mamba**: FlexAttention/FlashAttention num-blocks-first layouts (#42095), OOT MLA prefill backend registration (#43325), FlashAttention upstream sync (#44065), Mamba LINEAR attention-module refactor (#43556), corrupted MLA + linear attention fix (#43961), KDA conv-state unification (#44539) and gate/cumsum fusion (#43667), Mamba SSD `do_not_specialize` (#43803), Qwen3.5 mixed prefill+decode split routing (#44700), MiniMax-M2 gate kernel (#38445).\r\n* **KV cache & scheduler**: Pluggable `KVCacheSpec` (#37505), `scheduler_block_size` threaded into KVCacheManager/Coordinator (#44165), `max_concurrent_batches` moved to `VllmConfig` (#44274), config validation rejecting 0/negative knobs (#43794, #44057, #44207), KV-cache scale boilerplate removed from weight loading (#43167).\r\n* **Core**: Freeze the garbage collector in workers after model init (#44363), sparse NCCL weight transfer for in-place updates (#40096), graceful spinloop ext-load failure handling (#43659), scheduled-function deprecations (#43358).\r\n\r\n### Large Scale Serving & Distributed\r\n* **KV cache offloading**: Object-store secondary tier (#41968), HMA on by default for capable connectors (#41847) and tiering (#44287), per-request offloading policy (`on_new_request`) (#43205) and `on_schedule_end()` hook (#44206), token-offset selective offload (#39983), skip decode-phase blocks in CPU offload (#43797), page-size block alignment (#43689), Triton fast-path for small CPU→GPU `swap_blocks_batch` (#42212), stale sliding-window block fix (#42959).\r\n* **KV connectors / disaggregated serving**: PP-aware handshake aggregation and intermediate-PP output plumbing (#43720), multiple-async-KV-load deadlock fix (#44560), Nixl Mamba prefix-caching mode (#42554), NixlConnector `kv_both` role deprecation cycle (#43874), Mooncake fixes (#43742, #44103, #42694), LMCache `LMCacheMPConnector` (#42865), EC connector shutdown API (#42423) and non-blocking lookup (#41627), KV-transfer tokens excluded from `iteration_tokens_total` (#43346).\r\n* **EPLB**: Async EPLB by default (#43219), EPLB for DeepSeek-V4 Mega-MoE (#43339), Nixl zero-copy EPLB transfers (#41633).\r\n* **Data parallel**: DP Ray placement groups on specific nodes (#44669) and grouped-node allocation fix (#43998), SSL for the DP supervisor (#43688), DP-coordinator startup timeout raised to 120s (#42343), per-GPU-worker RDMA NIC selection (#42083).\r\n\r\n### Hardware & Performance\r\n* **NVIDIA / kernels**: FP8 FlashInfer attention for ViT (#38065), Triton MoE backend on Hopper by default (#44220), CUTLASS FP8 scaled-mm padding bypass (+20%) (#43706), MoE-permute buffer pre-allocation (+9–14%) (#43014), `Fp8BlockScaledMM` `new_empty()` optimization (#43677), TurboQuant shared dequant buffers (#40941), tuned `selective_state_update` for H200/RTX PRO (#44251), Inductor fast-path fallback for vLLM/AITER custom ops (#42129), Gemma RMS all-reduce fusion (#42646), NUMA auto-binding on DGX B300 (#43270).\r\n* **AMD ROCm**: ROCm 7.2.3 (#43136), AITER v0.1.13.post1 (#44265), native W4A16 (#41394) and fused-MoE W4A16 HIP (#44075) kernels for RDNA3 (gfx1100), AITER top-k/top-p sampler by default (#43331), attention-sink support in AITER FA (#43817), AITER hipBLASLt GEMM online tuning (#40426), `permute_cols` for ROCm (#44674), blocks-first KV layout for AMD (#43660), N=5 wvSplitK for spec decode (#40687), MoRI connector improvements (#43303, #41751, #40344).\r\n* **Intel XPU**: vllm-xpu-kernel v0.1.7 (#41019), `block_fp8_moe` (#42139), block-scaled W8A8 FP8 path (#39968), WNA16 oracle for GPTQ sym-int4 (#41426), rms_norm/act quant fusions (#43963), GDN-attention MTP (#43565), Triton selective-scan op (#43421), transparent sleep mode (#37149), CPU/tiering offloading on XPU (#36423), DeepSeek-V4 attention decode path (#42953).\r\n* **CPU & other architectures**: zentorch-accelerated W8A8/W4A16 on AMD Zen CPUs (#41813), CPU top-k/top-p Triton sampling (#43633), non-divisible GQA decode in mixed batches (#43032), `cpu_awq` folded into `awq_marlin` (#43841), RISC-V RVV WNA16 helpers (#42730), fused GDN gated-delta-rule kernels (#43534), PowerPC SHM communicator (#43754), arm64 CI image (#41303).\r\n* **TPU**: tpu-inference upgraded to v0.20.0 (#43394) then v0.21.0 (#44621).\r\n* **torch stable ABI**: Continued migration of kernels to the libtorch stable ABI — merge_attn_states/mamba/sampler [8/n] (#43361), attention/cache kernels [9/n] (#43717), header files (#44013), cuda_view/silu_and_mul [10/n] (#44334), custom all-reduce/DeepSeek-V4 fused MLA/MXFP8 MoE [10b/n] (#44365); ROCm fallback to regular ABI (#44648), `_has_module` trial-import verification (#44035).\r\n\r\n### Quantization\r\n* **ModelOpt**: LM-head quantization (#42124), MXFP8 non-gated MoE (#42958).\r\n* **compressed-tensors**: WNA8O8Int linears and WNInt embeddings (#44340), asymmetric MoE WNA16 Marlin (#44025), single-class NVFP4 linear refactor (#42443).\r\n* **Kernels & backends**: Triton W4A16 as CUDA fallback for non-Marlin-aligned shapes (#43731), Marlin MoE on SM 12.x (#40923), Machete W4A16 tests (#35450), fail-fast for unsupported NVFP4 KV-cache-dtype arch (#43669), CuteDSL compressor 128-split kernel optimization (#44230).\r\n* **MoE refactor (oracle)**: Migrated ModelOpt MXFP8 (#42768), W4A8-int8 (#42789), and WNA16 backend selection (#42553) into the modular-kernel oracle; removed `supports_expert_map` (#43108) and the inplace fused-experts mechanism (#43727).\r\n\r\n### API & Frontend\r\n* **Anthropic Messages API**: Structured output and effort support (#42396), system-role messages inside the messages array (#44283).\r\n* **OpenAI / Responses API**: `system_fingerprint` field (#40537), streaming tool/function calling with `required` (#40700), `chat_template_kwargs` in Responses (#43761), developer-to-system conversion in the HF renderer (#43590), unstreamed tool-call-args streaming fix (#44348).\r\n* **Parsers**: Unified reasoning + tool-call parsing behind `Parser.parse()` (#44267), Responses parser migrated to the unified interface (#42977), unstreamed tool-arg flush moved into the parser (#44017); new/fixed tool parsers — MiniCPM5 XML (#43175), Qwen3 XML JSON-args-first (#43243), DeepSeek DSML incremental streaming (#42879), first-args-chunk serializer fix (#42683), `tool_choice=\"none\"` honored in streaming (#42752), null-tool-args crash fix (#43862).\r\n* **Frontend**: `thinking_token_budget` validation (#43402), GPT-OSS instruction rendering (#44330), Harmony `stop_token_ids` cleanup (#44009), consistent `VLLMValidationError` in chat/completion validators (#36254), consolidation of dev entrypoints (#44170) and online-serving utils (#44479).\r\n* **Rust frontend**: Streaming `generate` endpoint (#43779), dynamic LoRA endpoints (#43778), `/version` (#43854) and `/server_info` (#43942), server-router extension hook (#43774), `--enable-request-id-headers` (#43883), recursive tool-parameter conversion (#44299), `include_reasoning=false` (#44391), `--language-model-only` skips the multimodal processor (#44500), per-engine batch auto-abort (#44591), UTF-8 char-boundary detokenizer fix (#44620), HF chat-template fixes (#44311), cross-DP aggregation of `is_sleeping`/`reset_prefix_cache` (#43429); new tool parsers — InternLM2 (#43481), hy_v3 (#43872), Phi-4-mini JSON (#44213), Gemma4 (#43850).\r\n* **Benchmarks**: Timed trace replay for Moonshot/Alibaba workloads in `vllm bench serve` (#39795), reasoning-model (thinking) benchmarking via `--chat-template-kwargs` (#44244).\r\n\r\n### Security\r\n* **Transport encryption**: SSL/TLS support for the data-parallel supervisor (#43688).\r\n* **Untrusted-input hardening**: Reject out-of-vocabulary token IDs before they reach the GPU logprob path (#44042) and fix a UTF-8 char-boundary panic in the Rust incremental detokenizer on malformed input (#44620), both of which prevent request-triggered crashes.\r\n* **Parameter validation**: Reject invalid `thinking_token_budget` values (#43402), non-positive `ParallelConfig` integer knobs (#44057), zero-valued config fields (#43794), and out-of-range `max_num_scheduled_tokens` (#44207).\r\n\r\n### Dependencies\r\n* FlashInfer v0.6.12 (#44036), ROCm 7.2.3 (#43136), AITER v0.1.13.post1 (#44265), tpu-inference v0.21.0 (#44621), mistral-common bump (#44649), fastsafetensors v0.3.2 (#43625).\r\n* Removed the stale cuDNN frontend upper bound (#42599); Docker fixes for flashinfer-jit-cache (#44366), FlashInfer CuTe DSL JIT `libcublas-dev` (#39855), and CUTLASS DSL cu13 install order (#45204).\r\n\r\n### Deprecations\r\n* Deprecate `JAISLMHeadModel` (#43784).\r\n* Begin the deprecation cycle for the NixlConnector `kv_both` role (#43874).\r\n* Remove functions previously scheduled for deprecation in v0.21.0 (#43358).\r\n\r\n## New Contributors\r\n\r\n* @aadwived made their first contribution in https://github.com/vllm-project/vllm/pull/41813\r\n* @adhithyamulticoreware made their first contribution in https://github.com/vllm-project/vllm/pull/44615\r\n* @adityasingh2400 made their first contribution in https://github.com/vllm-project/vllm/pull/43550\r\n* @adotdad made their first contribution in https://github.com/vllm-project/vllm/pull/43100\r\n* @amd-fuweiy made their first contribution in https://github.com/vllm-project/vllm/pull/43684\r\n* @andakai made their first contribution in https://github.com/vllm-project/vllm/pull/43617\r\n* @animeshtrivedi made their first contribution in https://github.com/vllm-project/vllm/pull/39795\r\n* @BramVanroy made their first contribution in https://github.com/vllm-project/vllm/pull/43087\r\n* @CienetStingLin made their first contribution in https://github.com/vllm-project/vllm/pull/43394\r\n* @devin-lai made their first contribution in https://github.com/vllm-project/vllm/pull/44213\r\n* @Dymasik made their first contribution in https://github.com/vllm-project/vllm/pull/43982\r\n* @ECMGit made their first contribution in https://github.com/vllm-project/vllm/pull/43332\r\n* @fallintoplace made their first contribution in https://github.com/vllm-project/vllm/pull/43540\r\n* @gagandhakrey made their first contribution in https://github.com/vllm-project/vllm/pull/43792\r\n* @galletas1712 made their first contribution in https://github.com/vllm-project/vllm/pull/43926\r\n* @garrygale made their first contribution in https://github.com/vllm-project/vllm/pull/44205\r\n* @Gruner-atero made their first contribution in https://github.com/vllm-project/vllm/pull/42967\r\n* @hanlin12-AMD made their first contribution in https://github.com/vllm-project/vllm/pull/40426\r\n* @harshaljanjani made their first contribution in https://github.com/vllm-project/vllm/pull/41459\r\n* @Holworth made their first contribution in https://github.com/vllm-project/vllm/pull/39562\r\n* @hoobnn made their first contribution in https://github.com/vllm-project/vllm/pull/42752\r\n* @HueCodes made their first contribution in https://github.com/vllm-project/vllm/pull/44591\r\n* @IdoAtadTD made their first contribution in https://github.com/vllm-project/vllm/pull/43978\r\n* @jasonboukheir made their first contribution in https://github.com/vllm-project/vllm/pull/41426\r\n* @Jie-Fang made their first contribution in https://github.com/vllm-project/vllm/pull/43584\r\n* @JINO-ROHIT made their first contribution in https://github.com/vllm-project/vllm/pull/43830\r\n* @JMonde made their first contribution in https://github.com/vllm-project/vllm/pull/37622\r\n* @JohnQinAMD made their first contribution in https://github.com/vllm-project/vllm/pull/43331\r\n* @jwzheng96 made their first contribution in https://github.com/vllm-project/vllm/pull/44057\r\n* @Kartavyasonar made their first contribution in https://github.com/vllm-project/vllm/pull/43669\r\n* @Krishnachaitanyakc made their first contribution in https://github.com/vllm-project/vllm/pull/38053\r\n* @linzm1007 made their first contribution in https://github.com/vllm-project/vllm/pull/43402\r\n* @MaciejBalaNV made their first contribution in https://github.com/vllm-project/vllm/pull/43356\r\n* @Majid-Taheri made their first contribution in https://github.com/vllm-project/vllm/pull/43803\r\n* @mfylcek made their first contribution in https://github.com/vllm-project/vllm/pull/43421\r\n* @MHYangAMD made their first contribution in https://github.com/vllm-project/vllm/pull/42595\r\n* @mikekg made their first contribution in https://github.com/vllm-project/vllm/pull/43330\r\n* @nightcityblade made their first contribution in https://github.com/vllm-project/vllm/pull/44118\r\n* @NolanHo made their first contribution in https://github.com/vllm-project/vllm/pull/43774\r\n* @oguzhankir made their first contribution in https://github.com/vllm-project/vllm/pull/41759\r\n* @okorzh-amd made their first contribution in https://github.com/vllm-project/vllm/pull/42129\r\n* @Oxygen56 made their first contribution in https://github.com/vllm-project/vllm/pull/44236\r\n* @QiliangCui2023 made their first contribution in https://github.com/vllm-project/vllm/pull/44476\r\n* @rajkiranjoshi made their first contribution in https://github.com/vllm-project/vllm/pull/42083\r\n* @Rukhaiya2004 made their first contribution in https://github.com/vllm-project/vllm/pull/43754\r\n* @ruocco made their first contribution in https://github.com/vllm-project/vllm/pull/39983\r\n* @sphinx07 made their first contribution in https://github.com/vllm-project/vllm/pull/43817\r\n* @SunskyXH made their first contribution in https://github.com/vllm-project/vllm/pull/44215\r\n* @ThibaultCastells made their first contribution in https://github.com/vllm-project/vllm/pull/43636\r\n* @tianyu-z made their first contribution in https://github.com/vllm-project/vllm/pull/43150\r\n* @tonyliu312 made their first contribution in https://github.com/vllm-project/vllm/pull/40923\r\n* @tushar00jain made their first contribution in https://github.com/vllm-project/vllm/pull/41980\r\n* @viiccwen made their first contribution in https://github.com/vllm-project/vllm/pull/44617\r\n* @Vikrantpalle made their first contribution in https://github.com/vllm-project/vllm/pull/38804\r\n* @wanghenshui made their first contribution in https://github.com/vllm-project/vllm/pull/44410\r\n* @willamhou made their first contribution in https://github.com/vllm-project/vllm/pull/43429\r\n* @william-rom made their first contribution in https://github.com/vllm-project/vllm/pull/43862\r\n* @xiaozcy made their first contribution in https://github.com/vllm-project/vllm/pull/43843\r\n* @XuZhou26 made their first contribution in https://github.com/vllm-project/vllm/pull/44618\r\n* @Yadan-Wei made their first contribution in https://github.com/vllm-project/vllm/pull/44559\r\n* @zhangtao2-1 made their first contribution in https://github.com/vllm-project/vllm/pull/43175\r\n* @zvik made their first contribution in https://github.com/vllm-project/vllm/pull/43519\r\n* @zzt93 made their first contribution in https://github.com/vllm-project/vllm/pull/43770\r\n\r\n## Contributors\r\n\r\nThank you to everyone who made this release possible!\r\n\r\n@AndreasKaratzas, @WoosukKwon, @BugenZhao, @yewentao256, @hmellor, @khluu, @njhill, @sfeng33, @bnellnm, @vadiklyutiy, @NickLucche, @JartX, @lucianommartins, @cleonard530, @wzhao18, @yma11, @simondanielsson, @jeejeelee, @zyongye, @chaunceyjiang, @bigPYJ1151, @ronensc, @taneem-ibrahim, @LucasWilkinson, @MatthewBonanni, @mmangkad, @chunyang-wen, @yzong-rh, @JaredforReal, @zixi-qi, @Isotr0py, @noooop, @chaojun-zhang, @Xunzhuo, @ivanium, @zufangzhu, @DaoyuanLi2816, @CienetStingLin, @aoshen02, @akii96, @benchislett, @MengqingCao, @rshavitt, @kliuae, @omerpaz95, @willamhou, @Majid-Taheri, @micah-wil, @ricky-chaoju, @mikekg, @mgoin, @mayuyuace, @Etelis, @ilmarkov, @tlrmchlsmth, @UranusSeven, @bedeks, @izhuhaoran, @ZJY0516, @fadara01, @pschlan-amd, @wangxiyuan, @Oxygen56, @charlifu, @varun-sundar-rabindranath, @shen-shanshan, @TheEpicDolphin, @adobrzyn, @XuZhou26, @tjtanaa, @Terrencezzj, @zhejiangxiaomai, @ILikeIneine, @yubofredwang, @chfeng-cs, @ThibaultCastells, @linzm1007, @javierdejesusda, @meenchen, @zhewenl, @xyang16, @angelayi, @nholmber, @zhangtao2-1, @adityasingh2400, @sts07142, @jatseng-ai, @fallintoplace, @andakai, @he-yufeng, @ignaciosica, @JINO-ROHIT, @tonyliu312, @QwertyJack, @animeshtrivedi, @jzakrzew, @juliendenize, @zexplorerhj, @ruocco, @mgehre-amd, @jasonboukheir, @MaciejBalaNV, @JohnQinAMD, @huanghua1994, @rajkiranjoshi, @rasmith, @harshaljanjani, @ltd0924, @wdhongtw, @yintong-lu, @tianmu-li, @jikunshang, @JMonde, @MHYangAMD, @frida-andersson, @gau-nernst, @Wauplin, @czhu-cohere, @gagandhakrey, @nemanjaudovic, @Liangliang-Ma, @liulanze, @sphinx07, @aadwived, @nightcityblade, @umut-polat, @jeffreywang88, @wcynb1023, @zzt93, @shadeMe, @Dao007forever, @alec-flowers, @Krishnachaitanyakc, @orozery, @BWAAEEEK, @cinnamonica02, @albertoperdomo2, @Rukhaiya2004, @mfylcek, @shreyas269, @Gruner-atero, @TomerBN-Nvidia, @wjinxu, @IdoAtadTD, @xiaozcy, @brian-dellabetta, @zhenwei-intel, @adotdad, @Kartavyasonar, @lesj0610, @ECMGit, @cakeng, @william-rom, @qiching, @NolanHo, @andylolu2, @xwu-intel, @linitra24, @hoobnn, @Dymasik, @wanghenshui, @maobaolong, @oguzhankir, @Jie-Fang, @okorzh-amd, @Kevin-XiongC, @jiahanc, @garrygale, @dsikka, @QiliangCui2023, @wjabbour, @zvik, @tc-mb, @jwzheng96, @divakar-amd, @tushar00jain, @galletas1712, @hanlin12-AMD, @tuukkjs, @viiccwen, @Sunt-ing, @HueCodes, @tianyu-z, @adhithyamulticoreware, @rishitdholakia13, @effi-ofer, @Vikrantpalle, @walterbm, @devin-lai, @Yadan-Wei, @amd-fuweiy, @maeehart, @qyYue1389, @BramVanroy, @SunskyXH, @Holworth, @majian4work, @xaguilar-amd, @Rohan138\r\n","reactions":{"url":"https://api.github.com/repos/vllm-project/vllm/releases/338844829/reactions","total_count":58,"+1":27,"-1":0,"laugh":0,"hooray":16,"confused":0,"heart":4,"rocket":6,"eyes":5},"mentions_count":199},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/334860934","assets_url":"https://api.github.com/repos/vllm-project/vllm/releases/334860934/assets","upload_url":"https://uploads.github.com/repos/vllm-project/vllm/releases/334860934/assets{?name,label}","html_url":"https://github.com/vllm-project/vllm/releases/tag/v0.22.1","id":334860934,"author":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"node_id":"RE_kwDOI7xefs4T9ZKG","tag_name":"v0.22.1","target_commitish":"main","name":"v0.22.1","draft":false,"immutable":false,"prerelease":false,"created_at":"2026-06-04T00:11:47Z","updated_at":"2026-06-05T10:18:46Z","published_at":"2026-06-05T10:10:00Z","assets":[{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/439189147","id":439189147,"node_id":"RA_kwDOI7xefs4aLX6b","name":"vllm-0.22.1+cpu-cp38-abi3-manylinux_2_34_aarch64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":49294101,"digest":"sha256:15c52a0e41babdce37201f279782ee183f09f44ba7c5dc2f202d40e898b9badc","download_count":114,"created_at":"2026-06-05T10:11:06Z","updated_at":"2026-06-05T10:11:11Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.22.1/vllm-0.22.1%2Bcpu-cp38-abi3-manylinux_2_34_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/439189142","id":439189142,"node_id":"RA_kwDOI7xefs4aLX6W","name":"vllm-0.22.1+cpu-cp38-abi3-manylinux_2_34_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":108468761,"digest":"sha256:c8661ca28e8e0a65fdf7f5462877150c9157be85507aa4b2aacfe6a8d25b1827","download_count":1138,"created_at":"2026-06-05T10:11:06Z","updated_at":"2026-06-05T10:11:15Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.22.1/vllm-0.22.1%2Bcpu-cp38-abi3-manylinux_2_34_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/439189143","id":439189143,"node_id":"RA_kwDOI7xefs4aLX6X","name":"vllm-0.22.1+cu129-cp38-abi3-manylinux_2_28_aarch64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":440436533,"digest":"sha256:b4cef4bf6264372d61382ebeb36e2be7183e3e736769f1d53fa4c897a0be8ce7","download_count":163,"created_at":"2026-06-05T10:11:06Z","updated_at":"2026-06-05T10:11:36Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.22.1/vllm-0.22.1%2Bcu129-cp38-abi3-manylinux_2_28_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/439189140","id":439189140,"node_id":"RA_kwDOI7xefs4aLX6U","name":"vllm-0.22.1+cu129-cp38-abi3-manylinux_2_28_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":473003163,"digest":"sha256:365ee929afd73bb5d146235b65053fa948788ec2ee00a2c3e957d3f43bf2b0cd","download_count":32432,"created_at":"2026-06-05T10:11:06Z","updated_at":"2026-06-05T10:11:40Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.22.1/vllm-0.22.1%2Bcu129-cp38-abi3-manylinux_2_28_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/439189145","id":439189145,"node_id":"RA_kwDOI7xefs4aLX6Z","name":"vllm-0.22.1-cp38-abi3-manylinux_2_28_aarch64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":252950342,"digest":"sha256:2d8e51693683e4d7c7e9ed311a2996200428fa3a28e154ddbff809aeb7c5382a","download_count":2725,"created_at":"2026-06-05T10:11:06Z","updated_at":"2026-06-05T10:11:28Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.22.1/vllm-0.22.1-cp38-abi3-manylinux_2_28_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/439189199","id":439189199,"node_id":"RA_kwDOI7xefs4aLX7P","name":"vllm-0.22.1-cp38-abi3-manylinux_2_28_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":261042548,"digest":"sha256:cc50536c93827a577b2bc45f038b8b866c31ed220631008521a337c9b2558f4d","download_count":3161,"created_at":"2026-06-05T10:11:12Z","updated_at":"2026-06-05T10:11:27Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.22.1/vllm-0.22.1-cp38-abi3-manylinux_2_28_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/439189284","id":439189284,"node_id":"RA_kwDOI7xefs4aLX8k","name":"vllm-0.22.1.tar.gz","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/x-gtar","state":"uploaded","size":36246058,"digest":"sha256:cd34902729f3767e3dfbb391956b2986bc638a54be9893e78bda9cc55669f15c","download_count":2001,"created_at":"2026-06-05T10:11:16Z","updated_at":"2026-06-05T10:11:18Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.22.1/vllm-0.22.1.tar.gz"}],"tarball_url":"https://api.github.com/repos/vllm-project/vllm/tarball/v0.22.1","zipball_url":"https://api.github.com/repos/vllm-project/vllm/zipball/v0.22.1","body":"## Highlights\r\n\r\nThis release features 8 commits from 6 contributors (1 new)!\r\n\r\nv0.22.1 is a patch release on top of v0.22.0 with targeted bug fixes plus a couple of additions: new model support for JetBrains' Mellum v2, zentorch-accelerated quantized linear inference on AMD Zen CPUs, and fixes for multi-node Ray data-parallel serving, DeepSeek-V4 initialization, and a few model-loading regressions.\r\n\r\n### Model Support\r\n* New model: JetBrains' **Mellum v2**, an open-weights Mixture-of-Experts code-generation model (#43992).\r\n* **DeepSeek-V4**: resolve a CUTLASS `fmin` compatibility issue that broke initialization (0decac0d).\r\n* Fix `OlmoHybridForCausalLM` failing to initialise after the checkpoint changed `rope_parameters` from `None` to `{\"rope_type\": None}` (#43846).\r\n* Fix **HyperCLOVAX** loading after the upstream HuggingFace repo removed its remote code (now native in `transformers >= 5.9.0`): register the `hyperclovax` model_type so vLLM uses its vendored config instead of the stale `auto_map` (#43860).\r\n\r\n### Hardware & Performance\r\n* **AMD Zen CPUs**: route W8A8 (int8 dynamic-symmetric) and W4A16 (GPTQ) linear inference through zentorch kernels, registered ahead of the generic oneDNN CPU kernels, with transparent fallback on non-Zen CPUs, GPUs, and XPU (#41813).\r\n\r\n### Large Scale Serving\r\n* Fix a deterministic hang in multi-node **Ray data-parallel** serving with `num_api_servers > 1` by excluding the Ray DP backend from the deferred (kernel-assigned) port allocation introduced in #42585 (#43864).\r\n\r\n### Build & CI\r\n* Docker: stop installing `flashinfer-jit-cache` via `--extra-index-url` while it is quarantined on PyPI, fixing image builds (#44366).\r\n* Normalize **NIXL** KV-connector wheel installs so only the wheel matching the image's CUDA major is kept, fixing `ImportError: libcudart.so.12` when importing `nixl_ep` on CUDA 13 images (#44266).\r\n\r\n## Contributors\r\n\r\n@khluu, @vadiklyutiy, @aadwived, @shadeMe, @alec-flowers, @hmellor\r\n\r\n## New Contributors\r\n\r\n* @aadwived made their first contribution in https://github.com/vllm-project/vllm/pull/41813","reactions":{"url":"https://api.github.com/repos/vllm-project/vllm/releases/334860934/reactions","total_count":17,"+1":9,"-1":0,"laugh":0,"hooray":0,"confused":0,"heart":2,"rocket":6,"eyes":0},"mentions_count":6},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/331396286","assets_url":"https://api.github.com/repos/vllm-project/vllm/releases/331396286/assets","upload_url":"https://uploads.github.com/repos/vllm-project/vllm/releases/331396286/assets{?name,label}","html_url":"https://github.com/vllm-project/vllm/releases/tag/v0.22.0","id":331396286,"author":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"node_id":"RE_kwDOI7xefs4TwLS-","tag_name":"v0.22.0","target_commitish":"main","name":"v0.22.0","draft":false,"immutable":false,"prerelease":false,"created_at":"2026-05-29T09:28:43Z","updated_at":"2026-06-02T18:14:39Z","published_at":"2026-05-29T10:28:13Z","assets":[{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/432937980","id":432937980,"node_id":"RA_kwDOI7xefs4Zzhv8","name":"vllm-0.22.0+cpu-cp38-abi3-manylinux_2_34_aarch64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":49286540,"digest":"sha256:5bbf549ce608fa4385e632b0f6fe7409321a8f811e414355494768ce4546d79f","download_count":98,"created_at":"2026-05-29T10:30:40Z","updated_at":"2026-05-29T10:30:45Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.22.0/vllm-0.22.0%2Bcpu-cp38-abi3-manylinux_2_34_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/432937982","id":432937982,"node_id":"RA_kwDOI7xefs4Zzhv-","name":"vllm-0.22.0+cpu-cp38-abi3-manylinux_2_34_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":108460954,"digest":"sha256:09f4434430e067f5b231b0d4be1db57df1cd9d2ca3335275cb5ecaf996f230a8","download_count":1984,"created_at":"2026-05-29T10:30:40Z","updated_at":"2026-05-29T10:30:48Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.22.0/vllm-0.22.0%2Bcpu-cp38-abi3-manylinux_2_34_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/432937981","id":432937981,"node_id":"RA_kwDOI7xefs4Zzhv9","name":"vllm-0.22.0+cu129-cp38-abi3-manylinux_2_28_aarch64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":440428989,"digest":"sha256:14f586820f3ed2cecdbda39651ecfe1cd097102200e25bbff5e740ccecb6143d","download_count":95137,"created_at":"2026-05-29T10:30:40Z","updated_at":"2026-05-29T10:31:10Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.22.0/vllm-0.22.0%2Bcu129-cp38-abi3-manylinux_2_28_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/432937979","id":432937979,"node_id":"RA_kwDOI7xefs4Zzhv7","name":"vllm-0.22.0+cu129-cp38-abi3-manylinux_2_28_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":472995605,"digest":"sha256:405319a7fde1af416eef6cbab0315d354d8b9c1a45b97097a9f1c3707bf05f74","download_count":154361,"created_at":"2026-05-29T10:30:40Z","updated_at":"2026-05-29T10:31:12Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.22.0/vllm-0.22.0%2Bcu129-cp38-abi3-manylinux_2_28_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/432937983","id":432937983,"node_id":"RA_kwDOI7xefs4Zzhv_","name":"vllm-0.22.0-cp38-abi3-manylinux_2_28_aarch64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":252942448,"digest":"sha256:0fbe1ff32e9ad82c56b002de11b061ca6b5b8a256cd11473946d2222115ed267","download_count":446,"created_at":"2026-05-29T10:30:40Z","updated_at":"2026-05-29T10:30:58Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.22.0/vllm-0.22.0-cp38-abi3-manylinux_2_28_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/432938013","id":432938013,"node_id":"RA_kwDOI7xefs4Zzhwd","name":"vllm-0.22.0-cp38-abi3-manylinux_2_28_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":261034920,"digest":"sha256:c387a977e35795e8f77b009e019e69722963819c26b55e4a679e09d4279ae35d","download_count":436,"created_at":"2026-05-29T10:30:46Z","updated_at":"2026-05-29T10:31:00Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.22.0/vllm-0.22.0-cp38-abi3-manylinux_2_28_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/432941182","id":432941182,"node_id":"RA_kwDOI7xefs4Zzih-","name":"vllm-0.22.0.tar.gz","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/x-gtar","state":"uploaded","size":36239170,"digest":"sha256:6d41581a9e5288cd69278518a550c6d7ce510ae27a506556a3427d01284be7fe","download_count":1957,"created_at":"2026-05-29T10:35:58Z","updated_at":"2026-05-29T10:36:00Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.22.0/vllm-0.22.0.tar.gz"}],"tarball_url":"https://api.github.com/repos/vllm-project/vllm/tarball/v0.22.0","zipball_url":"https://api.github.com/repos/vllm-project/vllm/zipball/v0.22.0","body":"## Highlights\r\n\r\nThis release features 459 commits from 230 contributors (63 new)!\r\n\r\n* **DeepSeek V4 maturity**: DeepSeek V4 received a major hardening pass this cycle — the model was reorganized into a dedicated `vllm/models/deepseek_v4/` package (#43004, #43039, #43073, #43077, #43149), gained NVFP4 fused MoE support (#42209), full + piecewise CUDA graph (#42604), and MTP speculative decoding (#43385). A large set of fused kernels (MegaMoE, `mhc`, Q-norm, indexer, sparse MLA) and ROCm parity fixes landed alongside accuracy fixes (#42810, #43710).\r\n* **Model Runner V2 advances toward default**: MRv2 is now default for Qwen3 dense models. vLLM will fall back to MRv1 for features that aren't yet supported in MRv2 (#39337). sleep-mode weight reload (#42673), `update_config` (#42783), and shared KV-cache layers (#35045), plus many correctness fixes.\r\n* **Experimental Rust frontend**: A new Rust front-end integration landed (#40848), with the implementation moved into the tree (#43283) and a DP Supervisor for data-parallel serving (#40841).\r\n* **Batch invariance, faster**: Batch-invariant inference gained Cutlass FP8 support for a **28.9% end-to-end latency improvement** (#40408), compile-mode support on SM80 (#42456), and an NVFP4 Cutlass linear path (#39912).\r\n* **Multi-tier KV cache offloading**: A new multi-tier KV cache offloading framework (#40020) with a Python filesystem secondary tier (#41735), DSv4 support (#43142), and Mooncake disk offloading (#42689) extends offloading beyond CPU memory.\r\n\r\n### Model Support\r\n* New architectures: MiniCPM-V 4.6 (#41254), InternS2 Preview (#42705), OpenVLA (#42654), MolmoWeb `hf_overrides` docs (#42163); EXAONE-4.5 aligned with Transformers update (#42246).\r\n* Speculative decoding: custom callable proposer backend (#39487), post-norm EAGLE-3 speculators (#42764), peagle speculators (#41826), hybrid-attention models in `extract_hidden_states` (#39949), non-MTP speculation for NemotronH (#43130), shared MTP weights in MRv2 (#42538).\r\n* DeepSeek V4: NVFP4 MoE (#42209), CUDA graph full/piecewise (#42604), MTP (#43385), model package refactor (#43004, #43039, #43073, #43077), sparse MLA + compressor refactor (#43149, #43710), MegaMoE input-prep kernel move (#43632).\r\n* Qwen3.5/3.6: GDN output-projection flatten (#42311), GatedDeltaNet Marlin TP≥2 fix (#36329), ViT full CUDA graph (#42151), runai-streamer weight loading for Qwen3.5/MTP/Qwen3-VL (#42521, #42716), KDA chunk-prefill exp2 semantics (#43195).\r\n* Gemma3/Gemma4: mixed-resolution image co-batching crash fix (#42217), MoE routing closure fix (#42250), tool-parser float-corruption fix (#42128), batched vision encoder for image/video (#43169), multi-GPU fix (#42630).\r\n* Kimi-K2.5: skip vision-tower dtype conversion under quantization (#42869), `mm_projector` dtype fix (#42081).\r\n* Cohere: enable Cohere MoE (#43143), pipeline parallelism for Cohere vision (#42819).\r\n* Tool calling: Apertus tool parser (#41154), Qwen3Coder `anyOf`/`oneOf`/`$ref` resolution re-land (#37831), shared `coerce_to_schema_type` across MiniMax-M2 / DeepSeek-V3.2 / Seed-OSS parsers (#43006, #43019, #43140).\r\n* ViT CUDA graph: Qwen2-VL (#41736), Step3-VL encoder (#42224), Qwen3.5 (#42151), FlashInfer metadata for Qwen2.5-VL vision attention (#42787).\r\n\r\n### Engine Core\r\n* Model Runner V2: Qwen3-dense-by-default oracle (#39337), sleep-mode reload weights (#42673), `update_config` (#42783), shared KV-cache layers (#35045), FP32 gumbel sampling (#41775), auto-fallback to MRv1 with connectors (#42955), `logprob_token_ids` correctness (#43125, #41761), prompt-logprobs size fix (#42778).\r\n* KV offloading: multi-tier framework (#40020), Python filesystem secondary tier (#41735), DSv4 support (#43142), tier-offload follow-up (#42529), prefer HND layout (#41928), `reset_cache()` (#41956), per-request tracking (#42507), store-deferral fix (#41945).\r\n* MoE refactor: `ExpertMapManager` (#41046), experts moved to `experts/` (#42334), `RoutedExperts` alias for FusedMoE (#40735), EPLB refactoring for FusedMoE (#41055).\r\n* Mamba: attention module refactor (#41126), Mamba2 SSD kernel warmup (#39822), bf16 SSM cache (#41680), GPU-side state postprocessing fused kernel (#40172), run single-token extends as decodes (#42430).\r\n* KV events: emit KV cache metadata (#40984).\r\n* Allocator: manual cumem allocator enable (#33648), stream-aware free callback (#43020).\r\n* elastic-EP: stage/commit MoE quant method on reconfigure (#40881).\r\n\r\n### Hardware & Performance\r\n* **NVIDIA Blackwell / SM12x**: FlashInfer b12x MoE + FP4 GEMM for SM120/121 (#40082), per-tensor FP8 CUTLASS on SM12.1 (#41215), `head_dim=512` for FlashInfer TRTLLM attention (#38822), FlashInfer Blackwell GDN prefill (#40717), GDN prefill kernel for SM100 (#43273).\r\n* **Performance**: batch-invariant Cutlass FP8 (+28.9% E2E) (#40408), CutlassFP8 padding pre-processing (+13.5% TTFT) (#42651), padded NVFP4 quant kernel (+2.4–5.7% E2E) (#42774), GPU<->CPU sync elimination 1/n (#41429) and 4/n (#42347), fused RoPE+KVCache+q_concat for MLA (#40392), MLA `compute_prefill_context` / `_v_up_proj` optimizations (#42460, #42561), penalties Triton kernel (#40657), `do_not_specialize` in fused FP8 RoPE (#42849), FULL CUDA graph capture for TRITON_MLA decode (#42885).\r\n* **AMD ROCm**: DSV4 functionality + accuracy fixes (#42810, #43679 Tilelang MHC), flash sparse MLA Triton kernels (#41812), gluon paged MQA logits on gfx950/MI355X (#42062), RMSNorm+Quant fusion for gfx950 (#41825), AITER FA backend cleanup (#41942), XGMI backend for MoRI connector (#41753), QuickReduce min-size override (#41675), DSV4 MTP (#43385).\r\n* **CPU / RISC-V**: RVV-optimized attention kernels for RISC-V Vector Extension (#40119) with VLEN=256 (#42943), fused GDN for AMX CPU (#42707), MXFP4 W4A16 MoE (#41922), experimental Triton + MRv2 on CPU (#43225), improved CPU thread utilization (#42666), `--cpu-distributed-timeout-seconds` (#42968).\r\n* **Intel XPU**: GPTQ int4 support (#37844), mxfp8 MoE (#41918), FP8 block-scaled quantization (#42952), custom-op collective behavior (#41354), multiple sparse-attention kernels (#37888), MoE topk routing + MXFP4 fallback (#42951), CT W4A4 MXFP4 path (#38896), reduced XPU MoE host overhead (#42915).\r\n* **Kernel ABI**: continued migration to libtorch stable ABI — 5/n (#42339), 6/n (#42663), 7/n (#43209).\r\n* **Experimental**: breakable CUDA graph (#42304).\r\n\r\n### Large Scale Serving\r\n* Disaggregated serving (NIXL): lease-renewal TTL for KV blocks on P (#41383), handshake-failure policy honoring (#40364), GDN support for PD with NIXL (#41869), multi-node TP>8 fix (#39907), side-channel host-selection fix (#41806).\r\n* Mooncake: disk offloading in MooncakeStoreConnector (#42689), HMA support for DSV4 (#42828), operation metrics (#43392), load-failure propagation (#42788), block-aligned full hits (#43494), finish-after-preemption handling (#43281).\r\n* Data parallel: DP Supervisor (#40841), publish request counts at engine-step start (#41626), forward `X-data-parallel-rank` header (#42330).\r\n* EPLB: change default EPLB communicator (#43110), VLM-wrapper init fix (#39805), remove dead `torch.accelerator.synchronize()` (#40733).\r\n* LoRA: one-shot Triton kernel for MoE LoRA (#42290), simultaneous 2D & 3D MoE LoRA adapters (#42242), reduced 2D-weight memory under EP (#42737), MoE LoRA align-kernel grid fix (#40131).\r\n\r\n### Quantization\r\n* **MXFP4**: linear layers + compressed-tensors integration (#41664), CPU W4A16 MoE (#41922), XPU mxfp8 MoE (#41918).\r\n* **NVFP4**: DeepSeek V4 fused MoE (#42209), ModelOpt W4A16 NVFP4 fused MoE + mixed-precision dispatch (#42566), batch-invariant NVFP4 Cutlass linear (#39912), FlashInfer TRTLLM NvFP4 monolithic MoE routing fix (#43223), TRTLLM NVFP4 MoE chunking fix (#43599).\r\n* **Quark**: load Quark NVFP4 checkpoints (#35859), W8A8 INT8 garbage-output fix on Step-3.5-Flash (#41892), W4A4 oracle refactor (#41436).\r\n* **AutoRound**: W4A16 support (#39778).\r\n* **ModelOpt**: Qwen3.5/3.6 VLM quantized prefix mapping (#42546).\r\n* **Framework**: rework `quantization_config` to use `QuantKey` with activation override (#41566), MoE W4A8 CT migrated to oracle (#42680), AWQ Marlin MoE onto modular WNA16 oracle (#42483), GPTQ consolidation (`gptq_marlin` → `auto_gptq`) (#38288).\r\n\r\n### API & Frontend\r\n* **Rust frontend**: integration (#40848), in-tree code move (#43283), utility call-ID newtype (#43405), simplified `AuthenticationMiddleware` path extraction (#43426).\r\n* **Responses API**: `chat_template_kwargs` support (#42272), message-merging fix (#42189), empty channel/recipient harmony fix (#35540).\r\n* **Completions**: `thinking_token_budget` support (#42116) with inverted-condition fix (#41674); map `reasoning_effort` to `enable_thinking` (#43401).\r\n* **Frontend**: truncation side for OpenAI endpoints (#43260), normalize `reasoning_content` → `reasoning` (#42664), reworked fastokens integration (#43168), consolidated Speech-to-Text entrypoints (#42370, #42274), beam-search consolidation via `BeamSearchMixin` (#42946), score/rerank chat-template instructions (#42412).\r\n* **Auth**: API-key authorization for `/v2` endpoints (#42594).\r\n* **Offline API**: pooling offline API split into `PoolingOfflineMixin` (#42267), split offline inference APIs/utils (#43553).\r\n\r\n### Build & Dependencies\r\n* CUDA 12.9 wheel builds switched to PyTorch `manylinux_2_28` base (#41668).\r\n* FlashInfer bumped to v0.6.11.post2 (#41711); `nvidia-cutlass-dsl` to 4.5.2 (#42991, #43230, #43745); llguidance to 1.7 (#42150); `triton_kernels` downgraded to v3.5.1 for gpt-oss (#43135).\r\n* Rust frontend build: `setuptools-rust` dependency (#43287, #43377), pinned `protoc` in rust-build stages (#43292).\r\n* Docker: non-root `vllm-openai` target (#40275), build `mooncake-transfer-engine` from source (#42114), AINIC & Thor NIC support (#40453); Python-only installation made optional (#42293).\r\n* vllm-tpu: disable build isolation for CUDA deps (#43038), tpu-inference docker build fix (#43360).\r\n* `humming` MoE backend dependency added, reverted, then restored with CuPy runtime fix (#42540, #43492, #43530).\r\n\r\n### Deprecations & Removals\r\n* Removed old locations of `get_tokenizer` and `resolve_hf_chat_template` (#35024).\r\n* Marked env vars now covered by `--moe-backend` / `--linear-backend` (#43148).\r\n* Removed deprecated MLA prefill arguments (#42555).\r\n* Removed dead CUDA kernels and dead code (#42767, #42889, #43144).\r\n\r\n## Contributors\r\n\r\n@yewentao256, @haosdent, @njhill, @mgoin, @jeejeelee, @AndreasKaratzas, @NickLucche, @sfeng33, @noooop, @WoosukKwon, @khluu, @taneem-ibrahim, @Dao007forever, @vadiklyutiy, @bnellnm, @ivanium, @tjtanaa, @mmangkad, @hmellor, @DarkLight1337, @hickeyma, @zhenwei-intel, @jikunshang, @ronensc, @benchislett, @hao-aaron, @arpera, @zyongye, @gau-nernst, @frida-andersson, @ZhanqiuHu, @cleonard530, @akii96, @bedeks, @Isotr0py, @JasonKeyiL, @bigPYJ1151, @zhewenl, @weizhoublue, @zxd1997066, @gnovack, @chaojun-zhang, @majian4work, @chaunceyjiang, @pschlan-amd, @amitz-nv, @yma11, @dsikka, @tc-mb, @shanjiaz, @jperezdealgaba, @yzong-rh, @viktorpusTT, @TheEpicDolphin, @MatthewBonanni, @shen-shanshan, @hallerite, @zufangzhu, @bbrowning, @divakar-amd, @ianliuy, @esmeetu, @rasmith, @louie-tsai, @pmaybank, @liulanze, @ZJY0516, @TheDuyIT, @wzhao18, @jinzhen-lin, @BugenZhao, @ashwing, @fuergaosi233, @hqhq1025, @shaharmor98, @pisceskkk, @lkm2835, @noa-neria, @Rohan138, @whx-sjtu, @vrdn-23, @alexagriffith, @Flink-ddd, @jeffreywang-anyscale, @skyloevil, @ymoslem, @Lucaskabela, @kg6-sleipnir, @woernfl, @tdoublep, @GOavi101, @jmamou, @PeaBrane, @KaivalyaMDabhadkar, @BWAAEEEK, @MrZ20, @afierka-intel, @JoursBleu, @hissu-hyvarinen, @mwawrzos, @CynicDora, @NoeliaBentancor, @johncalesp, @fynnsu, @fxmarty-amd, @walterbm, @liangel-02, @lgeiger, @he-yufeng, @abinggo, @KrxGu, @hks-9697-v2, @Sarah-Salah, @rebklee, @aoshen02, @haic0, @libinta, @Zhenzhong1, @xhx1022, @b-mu, @WindChimeRan, @tpopp, @charlifu, @chengyinie, @ricky-chaoju, @lyd1992, @daniel-devlab, @paulyu12, @bobofang11235, @laudney, @BadrBasowid, @maeehart, @PatchouliTIS, @chunxiaozheng, @blake-snc, @southfreebird, @rbrugaro-amd, @rasdani, @dusthunter, @qizzzh, @ProExpertProg, @qianlihuang, @alec-flowers, @JisoLya, @gaozihao-shy, @rishaps, @xyang16, @wendyliu235, @hlin99, @tianmu-li, @yuwenzho, @inisis, @kfirtoledo, @roikoren755, @liranschour, @vllm-agent, @blancsw, @netanel-haber, @BowenBao, @czhu-cohere, @amitport, @tuukkjs, @revit13, @ofirzaf, @qyYue1389, @junyanxu, @gracie-guo, @sagearc, @xinyu-intel, @yiwen101, @DomBrown, @tomeras91, @Dogacel, @maxdebayser, @fadara01, @Terrencezzj, @izikgo, @wangrui6, @kebe7jun, @rishitdholakia13, @j9smith, @meena-at-work, @dllehr-amd, @alexeldeib, @sonusflow, @lucianommartins, @AAISSJ, @DaoyuanLi2816, @zexplorerhj, @zhangxin81, @velonica0, @fuscof-ibm, @anishesg, @zhengluo-nv, @ylangtsou, @fangyuchu, @zx3xyy, @simondanielsson, @ruizhang99, @zixi-qi, @xwu-intel, @yufufi, @wdhongtw, @mrjunwan-lang, @wangxiyuan, @wasnertobias, @ilmarkov, @sychen52, @zhandaz, @russellb, @SandishKumarHN, @juhi10071998, @itayalroy, @djmmoss, @SumanthRH, @mayuyuace, @zhougit86, @meenchen, @lucifer1004, @popkart-EZ, @jzakrzew, @ffggs, @huanghua1994, @orozery, @danisereb, @rshavitt, @Yihuki, @QingZhou-YangHY, @Jie-Fang, @bbartels\r\n\r\n## New Contributors\r\n\r\n* @abinggo made their first contribution in https://github.com/vllm-project/vllm/pull/42128\r\n* @afierka-intel made their first contribution in https://github.com/vllm-project/vllm/pull/40327\r\n* @alexagriffith made their first contribution in https://github.com/vllm-project/vllm/pull/41987\r\n* @alexeldeib made their first contribution in https://github.com/vllm-project/vllm/pull/43255\r\n* @amitport made their first contribution in https://github.com/vllm-project/vllm/pull/41666\r\n* @anishesg made their first contribution in https://github.com/vllm-project/vllm/pull/43079\r\n* @bedeks made their first contribution in https://github.com/vllm-project/vllm/pull/40269\r\n* @blake-snc made their first contribution in https://github.com/vllm-project/vllm/pull/35568\r\n* @blancsw made their first contribution in https://github.com/vllm-project/vllm/pull/41154\r\n* @bobofang11235 made their first contribution in https://github.com/vllm-project/vllm/pull/42604\r\n* @BWAAEEEK made their first contribution in https://github.com/vllm-project/vllm/pull/42233\r\n* @CynicDora made their first contribution in https://github.com/vllm-project/vllm/pull/39487\r\n* @daniel-devlab made their first contribution in https://github.com/vllm-project/vllm/pull/42479\r\n* @DaoyuanLi2816 made their first contribution in https://github.com/vllm-project/vllm/pull/42905\r\n* @Dogacel made their first contribution in https://github.com/vllm-project/vllm/pull/42764\r\n* @DomBrown made their first contribution in https://github.com/vllm-project/vllm/pull/42080\r\n* @dusthunter made their first contribution in https://github.com/vllm-project/vllm/pull/42594\r\n* @ffggs made their first contribution in https://github.com/vllm-project/vllm/pull/43414\r\n* @frida-andersson made their first contribution in https://github.com/vllm-project/vllm/pull/41825\r\n* @fuergaosi233 made their first contribution in https://github.com/vllm-project/vllm/pull/43488\r\n* @gaozihao-shy made their first contribution in https://github.com/vllm-project/vllm/pull/42869\r\n* @gracie-guo made their first contribution in https://github.com/vllm-project/vllm/pull/42626\r\n* @haic0 made their first contribution in https://github.com/vllm-project/vllm/pull/40453\r\n* @hks-9697-v2 made their first contribution in https://github.com/vllm-project/vllm/pull/42521\r\n* @hlin99 made their first contribution in https://github.com/vllm-project/vllm/pull/42740\r\n* @inisis made their first contribution in https://github.com/vllm-project/vllm/pull/41710\r\n* @izikgo made their first contribution in https://github.com/vllm-project/vllm/pull/42938\r\n* @j9smith made their first contribution in https://github.com/vllm-project/vllm/pull/41215\r\n* @junyanxu made their first contribution in https://github.com/vllm-project/vllm/pull/42671\r\n* @KaivalyaMDabhadkar made their first contribution in https://github.com/vllm-project/vllm/pull/42333\r\n* @libinta made their first contribution in https://github.com/vllm-project/vllm/pull/41689\r\n* @lucifer1004 made their first contribution in https://github.com/vllm-project/vllm/pull/43433\r\n* @meena-at-work made their first contribution in https://github.com/vllm-project/vllm/pull/40082\r\n* @mrjunwan-lang made their first contribution in https://github.com/vllm-project/vllm/pull/43360\r\n* @MrZ20 made their first contribution in https://github.com/vllm-project/vllm/pull/42394\r\n* @mwawrzos made their first contribution in https://github.com/vllm-project/vllm/pull/42498\r\n* @NoeliaBentancor made their first contribution in https://github.com/vllm-project/vllm/pull/42250\r\n* @ovidiusm made their first contribution in https://github.com/vllm-project/vllm/pull/42542\r\n* @paulyu12 made their first contribution in https://github.com/vllm-project/vllm/pull/42306\r\n* @QingZhou-YangHY made their first contribution in https://github.com/vllm-project/vllm/pull/43579\r\n* @qizzzh made their first contribution in https://github.com/vllm-project/vllm/pull/41680\r\n* @qyYue1389 made their first contribution in https://github.com/vllm-project/vllm/pull/42289\r\n* @rasdani made their first contribution in https://github.com/vllm-project/vllm/pull/42481\r\n* @rebklee made their first contribution in https://github.com/vllm-project/vllm/pull/42098\r\n* @revit13 made their first contribution in https://github.com/vllm-project/vllm/pull/42926\r\n* @ruizhang99 made their first contribution in https://github.com/vllm-project/vllm/pull/43260\r\n* @Sarah-Salah made their first contribution in https://github.com/vllm-project/vllm/pull/42441\r\n* @sonusflow made their first contribution in https://github.com/vllm-project/vllm/pull/36329\r\n* @TheDuyIT made their first contribution in https://github.com/vllm-project/vllm/pull/40131\r\n* @tuukkjs made their first contribution in https://github.com/vllm-project/vllm/pull/42880\r\n* @vllm-agent made their first contribution in https://github.com/vllm-project/vllm/pull/42913\r\n* @wangrui6 made their first contribution in https://github.com/vllm-project/vllm/pull/40326\r\n* @wasnertobias made their first contribution in https://github.com/vllm-project/vllm/pull/43001\r\n* @weizhoublue made their first contribution in https://github.com/vllm-project/vllm/pull/42830\r\n* @woernfl made their first contribution in https://github.com/vllm-project/vllm/pull/42397\r\n* @xwu-intel made their first contribution in https://github.com/vllm-project/vllm/pull/37888\r\n* @Yihuki made their first contribution in https://github.com/vllm-project/vllm/pull/42933\r\n* @yiwen101 made their first contribution in https://github.com/vllm-project/vllm/pull/42654\r\n* @ylangtsou made their first contribution in https://github.com/vllm-project/vllm/pull/43038\r\n* @yufufi made their first contribution in https://github.com/vllm-project/vllm/pull/42972\r\n* @zhengluo-nv made their first contribution in https://github.com/vllm-project/vllm/pull/43105\r\n* @zhougit86 made their first contribution in https://github.com/vllm-project/vllm/pull/42739\r\n* @zx3xyy made their first contribution in https://github.com/vllm-project/vllm/pull/42855\r\n","reactions":{"url":"https://api.github.com/repos/vllm-project/vllm/releases/331396286/reactions","total_count":47,"+1":14,"-1":0,"laugh":0,"hooray":16,"confused":0,"heart":1,"rocket":12,"eyes":4},"mentions_count":200},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/322898436","assets_url":"https://api.github.com/repos/vllm-project/vllm/releases/322898436/assets","upload_url":"https://uploads.github.com/repos/vllm-project/vllm/releases/322898436/assets{?name,label}","html_url":"https://github.com/vllm-project/vllm/releases/tag/v0.21.0","id":322898436,"author":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"node_id":"RE_kwDOI7xefs4TPwoE","tag_name":"v0.21.0","target_commitish":"main","name":"v0.21.0","draft":false,"immutable":false,"prerelease":false,"created_at":"2026-05-15T04:28:34Z","updated_at":"2026-05-15T08:54:45Z","published_at":"2026-05-15T08:44:26Z","assets":[{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/420864624","id":420864624,"node_id":"RA_kwDOI7xefs4ZFeJw","name":"vllm-0.21.0+cpu-cp38-abi3-manylinux_2_34_aarch64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":36618044,"digest":"sha256:eeee9f62cbff8e573e9389da4d270ce0f086aacd4f4975811ed0ae377e2683f1","download_count":207,"created_at":"2026-05-15T08:44:37Z","updated_at":"2026-05-15T08:44:40Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.21.0/vllm-0.21.0%2Bcpu-cp38-abi3-manylinux_2_34_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/420864633","id":420864633,"node_id":"RA_kwDOI7xefs4ZFeJ5","name":"vllm-0.21.0+cpu-cp38-abi3-manylinux_2_34_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":95931170,"digest":"sha256:f7473fcd27cc13796e0f13ac79ee55185c283d8821d747d6c9515bef4e10be64","download_count":11172,"created_at":"2026-05-15T08:44:37Z","updated_at":"2026-05-15T08:44:46Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.21.0/vllm-0.21.0%2Bcpu-cp38-abi3-manylinux_2_34_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/420864631","id":420864631,"node_id":"RA_kwDOI7xefs4ZFeJ3","name":"vllm-0.21.0+cu129-cp38-abi3-manylinux_2_34_aarch64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":424234672,"digest":"sha256:de0af3ab4c0cc86e98712bbe89bb30eae967b5bf87873920b7cf13bbfd096aaa","download_count":25684,"created_at":"2026-05-15T08:44:37Z","updated_at":"2026-05-15T08:45:05Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.21.0/vllm-0.21.0%2Bcu129-cp38-abi3-manylinux_2_34_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/420864632","id":420864632,"node_id":"RA_kwDOI7xefs4ZFeJ4","name":"vllm-0.21.0+cu129-cp38-abi3-manylinux_2_34_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":457203916,"digest":"sha256:920777691e340df7a8328adfb1e57b9996dbb537edfb654dd32f70844f5f423d","download_count":159205,"created_at":"2026-05-15T08:44:37Z","updated_at":"2026-05-15T08:45:12Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.21.0/vllm-0.21.0%2Bcu129-cp38-abi3-manylinux_2_34_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/420864620","id":420864620,"node_id":"RA_kwDOI7xefs4ZFeJs","name":"vllm-0.21.0-cp38-abi3-manylinux_2_24_aarch64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":239758862,"digest":"sha256:dc62135a50dc4b412b4f79549208e782f1665e49e8c13c2d29d2c3d94ff8ac97","download_count":1646,"created_at":"2026-05-15T08:44:37Z","updated_at":"2026-05-15T08:44:57Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.21.0/vllm-0.21.0-cp38-abi3-manylinux_2_24_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/420864669","id":420864669,"node_id":"RA_kwDOI7xefs4ZFeKd","name":"vllm-0.21.0-cp38-abi3-manylinux_2_24_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":248151215,"digest":"sha256:f4a75b1391f44c67dc1ca268f5ffed9f6b7fdbc657c93db64e6892c5d1bc320b","download_count":2237,"created_at":"2026-05-15T08:44:41Z","updated_at":"2026-05-15T08:44:57Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.21.0/vllm-0.21.0-cp38-abi3-manylinux_2_24_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/420864701","id":420864701,"node_id":"RA_kwDOI7xefs4ZFeK9","name":"vllm-0.21.0.tar.gz","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/x-gtar","state":"uploaded","size":34452698,"digest":"sha256:ec9ed5eaef1fdc5108131b895958cff97f025e8813ab32bfb5e383c1901e40d4","download_count":2851,"created_at":"2026-05-15T08:44:47Z","updated_at":"2026-05-15T08:44:50Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.21.0/vllm-0.21.0.tar.gz"}],"tarball_url":"https://api.github.com/repos/vllm-project/vllm/tarball/v0.21.0","zipball_url":"https://api.github.com/repos/vllm-project/vllm/zipball/v0.21.0","body":"## Highlights\r\n\r\nThis release features 367 commits from 202 contributors (49 new)!\r\n\r\n* **Transformers v4 deprecated**: This release formally deprecates `transformers` v4 support (#40389). Users should migrate to `transformers` v5.\r\n* **C++20 build requirement**: vLLM now requires a C++20-compatible compiler for compatibility with PyTorch (#40380). This is a **breaking build change**.\r\n* **KV Offload + Hybrid Memory Allocator (HMA)**: The KV offloading subsystem now integrates with the Hybrid Memory Allocator, including scheduler-side sliding window group support and full HMA enablement (#41228, #41445, #39571).\r\n* **Speculative decoding with thinking budget**: Speculative decoding now respects reasoning/thinking budgets, enabling correct spec decode for reasoning models (#34668).\r\n* **TOKENSPEED_MLA backend on Blackwell**: A new TOKENSPEED_MLA attention backend is available for DeepSeek-R1/Kimi-K25 prefill + decode on Blackwell GPUs (#41778).\r\n\r\n### Model Support\r\n* New architectures: MiMo-V2.5 (#40967), Laguna XS.2 (#41129, #41880), Moondream3 (#32325), Qianfan-OCR (#40136), Cohere MoE (#40817), Cohere Eagle (#42078).\r\n* Speculative decoding: EAGLE for Mistral (#41024), Gemma4 MTP (#41745), MTP for MiMo-V2.5 (#41905), Cohere Eagle (#42078).\r\n* DeepSeek V4: AMD/ROCm support (#40871), pipeline parallelism (#41694), `max` reasoning effort (#40982), disaggregated serving fixes (#41957).\r\n* Tool calling: Cohere reasoning and tool parsers (#40422), LFM2/2.5 tool parser (#39243).\r\n* Gemma3/Gemma4: `hidden_act` variant support (#40588), pipeline parallelism fix (#40786), MoE fixes (#41206, #41574, #41401), tool parser crash fix (#41991, #42188).\r\n* Model Runner V2: Qwen3.5/Mamba hybrid model support (#35520), `logprob_token_ids` support (#40559).\r\n* CUDA graph: ViT CUDA graph support for Qwen2.5-VL (#40830).\r\n* Compatibility: Vendor HCXVisionConfig for Transformers v5 (#38447), legacy `rope_type` checkpoint support (#41734).\r\n\r\n### Engine Core\r\n* KV offloading + HMA: Scheduler-side sliding window groups (#41228), full HMA enablement (#41445), multi-connector HMA (#39571), per-job store completion (#39186), DCP/PCP support in OffloadingConnector (#41549), MooncakeStoreConnector for distributed KV offloading (#40900).\r\n* Speculative decoding: Thinking budget support (#34668), independent drafter attention backend selection (#39930), multimodal model support with warning (#41752), per-step allocation elimination (#41043).\r\n* Model Runner V2: Rejection sampling acceptance rate fix (#40651), skip metadata rebuild before draft prefill (#40410), rebuild metadata between draft decode steps (#41162), Qwen3.5/Mamba hybrid support (#35520).\r\n* Routing: Replace routing replay with device cache and async D2H pipeline (#39917).\r\n* Ray: RayExecutorV2 enabled by default (#41421), actor name collision fix for DP > 1 (#40398).\r\n* Stability: Two-phase pause to prevent scheduler deadlock (#39366), thread-safe HF tokenizer wrappers (#41181), OOM prevention via `max_split_size_mb` during model loading (#41268).\r\n* IndexCache support for DSA models (#37735).\r\n\r\n### Hardware & Performance\r\n* **NVIDIA Blackwell**: TOKENSPEED_MLA backend for DSR1/Kimi-K25 (#41778), faster per-token FP8 group quant packed kernel (#41326), FP8 on NVIDIA Thor/SM110 (#39712), CUTLASS scaled mm for non-compatible sizes (#41868).\r\n* **Performance**: FlashInfer top-k/top-p sampler enabled by default (#40376), FP8 FlashInfer attention for ViT (#38065), TurboQuant shared dequant buffers (#40941), `AllPool.forward` 51% faster (#41163), GPU<->CPU sync elimination in pooling (#41433) and attention (#41434), numpy zero-copy embedding serialization (#41681), multimodal processor skip for text-only (#41246), FlashInfer FP8 async TP fusion (#39505), NVFP4 all-gather GEMM fusion for AsyncTP (#41882), re-enable allreduce+RMS fusion for DP/PP (#41458), DeepSeek bf16→fp32 via `torch.mm` (#41300), persistent MLA for sparse backend (#41990), configurable safetensors checkpoint prefetch (#41499), fused mhc_post_pre kernel (#41536), 2D-grid W8W8 group quant kernel (#42153), relaxed memory ordering for KV cache swaps (#39306).\r\n* **AMD ROCm**: ROCm 7.2.2 (#41386), DBO (Dynamic Batch Optimization) (#34726), AITER Fused Allreduce+RMSNorm (#37646), Fused Shared Expert (FSE) for Qwen3-Next (#39280), DeepSeek V3.2 TP4 AITER MLA (#41835), GDN linear attention fusion (#40711), eliminate redundant MoE buffer copies in AITER (#41713), CPU offloading support (#40549), DeepEP API update (#39721), cap Triton paged attention block size to fix shared memory OOM (#38502).\r\n* **CPU**: FP8 attention for AMX/AVX-512 (#39445), FP8 W8A16 linear (#41186), FP8 W8A16 MoE (#41314), DNNL AVX2 W8A8 Int8 (#41318), Gated DeltaNet Attention for Qwen 3.5/3.6 (#41025), RISC-V OMP thread auto-binding (#40569).\r\n* **Intel XPU**: Top-k/top-p sample kernel (#39285), out-of-place all-reduce (#41808), LoRA support (#38206).\r\n* **IBM Power**: VSX attention backend (#40451).\r\n* **FlexAttention**: Re-enabled for batch invariant mode (#40842).\r\n* **MLA**: Abstracted MLA prefill backends, eliminated cuDNN dependency (#32623).\r\n\r\n### Large Scale Serving\r\n* Disaggregated serving: Bi-directional KV cache transfers between P and D (#32553), NIXL transfer redesign (#40731), EPLB memory overhead optimization (#40013), NIXL connector bumped to 1.x (#42364), Mooncake KVConnectorStats for transfer observability (#40414), NIXL P-node pre-admission rejection notification (#41269), KV block release for skipped P-ranks (#40449).\r\n* DCP: Pack output and LSE in DCP A2A (#41160).\r\n* MoE: PluggableLayer interface for out-of-tree MoE runners (#35178).\r\n* LoRA: Initial expert parallel (EP) support (#40867), Qwen3.5 LoRA fusion fix (#37912).\r\n\r\n### Quantization\r\n* **NVFP4**: KV cache support (#40177), Triton dequant/QDQ emulation kernels for Hopper and AMD (#40033), GELU on TRT-LLM NvFP4 fused MoE for Gemma4 (#41050), ModelOpt NVFP4 W4A16 (#41769), NVFP4 all-gather GEMM fusion for AsyncTP (#41882), GLM4-MoE NVFP4 loading fix (#41755).\r\n* **MXFP4**: Humming MXFP4 MoE backend (#41083), FlashInfer CUTLASS MXFP4-MXFP8 MoE fix (#42089).\r\n* **TurboQuant**: Hybrid model and uniform quantization support (#39931).\r\n* **Compressed tensors**: Allow configs with non-explicit ignores (#41965).\r\n* **FP8**: Bias loading fix (#41424), FlashInfer autotune temporarily disabled for correctness (#41524).\r\n* **DSV4**: Improved fused Indexer Q quant kernel (#41428).\r\n\r\n### API & Frontend\r\n* **Responses API**: Streaming tool/function calling with `required` (#40700) and named tool/function choice (#41110), resubmitting output items with missing fields (#41355).\r\n* **OpenAI compatibility**: `system_fingerprint` field in responses (#40537), `prompt_embeds` content part support (#40720), `defer_loading` and `tool_reference` support (#40190), rendered prompt text in chat completion response (#42052), tolerate empty content in forced tool choice (#40148).\r\n* **Tool calling**: XGrammar 0.2.0 with structural tags for strict tool calling + reasoning (#40894), Cohere reasoning/tool parsers (#40422), LFM2/2.5 tool parser (#39243).\r\n* **Tokenizer**: Fastokens support (#41741).\r\n* **RLHF**: Explicit `/start_weight_update` and `/finish_weight_update` APIs (#39212).\r\n* **ASR**: Engine request abort on cancellation (#41266).\r\n* **Configuration**: `VLLM_SKIP_MODEL_NAME_VALIDATION` env var (#34676), configurable model weights loading tracking (#41086), Triton JIT compilation monitor (#40137).\r\n\r\n### Build & Dependencies\r\n* **Breaking**: C++20 required for PyTorch compatibility (#40380).\r\n* **Breaking**: Transformers v4 deprecated (#40389).\r\n* Docker image size reduced by ~2.5 GB via deferred FlashInfer cubin download (#41134).\r\n* CUDA 13.0 wheels switched to PyTorch manylinux_2_28 base (#41416).\r\n* DeepGEMM bundled wheel built per-Python for CPython compatibility (#41516).\r\n* Container image provenance metadata embedded (#40653).\r\n* tpu-inference upgraded to v0.19.0 (#41844).\r\n* NIXL connector bumped to 1.x (#42364).\r\n* ROCm 7.2.2 (#41386).\r\n\r\n## Contributors\r\n\r\n@AndreasKaratzas, @haosdent, @khluu, @yewentao256, @stecasta, @mgoin, @Isotr0py, @hmellor, @chaunceyjiang, @jeejeelee, @noooop, @MatthewBonanni, @njhill, @zyongye, @yzong-rh, @ronensc, @NickLucche, @chaojun-zhang, @dzhengAP, @chfeng-cs, @TheEpicDolphin, @esmeetu, @wzhao18, @ZJY0516, @juliendenize, @kylesayrs, @fadara01, @Etelis, @tianmu-li, @arpera, @ekagra-ranjan, @orozery, @wxsIcey, @jikunshang, @izhuhaoran, @rasmith, @russellb, @Lucaskabela, @Harry-Chen, @alec-flowers, @pmaybank, @Terrencezzj, @hickeyma, @Baekpica, @itej89, @fxmarty-amd, @WoosukKwon, @juhi10071998, @sychen52, @baonudesifeizhai, @vllmellm, @johncalesp, @the-david-oy, @lucianommartins, @bittoby, @Dao007forever, @lyd1992, @yuwenzho, @lesj0610, @sfeng33, @micah-wil, @akii96, @yma11, @SoluMilken, @mmangkad, @SiluPanda, @ojhaanshika, @zhandaz, @bhoomit, @simon-mo, @msanft, @angelayi, @anthonsu, @artem-spector, @zhangxin81, @benoittgt, @joerowell, @yangrz7, @chelnnexy, @liangel-02, @walterbm, @rishitdholakia13, @SKRohit, @BugenZhao, @JaredforReal, @amd-lalithnc, @frgossen, @h-avsha, @DarkLight1337, @danisereb, @laithsakka, @Bortlesboat, @wangluochao902, @Rohan138, @hao-aaron, @puririshi98, @roikoren755, @heachary, @UranusSeven, @dsingal0, @ChenxiQ, @snadampal, @ilmarkov, @wendyliu235, @lequytra, @JisoLya, @LuisRobaina, @sniper35, @eicherseiji, @Yuyi-Ao, @raviguptaamd, @sungsooha, @ganyi1996ppo, @andylolu2, @FredericOdermatt, @ProExpertProg, @rbrugaro-amd, @mcsantiago, @hnt2601, @jinzhen-lin, @taneem-ibrahim, @tomeras91, @alex-jw-brooks, @Aktsvigun, @HanFa, @netanel-haber, @JasonKeyiL, @gshtras, @joa-stdn, @Seven-Streams, @JartX, @xuechendi, @BowenBao, @Akashcodes732, @jeffreywang-anyscale, @czhu-cohere, @zhewenl, @marvinzh, @Lidang-Jiang, @gcanlin, @whx-sjtu, @S1ro1, @liulanze, @Dhruvilbhatt, @laviier, @wi-adam, @aaab8b, @yuankaichen-amd, @ZhanqiuHu, @QwertyJack, @viktorpusTT, @divakar-amd, @starkwj, @benchislett, @jcyang43, @JLiu4Coding, @xy3xy3, @hongxiayang, @amd-mghanimi, @wenyili, @bigPYJ1151, @s-yanev, @AlonKejzman, @noobHappylife, @TomerBN-Nvidia, @MeganEFlynn, @liuzijing2014, @jbuchananr, @lokashrinav, @ssam18, @dllehr-amd, @gmagogsfm, @tpopp, @tjtanaa, @simondanielsson, @zhenwei-intel, @HiroakiMikami, @nholmber, @SumanthRH, @LucasWilkinson, @maeehart, @rishaps, @r-barnes, @gau-nernst, @Kermit-C, @tdoublep, @aoshen02, @Naveassaf, @wangxingran222, @cvan20191, @AbhiOnGithub, @abdulrahman-cohere, @jmamou, @Flink-ddd, @bnellnm, @hqhq1025, @gnovack, @wangxiyuan, @princepride, @jiahanc, @LCAIZJ, @ovidiusm\r\n\r\n## New Contributors\r\n\r\n* @abdulrahman-cohere made their first contribution in https://github.com/vllm-project/vllm/pull/41266\r\n* @AbhiOnGithub made their first contribution in https://github.com/vllm-project/vllm/pull/42180\r\n* @Aktsvigun made their first contribution in https://github.com/vllm-project/vllm/pull/40788\r\n* @amd-mghanimi made their first contribution in https://github.com/vllm-project/vllm/pull/41713\r\n* @Baekpica made their first contribution in https://github.com/vllm-project/vllm/pull/41206\r\n* @benoittgt made their first contribution in https://github.com/vllm-project/vllm/pull/41134\r\n* @bittoby made their first contribution in https://github.com/vllm-project/vllm/pull/41690\r\n* @chelnnexy made their first contribution in https://github.com/vllm-project/vllm/pull/40754\r\n* @ChenxiQ made their first contribution in https://github.com/vllm-project/vllm/pull/40956\r\n* @chfeng-cs made their first contribution in https://github.com/vllm-project/vllm/pull/42066\r\n* @cvan20191 made their first contribution in https://github.com/vllm-project/vllm/pull/40951\r\n* @dzhengAP made their first contribution in https://github.com/vllm-project/vllm/pull/41423\r\n* @ghphotoframe made their first contribution in https://github.com/vllm-project/vllm/pull/40859\r\n* @HiroakiMikami made their first contribution in https://github.com/vllm-project/vllm/pull/40588\r\n* @itej89 made their first contribution in https://github.com/vllm-project/vllm/pull/39721\r\n* @JasonKeyiL made their first contribution in https://github.com/vllm-project/vllm/pull/41068\r\n* @jbuchananr made their first contribution in https://github.com/vllm-project/vllm/pull/39243\r\n* @JisoLya made their first contribution in https://github.com/vllm-project/vllm/pull/41363\r\n* @JLiu4Coding made their first contribution in https://github.com/vllm-project/vllm/pull/41832\r\n* @juhi10071998 made their first contribution in https://github.com/vllm-project/vllm/pull/41050\r\n* @Kermit-C made their first contribution in https://github.com/vllm-project/vllm/pull/42076\r\n* @lequytra made their first contribution in https://github.com/vllm-project/vllm/pull/41401\r\n* @Lidang-Jiang made their first contribution in https://github.com/vllm-project/vllm/pull/38099\r\n* @liulanze made their first contribution in https://github.com/vllm-project/vllm/pull/41571\r\n* @lokashrinav made their first contribution in https://github.com/vllm-project/vllm/pull/41681\r\n* @LuisRobaina made their first contribution in https://github.com/vllm-project/vllm/pull/40720\r\n* @maeehart made their first contribution in https://github.com/vllm-project/vllm/pull/42061\r\n* @marvinzh made their first contribution in https://github.com/vllm-project/vllm/pull/40136\r\n* @mcsantiago made their first contribution in https://github.com/vllm-project/vllm/pull/41492\r\n* @MeganEFlynn made their first contribution in https://github.com/vllm-project/vllm/pull/41880\r\n* @nholmber made their first contribution in https://github.com/vllm-project/vllm/pull/39280\r\n* @pmaybank made their first contribution in https://github.com/vllm-project/vllm/pull/41012\r\n* @raviguptaamd made their first contribution in https://github.com/vllm-project/vllm/pull/34726\r\n* @s-yanev made their first contribution in https://github.com/vllm-project/vllm/pull/41755\r\n* @S1ro1 made their first contribution in https://github.com/vllm-project/vllm/pull/39213\r\n* @Seven-Streams made their first contribution in https://github.com/vllm-project/vllm/pull/40894\r\n* @SiluPanda made their first contribution in https://github.com/vllm-project/vllm/pull/40907\r\n* @SKRohit made their first contribution in https://github.com/vllm-project/vllm/pull/40786\r\n* @snadampal made their first contribution in https://github.com/vllm-project/vllm/pull/32553\r\n* @sniper35 made their first contribution in https://github.com/vllm-project/vllm/pull/32325\r\n* @ssam18 made their first contribution in https://github.com/vllm-project/vllm/pull/41486\r\n* @the-david-oy made their first contribution in https://github.com/vllm-project/vllm/pull/40737\r\n* @wangluochao902 made their first contribution in https://github.com/vllm-project/vllm/pull/41043\r\n* @wenyili made their first contribution in https://github.com/vllm-project/vllm/pull/41901\r\n* @wi-adam made their first contribution in https://github.com/vllm-project/vllm/pull/40749\r\n* @xy3xy3 made their first contribution in https://github.com/vllm-project/vllm/pull/40820\r\n* @yangrz7 made their first contribution in https://github.com/vllm-project/vllm/pull/40449\r\n* @yuankaichen-amd made their first contribution in https://github.com/vllm-project/vllm/pull/40390\r\n* @zhangxin81 made their first contribution in https://github.com/vllm-project/vllm/pull/39904\r\n","reactions":{"url":"https://api.github.com/repos/vllm-project/vllm/releases/322898436/reactions","total_count":34,"+1":12,"-1":0,"laugh":0,"hooray":11,"confused":0,"heart":0,"rocket":11,"eyes":0},"mentions_count":200},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/319885695","assets_url":"https://api.github.com/repos/vllm-project/vllm/releases/319885695/assets","upload_url":"https://uploads.github.com/repos/vllm-project/vllm/releases/319885695/assets{?name,label}","html_url":"https://github.com/vllm-project/vllm/releases/tag/v0.20.2","id":319885695,"author":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"node_id":"RE_kwDOI7xefs4TERF_","tag_name":"v0.20.2","target_commitish":"main","name":"v0.20.2","draft":false,"immutable":false,"prerelease":false,"created_at":"2026-05-08T01:25:36Z","updated_at":"2026-05-10T07:39:12Z","published_at":"2026-05-10T07:37:57Z","assets":[{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/416521133","id":416521133,"node_id":"RA_kwDOI7xefs4Y05ut","name":"vllm-0.20.2+cpu-cp38-abi3-manylinux_2_35_aarch64.whl","label":"","uploader":{"login":"ywang96","id":136131678,"node_id":"U_kgDOCB00Xg","avatar_url":"https://avatars.githubusercontent.com/u/136131678?v=4","gravatar_id":"","url":"https://api.github.com/users/ywang96","html_url":"https://github.com/ywang96","followers_url":"https://api.github.com/users/ywang96/followers","following_url":"https://api.github.com/users/ywang96/following{/other_user}","gists_url":"https://api.github.com/users/ywang96/gists{/gist_id}","starred_url":"https://api.github.com/users/ywang96/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/ywang96/subscriptions","organizations_url":"https://api.github.com/users/ywang96/orgs","repos_url":"https://api.github.com/users/ywang96/repos","events_url":"https://api.github.com/users/ywang96/events{/privacy}","received_events_url":"https://api.github.com/users/ywang96/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":36045171,"digest":"sha256:9c6b1acd4e9c38f6aede6bd63301831bb01d5fd414ca4a652426ffa13e047100","download_count":212,"created_at":"2026-05-10T07:37:15Z","updated_at":"2026-05-10T07:37:19Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.20.2/vllm-0.20.2%2Bcpu-cp38-abi3-manylinux_2_35_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/416521131","id":416521131,"node_id":"RA_kwDOI7xefs4Y05ur","name":"vllm-0.20.2+cpu-cp38-abi3-manylinux_2_35_x86_64.whl","label":"","uploader":{"login":"ywang96","id":136131678,"node_id":"U_kgDOCB00Xg","avatar_url":"https://avatars.githubusercontent.com/u/136131678?v=4","gravatar_id":"","url":"https://api.github.com/users/ywang96","html_url":"https://github.com/ywang96","followers_url":"https://api.github.com/users/ywang96/followers","following_url":"https://api.github.com/users/ywang96/following{/other_user}","gists_url":"https://api.github.com/users/ywang96/gists{/gist_id}","starred_url":"https://api.github.com/users/ywang96/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/ywang96/subscriptions","organizations_url":"https://api.github.com/users/ywang96/orgs","repos_url":"https://api.github.com/users/ywang96/repos","events_url":"https://api.github.com/users/ywang96/events{/privacy}","received_events_url":"https://api.github.com/users/ywang96/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":75799316,"digest":"sha256:d5272bf68bd628d0fe0f13054c680e4c4e8993f99e85e523d3781a531598a325","download_count":1654,"created_at":"2026-05-10T07:37:15Z","updated_at":"2026-05-10T07:37:22Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.20.2/vllm-0.20.2%2Bcpu-cp38-abi3-manylinux_2_35_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/416521129","id":416521129,"node_id":"RA_kwDOI7xefs4Y05up","name":"vllm-0.20.2+cu129-cp38-abi3-manylinux_2_31_aarch64.whl","label":"","uploader":{"login":"ywang96","id":136131678,"node_id":"U_kgDOCB00Xg","avatar_url":"https://avatars.githubusercontent.com/u/136131678?v=4","gravatar_id":"","url":"https://api.github.com/users/ywang96","html_url":"https://github.com/ywang96","followers_url":"https://api.github.com/users/ywang96/followers","following_url":"https://api.github.com/users/ywang96/following{/other_user}","gists_url":"https://api.github.com/users/ywang96/gists{/gist_id}","starred_url":"https://api.github.com/users/ywang96/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/ywang96/subscriptions","organizations_url":"https://api.github.com/users/ywang96/orgs","repos_url":"https://api.github.com/users/ywang96/repos","events_url":"https://api.github.com/users/ywang96/events{/privacy}","received_events_url":"https://api.github.com/users/ywang96/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":422241519,"digest":"sha256:8a58a086c5c4ed2883eee36aaaf6b79c83463d02da3015454acf92afcc8e150e","download_count":19871,"created_at":"2026-05-10T07:37:15Z","updated_at":"2026-05-10T07:37:43Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.20.2/vllm-0.20.2%2Bcu129-cp38-abi3-manylinux_2_31_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/416521130","id":416521130,"node_id":"RA_kwDOI7xefs4Y05uq","name":"vllm-0.20.2+cu129-cp38-abi3-manylinux_2_31_x86_64.whl","label":"","uploader":{"login":"ywang96","id":136131678,"node_id":"U_kgDOCB00Xg","avatar_url":"https://avatars.githubusercontent.com/u/136131678?v=4","gravatar_id":"","url":"https://api.github.com/users/ywang96","html_url":"https://github.com/ywang96","followers_url":"https://api.github.com/users/ywang96/followers","following_url":"https://api.github.com/users/ywang96/following{/other_user}","gists_url":"https://api.github.com/users/ywang96/gists{/gist_id}","starred_url":"https://api.github.com/users/ywang96/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/ywang96/subscriptions","organizations_url":"https://api.github.com/users/ywang96/orgs","repos_url":"https://api.github.com/users/ywang96/repos","events_url":"https://api.github.com/users/ywang96/events{/privacy}","received_events_url":"https://api.github.com/users/ywang96/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":455085432,"digest":"sha256:2f8c2bf2ac6d3d16f930535e66822abd71065468521884eb5b910225b2abef4b","download_count":32749,"created_at":"2026-05-10T07:37:15Z","updated_at":"2026-05-10T07:37:43Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.20.2/vllm-0.20.2%2Bcu129-cp38-abi3-manylinux_2_31_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/416521132","id":416521132,"node_id":"RA_kwDOI7xefs4Y05us","name":"vllm-0.20.2-cp38-abi3-manylinux_2_35_aarch64.whl","label":"","uploader":{"login":"ywang96","id":136131678,"node_id":"U_kgDOCB00Xg","avatar_url":"https://avatars.githubusercontent.com/u/136131678?v=4","gravatar_id":"","url":"https://api.github.com/users/ywang96","html_url":"https://github.com/ywang96","followers_url":"https://api.github.com/users/ywang96/followers","following_url":"https://api.github.com/users/ywang96/following{/other_user}","gists_url":"https://api.github.com/users/ywang96/gists{/gist_id}","starred_url":"https://api.github.com/users/ywang96/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/ywang96/subscriptions","organizations_url":"https://api.github.com/users/ywang96/orgs","repos_url":"https://api.github.com/users/ywang96/repos","events_url":"https://api.github.com/users/ywang96/events{/privacy}","received_events_url":"https://api.github.com/users/ywang96/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":235788487,"digest":"sha256:76ccf4c0554556c06f6b0fb1643742d4cf97dcc69f6ef3f04556d0764126035a","download_count":14251,"created_at":"2026-05-10T07:37:15Z","updated_at":"2026-05-10T07:37:34Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.20.2/vllm-0.20.2-cp38-abi3-manylinux_2_35_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/416521160","id":416521160,"node_id":"RA_kwDOI7xefs4Y05vI","name":"vllm-0.20.2-cp38-abi3-manylinux_2_35_x86_64.whl","label":"","uploader":{"login":"ywang96","id":136131678,"node_id":"U_kgDOCB00Xg","avatar_url":"https://avatars.githubusercontent.com/u/136131678?v=4","gravatar_id":"","url":"https://api.github.com/users/ywang96","html_url":"https://github.com/ywang96","followers_url":"https://api.github.com/users/ywang96/followers","following_url":"https://api.github.com/users/ywang96/following{/other_user}","gists_url":"https://api.github.com/users/ywang96/gists{/gist_id}","starred_url":"https://api.github.com/users/ywang96/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/ywang96/subscriptions","organizations_url":"https://api.github.com/users/ywang96/orgs","repos_url":"https://api.github.com/users/ywang96/repos","events_url":"https://api.github.com/users/ywang96/events{/privacy}","received_events_url":"https://api.github.com/users/ywang96/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":244425988,"digest":"sha256:22a7dd06eb03371298e13d6100f3dedbf307352342aaf08e87c929c60aae9b4d","download_count":31346,"created_at":"2026-05-10T07:37:20Z","updated_at":"2026-05-10T07:37:34Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.20.2/vllm-0.20.2-cp38-abi3-manylinux_2_35_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/416521186","id":416521186,"node_id":"RA_kwDOI7xefs4Y05vi","name":"vllm-0.20.2.tar.gz","label":"","uploader":{"login":"ywang96","id":136131678,"node_id":"U_kgDOCB00Xg","avatar_url":"https://avatars.githubusercontent.com/u/136131678?v=4","gravatar_id":"","url":"https://api.github.com/users/ywang96","html_url":"https://github.com/ywang96","followers_url":"https://api.github.com/users/ywang96/followers","following_url":"https://api.github.com/users/ywang96/following{/other_user}","gists_url":"https://api.github.com/users/ywang96/gists{/gist_id}","starred_url":"https://api.github.com/users/ywang96/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/ywang96/subscriptions","organizations_url":"https://api.github.com/users/ywang96/orgs","repos_url":"https://api.github.com/users/ywang96/repos","events_url":"https://api.github.com/users/ywang96/events{/privacy}","received_events_url":"https://api.github.com/users/ywang96/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/x-gtar","state":"uploaded","size":33526607,"digest":"sha256:58809377798c5335c6e2fe30092abda54d9200b5b8a717b3735a63f5daa0e383","download_count":1253,"created_at":"2026-05-10T07:37:23Z","updated_at":"2026-05-10T07:37:25Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.20.2/vllm-0.20.2.tar.gz"}],"tarball_url":"https://api.github.com/repos/vllm-project/vllm/tarball/v0.20.2","zipball_url":"https://api.github.com/repos/vllm-project/vllm/zipball/v0.20.2","body":"# vLLM v0.20.2\r\n\r\n## Highlights\r\nThis release features 6 commits from 6 contributors (0 new)!\r\n\r\nThis is a small patch release with bug fixes for DeepSeek V4, gpt-oss, and Qwen3-VL\r\n\r\n### Bug Fixes\r\n* **DeepSeek V4 sparse attention**: Re-enable the persistent topk path on Hopper and ensure the memset kernel runs at CUDA graph capture time regardless of `max_seq_len`, fixing the MTP=1 hang on DeepSeek V4 (#41665, revert of #41605).\r\n* **DeepSeek V4 KV cache**: Fixed a \"failure to allocate KV blocks\" error in the V1 engine KV cache manager (#41282).\r\n* **gpt-oss MXFP4 + torch.compile**: Plumbed `hidden_dim_unpadded` through the `moe_forward` fake op so MXFP4 works under `torch.compile` on v0.20.x (#42002, backport of #41646).\r\n* **Qwen3-VL**: Removed an invalid deepstack boundary check that could fail under heavy load (#40932).\r\n\r\n## Contributors\r\n@ywang96, @zyongye, @stecasta, @wzhao18, @Isotr0py, @khluu\r\n","reactions":{"url":"https://api.github.com/repos/vllm-project/vllm/releases/319885695/reactions","total_count":33,"+1":15,"-1":0,"laugh":0,"hooray":5,"confused":0,"heart":4,"rocket":9,"eyes":0},"mentions_count":6},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/316580679","assets_url":"https://api.github.com/repos/vllm-project/vllm/releases/316580679/assets","upload_url":"https://uploads.github.com/repos/vllm-project/vllm/releases/316580679/assets{?name,label}","html_url":"https://github.com/vllm-project/vllm/releases/tag/v0.20.1","id":316580679,"author":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"node_id":"RE_kwDOI7xefs4S3qNH","tag_name":"v0.20.1","target_commitish":"main","name":"v0.20.1","draft":false,"immutable":false,"prerelease":false,"created_at":"2026-05-04T08:56:49Z","updated_at":"2026-05-04T10:36:59Z","published_at":"2026-05-04T10:36:26Z","assets":[{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/411742497","id":411742497,"node_id":"RA_kwDOI7xefs4YirEh","name":"vllm-0.20.1+cpu-cp38-abi3-manylinux_2_35_aarch64.whl","label":"","uploader":{"login":"ywang96","id":136131678,"node_id":"U_kgDOCB00Xg","avatar_url":"https://avatars.githubusercontent.com/u/136131678?v=4","gravatar_id":"","url":"https://api.github.com/users/ywang96","html_url":"https://github.com/ywang96","followers_url":"https://api.github.com/users/ywang96/followers","following_url":"https://api.github.com/users/ywang96/following{/other_user}","gists_url":"https://api.github.com/users/ywang96/gists{/gist_id}","starred_url":"https://api.github.com/users/ywang96/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/ywang96/subscriptions","organizations_url":"https://api.github.com/users/ywang96/orgs","repos_url":"https://api.github.com/users/ywang96/repos","events_url":"https://api.github.com/users/ywang96/events{/privacy}","received_events_url":"https://api.github.com/users/ywang96/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":36044609,"digest":"sha256:51a02c78cb8d688ba299f83b495e261abd862273e1bd1d2974c3875771a01d13","download_count":226,"created_at":"2026-05-04T10:36:29Z","updated_at":"2026-05-04T10:36:33Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.20.1/vllm-0.20.1%2Bcpu-cp38-abi3-manylinux_2_35_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/411742492","id":411742492,"node_id":"RA_kwDOI7xefs4YirEc","name":"vllm-0.20.1+cpu-cp38-abi3-manylinux_2_35_x86_64.whl","label":"","uploader":{"login":"ywang96","id":136131678,"node_id":"U_kgDOCB00Xg","avatar_url":"https://avatars.githubusercontent.com/u/136131678?v=4","gravatar_id":"","url":"https://api.github.com/users/ywang96","html_url":"https://github.com/ywang96","followers_url":"https://api.github.com/users/ywang96/followers","following_url":"https://api.github.com/users/ywang96/following{/other_user}","gists_url":"https://api.github.com/users/ywang96/gists{/gist_id}","starred_url":"https://api.github.com/users/ywang96/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/ywang96/subscriptions","organizations_url":"https://api.github.com/users/ywang96/orgs","repos_url":"https://api.github.com/users/ywang96/repos","events_url":"https://api.github.com/users/ywang96/events{/privacy}","received_events_url":"https://api.github.com/users/ywang96/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":75798752,"digest":"sha256:fbd97e4bf26ecaa32e6d90beae4383fc6e627b84ded6413972505280ee751ca2","download_count":1772,"created_at":"2026-05-04T10:36:29Z","updated_at":"2026-05-04T10:36:36Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.20.1/vllm-0.20.1%2Bcpu-cp38-abi3-manylinux_2_35_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/411742496","id":411742496,"node_id":"RA_kwDOI7xefs4YirEg","name":"vllm-0.20.1+cu129-cp38-abi3-manylinux_2_31_aarch64.whl","label":"","uploader":{"login":"ywang96","id":136131678,"node_id":"U_kgDOCB00Xg","avatar_url":"https://avatars.githubusercontent.com/u/136131678?v=4","gravatar_id":"","url":"https://api.github.com/users/ywang96","html_url":"https://github.com/ywang96","followers_url":"https://api.github.com/users/ywang96/followers","following_url":"https://api.github.com/users/ywang96/following{/other_user}","gists_url":"https://api.github.com/users/ywang96/gists{/gist_id}","starred_url":"https://api.github.com/users/ywang96/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/ywang96/subscriptions","organizations_url":"https://api.github.com/users/ywang96/orgs","repos_url":"https://api.github.com/users/ywang96/repos","events_url":"https://api.github.com/users/ywang96/events{/privacy}","received_events_url":"https://api.github.com/users/ywang96/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":422241415,"digest":"sha256:ff6bf51f62b78088e94393b2ec1404bf2ad7bc6c58ab673180b8dac44c314952","download_count":3264,"created_at":"2026-05-04T10:36:29Z","updated_at":"2026-05-04T10:36:55Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.20.1/vllm-0.20.1%2Bcu129-cp38-abi3-manylinux_2_31_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/411742494","id":411742494,"node_id":"RA_kwDOI7xefs4YirEe","name":"vllm-0.20.1+cu129-cp38-abi3-manylinux_2_31_x86_64.whl","label":"","uploader":{"login":"ywang96","id":136131678,"node_id":"U_kgDOCB00Xg","avatar_url":"https://avatars.githubusercontent.com/u/136131678?v=4","gravatar_id":"","url":"https://api.github.com/users/ywang96","html_url":"https://github.com/ywang96","followers_url":"https://api.github.com/users/ywang96/followers","following_url":"https://api.github.com/users/ywang96/following{/other_user}","gists_url":"https://api.github.com/users/ywang96/gists{/gist_id}","starred_url":"https://api.github.com/users/ywang96/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/ywang96/subscriptions","organizations_url":"https://api.github.com/users/ywang96/orgs","repos_url":"https://api.github.com/users/ywang96/repos","events_url":"https://api.github.com/users/ywang96/events{/privacy}","received_events_url":"https://api.github.com/users/ywang96/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":455084882,"digest":"sha256:9b4e40078eea9777ec0e1ffebbcc352b60903ad044bdc3eb235a8a30e8da259a","download_count":87284,"created_at":"2026-05-04T10:36:29Z","updated_at":"2026-05-04T10:36:59Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.20.1/vllm-0.20.1%2Bcu129-cp38-abi3-manylinux_2_31_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/411742495","id":411742495,"node_id":"RA_kwDOI7xefs4YirEf","name":"vllm-0.20.1-cp38-abi3-manylinux_2_35_aarch64.whl","label":"","uploader":{"login":"ywang96","id":136131678,"node_id":"U_kgDOCB00Xg","avatar_url":"https://avatars.githubusercontent.com/u/136131678?v=4","gravatar_id":"","url":"https://api.github.com/users/ywang96","html_url":"https://github.com/ywang96","followers_url":"https://api.github.com/users/ywang96/followers","following_url":"https://api.github.com/users/ywang96/following{/other_user}","gists_url":"https://api.github.com/users/ywang96/gists{/gist_id}","starred_url":"https://api.github.com/users/ywang96/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/ywang96/subscriptions","organizations_url":"https://api.github.com/users/ywang96/orgs","repos_url":"https://api.github.com/users/ywang96/repos","events_url":"https://api.github.com/users/ywang96/events{/privacy}","received_events_url":"https://api.github.com/users/ywang96/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":235787780,"digest":"sha256:4a5c24a94be8413ce682e72d6c22762ac7be64635cf191f4807d30910f928268","download_count":118,"created_at":"2026-05-04T10:36:29Z","updated_at":"2026-05-04T10:36:46Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.20.1/vllm-0.20.1-cp38-abi3-manylinux_2_35_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/411742540","id":411742540,"node_id":"RA_kwDOI7xefs4YirFM","name":"vllm-0.20.1-cp38-abi3-manylinux_2_35_x86_64.whl","label":"","uploader":{"login":"ywang96","id":136131678,"node_id":"U_kgDOCB00Xg","avatar_url":"https://avatars.githubusercontent.com/u/136131678?v=4","gravatar_id":"","url":"https://api.github.com/users/ywang96","html_url":"https://github.com/ywang96","followers_url":"https://api.github.com/users/ywang96/followers","following_url":"https://api.github.com/users/ywang96/following{/other_user}","gists_url":"https://api.github.com/users/ywang96/gists{/gist_id}","starred_url":"https://api.github.com/users/ywang96/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/ywang96/subscriptions","organizations_url":"https://api.github.com/users/ywang96/orgs","repos_url":"https://api.github.com/users/ywang96/repos","events_url":"https://api.github.com/users/ywang96/events{/privacy}","received_events_url":"https://api.github.com/users/ywang96/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":244425444,"digest":"sha256:11907857c94c226caf82ada92ab09b1e0dbf538b5b80a93aed74be29e861020c","download_count":382,"created_at":"2026-05-04T10:36:34Z","updated_at":"2026-05-04T10:36:52Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.20.1/vllm-0.20.1-cp38-abi3-manylinux_2_35_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/411742554","id":411742554,"node_id":"RA_kwDOI7xefs4YirFa","name":"vllm-0.20.1.tar.gz","label":"","uploader":{"login":"ywang96","id":136131678,"node_id":"U_kgDOCB00Xg","avatar_url":"https://avatars.githubusercontent.com/u/136131678?v=4","gravatar_id":"","url":"https://api.github.com/users/ywang96","html_url":"https://github.com/ywang96","followers_url":"https://api.github.com/users/ywang96/followers","following_url":"https://api.github.com/users/ywang96/following{/other_user}","gists_url":"https://api.github.com/users/ywang96/gists{/gist_id}","starred_url":"https://api.github.com/users/ywang96/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/ywang96/subscriptions","organizations_url":"https://api.github.com/users/ywang96/orgs","repos_url":"https://api.github.com/users/ywang96/repos","events_url":"https://api.github.com/users/ywang96/events{/privacy}","received_events_url":"https://api.github.com/users/ywang96/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/x-gtar","state":"uploaded","size":33518241,"digest":"sha256:d7bfe0c1445cb054db99d02d54d7c53abfec49c6f40171ff8414ffccf5eb7067","download_count":1209,"created_at":"2026-05-04T10:36:37Z","updated_at":"2026-05-04T10:36:39Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.20.1/vllm-0.20.1.tar.gz"}],"tarball_url":"https://api.github.com/repos/vllm-project/vllm/tarball/v0.20.1","zipball_url":"https://api.github.com/repos/vllm-project/vllm/zipball/v0.20.1","body":"# vLLM v0.20.1\r\n\r\nThis is a patch release on top of `v0.20.0` primarily focused on **DeepSeek V4 stabilization and performance improvements**, along with several important bug fixes.\r\n\r\n### DeepSeek V4\r\n* Base model support (#41006).\r\n* Multi-stream pre-attention GEMM (#41061), configurable pre-attn GEMM knob (#41443), and tuned default `VLLM_MULTI_STREAM_GEMM_TOKEN_THRESHOLD` (#41526).\r\n* BF16 and MXFP8 all-to-all support for FlashInfer one-sided communication (#40960).\r\n* PTX `cvt` instruction for faster FP32->FP4 conversion (#41015).\r\n* Integrated tile kernels (`head_compute_mix_kernel`) for optimized head computation (#41255).\r\n* Guard megamoe flag with Pure TP (#41522).\r\n* Fixed persistent topk cooperative deadlock at TopK=1024 (#41189) and inter-CTA init race on RadixRowState (#41444), with temporary disable of persistent topk as a workaround (#41442).\r\n* Fixed import error due to AOT compile cache loading (#41090).\r\n* Fixed torch inductor error (#41135).\r\n* Fixed repeated RoPE cache initialization (#41148).\r\n* Fixed missing type conversion for non-streaming tool calls in DSV3.2/V4 (#41198).\r\n\r\n### Bug Fixes\r\n* Fixed `max_num_batched_token` not being captured in CUDA graph (#40734).\r\n* Fixed `num_gpu_blocks_override` not accounted for in `max_model_len` checks (#41069).\r\n* Auto-disable `expandable_segments` around cumem memory pool (#40812).\r\n* Fixed BailingMoE linear layer (#40859) and MLA RoPE rotation for BailingMoE V2.5 (#41185).\r\n* Fixed reasoning parser kwargs not being passed to structured output (#41199).\r\n* [ROCm] Fixed `input_ids` and `expert_map` args for Quark W4A8 GPT-OSS (#41165).\r\n\r\n## List of contributors\r\n@BugenZhao, @chaunceyjiang, @gau-nernst, @ghphotoframe, @Isotr0py, @jeejeelee, @khluu, @njhill, @Rohan138, @wzhao18, @youkaichao, @ywang96, @ZJY0516, @zixi-qi, @zyongye   ","reactions":{"url":"https://api.github.com/repos/vllm-project/vllm/releases/316580679/reactions","total_count":25,"+1":25,"-1":0,"laugh":0,"hooray":0,"confused":0,"heart":0,"rocket":0,"eyes":0},"mentions_count":15},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/312943348","assets_url":"https://api.github.com/repos/vllm-project/vllm/releases/312943348/assets","upload_url":"https://uploads.github.com/repos/vllm-project/vllm/releases/312943348/assets{?name,label}","html_url":"https://github.com/vllm-project/vllm/releases/tag/v0.20.0","id":312943348,"author":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"node_id":"RE_kwDOI7xefs4SpyL0","tag_name":"v0.20.0","target_commitish":"main","name":"v0.20.0","draft":false,"immutable":false,"prerelease":false,"created_at":"2026-04-27T19:04:32Z","updated_at":"2026-04-27T21:20:28Z","published_at":"2026-04-27T21:20:28Z","assets":[{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/406528980","id":406528980,"node_id":"RA_kwDOI7xefs4YOyPU","name":"vllm-0.20.0+cpu-cp38-abi3-manylinux_2_35_aarch64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":36038767,"digest":"sha256:a583de9da61cd80c5f1146b55786bcabc635eb7c20b3ebefe1794a594e56485b","download_count":153,"created_at":"2026-04-27T11:05:22Z","updated_at":"2026-04-27T11:05:25Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.20.0/vllm-0.20.0%2Bcpu-cp38-abi3-manylinux_2_35_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/406528982","id":406528982,"node_id":"RA_kwDOI7xefs4YOyPW","name":"vllm-0.20.0+cpu-cp38-abi3-manylinux_2_35_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":75793141,"digest":"sha256:46010d28b1ec4b1cff7db306dbe87bd2bd36b2310cbe9e6780fd405a20d9c03b","download_count":1947,"created_at":"2026-04-27T11:05:22Z","updated_at":"2026-04-27T11:05:28Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.20.0/vllm-0.20.0%2Bcpu-cp38-abi3-manylinux_2_35_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/406528979","id":406528979,"node_id":"RA_kwDOI7xefs4YOyPT","name":"vllm-0.20.0+cu129-cp38-abi3-manylinux_2_31_aarch64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":422288789,"digest":"sha256:c9aef095f1cdf466cdc29f455a3a4017c1e61cd3ca9577bba3d55618dd9573b6","download_count":26245,"created_at":"2026-04-27T11:05:22Z","updated_at":"2026-04-27T11:05:50Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.20.0/vllm-0.20.0%2Bcu129-cp38-abi3-manylinux_2_31_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/406528981","id":406528981,"node_id":"RA_kwDOI7xefs4YOyPV","name":"vllm-0.20.0+cu129-cp38-abi3-manylinux_2_31_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":455151305,"digest":"sha256:b687351527b0bca7f6b0f367b6c93499e19b06f0c9a05b01084c043f7ec50ce7","download_count":47012,"created_at":"2026-04-27T11:05:22Z","updated_at":"2026-04-27T11:05:49Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.20.0/vllm-0.20.0%2Bcu129-cp38-abi3-manylinux_2_31_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/406528984","id":406528984,"node_id":"RA_kwDOI7xefs4YOyPY","name":"vllm-0.20.0-cp38-abi3-manylinux_2_35_aarch64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":235776358,"digest":"sha256:29a135ca0d70650f057f15c7c0b560d24659524c771f70fbddc24597c861c118","download_count":206134,"created_at":"2026-04-27T11:05:22Z","updated_at":"2026-04-27T11:05:39Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.20.0/vllm-0.20.0-cp38-abi3-manylinux_2_35_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/406529043","id":406529043,"node_id":"RA_kwDOI7xefs4YOyQT","name":"vllm-0.20.0-cp38-abi3-manylinux_2_35_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":244415937,"digest":"sha256:24d28892e210200f6e1bd13f699c42a74cd2bb7364c11248e2348f677c7f6dfb","download_count":206008,"created_at":"2026-04-27T11:05:26Z","updated_at":"2026-04-27T11:05:41Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.20.0/vllm-0.20.0-cp38-abi3-manylinux_2_35_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/406529068","id":406529068,"node_id":"RA_kwDOI7xefs4YOyQs","name":"vllm-0.20.0.tar.gz","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/x-gtar","state":"uploaded","size":33508260,"digest":"sha256:a6d50152936ee292455af3ffbe359f7a284ac43bf3b68caccf29f368e196cc72","download_count":1818,"created_at":"2026-04-27T11:05:28Z","updated_at":"2026-04-27T11:05:32Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.20.0/vllm-0.20.0.tar.gz"}],"tarball_url":"https://api.github.com/repos/vllm-project/vllm/tarball/v0.20.0","zipball_url":"https://api.github.com/repos/vllm-project/vllm/zipball/v0.20.0","body":"# vLLM v0.20.0\r\n\r\n## Highlights\r\nThis release features 752 commits from 320 contributors (123 new)!\r\n\r\n* **DeepSeek V4**: Initial DeepSeek V4 support landed (#40860), with DSML token-leakage fix in DSV4/3.2 (#40806), DSA + MTP IMA fix (#40772), and a silu clamp limit on the shared expert (#40950).\r\n* **CUDA 13.0 default**: Default CUDA wheel on PyPI and `vllm/vllm-openai:v0.20.0` image switched to CUDA 13.0; architecture lists and build-args cleaned up (#39878), and CUDA bumped to 13.0.2 to match PyTorch 2.11.0 (#40669). As a general rule of thumb, our CUDA version policy follows PyTorch's. We highly recommend to install vLLM with `uv` and use `--torch-backend=cu129` if you are on CUDA 12.9.\r\n* **PyTorch 2.11 upgrade** (#34644): vLLM ships on torch 2.11 for CUDA, and XPU is now also on torch 2.11 (#37947) — XPU is no longer pinned to 2.10. This is a breaking change for environment dependency.\r\n* **Python 3.14**: Added to the supported Python version list (#34770).\r\n* **Transformers v5**: vLLM now runs on HuggingFace `transformers>=5` (#30566), with vision-encoder torch.compile bypass (#30518) and continued v4/v5 compat fixes including PaddleOCR-VL image processor `max_pixels` (#38629), Mistral YaRN warning (#37292), and Jina ColBERT rotary inv_freq recompute (#39176).\r\n* **New large models**: Hunyuan v3 (Hy3) preview (#40681) with HYV3 reasoning parser (#40713); Granite 4.1 Vision as a built-in multimodal model (#40282).\r\n* **FlashAttention 4 as default MLA prefill**: FA4 re-enabled as the default MLA prefill backend (#38819) with head-dim 512 and paged-KV support on SM90+ (#38835), plus an upstream FA4 sync (#38690).\r\n* **TurboQuant 2-bit KV cache**: New attention backend delivering 2-bit KV cache compression with 4× capacity (#38479), now with FA3/FA4 prefill support (#40092).\r\n* **Online quantization frontend**: New end-to-end online quantization frontend (#38138), with docs (#39736); experts_int8 consolidated into the FP8 online path (#38463); MXFP8 online quant moved to the new frontend (#40152).\r\n* **vLLM IR**: Initial IR skeleton with rms_norm op (#33825), OOT-platform kernel imports (#38807), gemma_rms_norm reworked on IR (#39014), and IR op testing/benchmarking infra added (#40167) — foundation for future kernel work.\r\n* **Model Runner V2 advances**: Eagle prefill full-CUDA-graph (#37588), auto-resolve cudagraph mode/sizes from attention backend (#32936), fused probabilistic rejection sample kernels (#38496), config validation for unsupported features (#38758), piecewise-fallback disabled for eagle draft decodes (#39773), multiple prompt-logprobs support (#39937), prefill warmup coverage (#40746), and a fix for accuracy regression caused by stale sampled/draft tokens (#39833).\r\n* **MoE refactor series**: Unquantized migrated to Full Oracle Flow (#36286), CT W8A8 to Oracle (#39187), SharedExperts class (#35153), `SharedFusedMoE` removed (#35782), DefaultMoERunner split (#35326) and later combined back into `MoERunnerBase` (#40560), shared/fused expert output sum moved into `MoERunnerBase` (#35949), ZeroExpertFusedMoE in new framework (#35549), `compressed_tensors_moe.py` split (#38960), `GPTQMarlinMoEMethod` reworked with MK (#37990), XPU & CUTLASS MoE relocated to `fused_moe/experts/` (#40568, #40574), `make_expert_params_mapping` renamed (#40671), MoE LoRA refactor (#40338), and MoE DP chunking removed (#39107).\r\n* **Performance**: Optimize batch invariant with fused rms norm — 2.1% E2E latency improvement (#40413); avoid `seq_lens_cpu` GPU→CPU sync (#40654); cache `InductorPass.hash_source` (#39328); skip FX-graph deserialization on loading for faster warm compile (#40151); CUDAGraph memory profiling enabled by default for clearer startup memory accounting (#38284).\r\n\r\n### Model Support\r\n* New architectures: DeepSeek V4 (#40860), Hunyuan v3 preview (#40681), Granite 4.1 Vision (#40282), EXAONE-4.5 (#39388), BharatGen Param2MoE (#38000), Phi-4-reasoning-vision-15B (#38306), Cheers multimodal (#38788), telechat3 (#38510), FireRedLID (#39290), jina-reranker-v3 (#38800), Jina Embeddings v5 (#39575), Nemotron-v3 VL Nano/Super (#39747).\r\n* Gemma4 series: fast prefill (#38879), quantized MoE (#39045), Eagle3 (#39450), block-local attention + YaRN for Gemma3 (#39823), bidirectional vision attention for sliding layers (#40534), token-repetition fix via dynamic BOS (#39842), multimodal embedder norm-order fix (#40411), plus a string of streaming/tool-call fixes (#38844, #38909, #38992, #39114, #39679, #39027).\r\n* Quantization formats: GGUF support for MiniMax-M2.1 (#36965), non-standard GGUF quant types with prefix such as UD-IQ1_S (#39471).\r\n* Speculative decoding: Eagle3 for MiniMax-M2 (#37512), Eagle3 for Gemma4 (#39450).\r\n* LoRA: Qwen3ASRForConditionalGeneration (#37247), Gemma4ForConditionalGeneration (#39291, #38844), DeepSeek V3.2 (#35077), Qwen3.5 / Step3.x expert base_layer extension (#37114), MoE LoRA refactor (#40338), dual-CUDA-streams linear layer (#35721).\r\n* Multimodal MRoPE refresh: mm_features-based MRoPE for Ernie-4.5 VL (#39753), Keye-VL / Keye-1.5-VL (#39869), PaddleOCR-VL (#39888).\r\n* Other: Nano-Nemotron-VL static image inputs fix (#40724); Qwen3 MoE no longer calls gate twice (#40664); DeepSeek V2-Lite accuracy drop fix (#40673); Parakeet UX / perf enhancements (#39423); ColModernVBERT updated for latest HF checkpoint (#39307); NemotronH default `mamba_ssm_cache_dtype=float32` with NemotronHNanoVLV2 auto-hook (#39032); new TP plan styles for the Transformers backend (#40467); GLM-5.1 fix on ROCm (#40763).\r\n\r\n### Engine Core\r\n* **Model Runner V2**: Full CUDA graph for eagle prefill (#37588), auto cudagraph mode/sizes based on attention backend (#32936), fused probabilistic rejection-sample kernels (#38496), config validation (#38758), eagle-draft piecewise fallback disabled (#39773), multiple prompt logprobs (#39937), prefill warmup coverage (#40746), stale sampled/draft tokens accuracy fix (#39833).\r\n* **vLLM IR**: IR skeleton + rms_norm (#33825), OOT kernel import hooks (#38807), gemma_rms_norm on IR (#39014), IR op testing/benchmarking infra (#40167).\r\n* **torch.compile**: Opaque Objects on torch 2.11 (#39286), AOT compile with batch-invariance mode (#39201), Inductor cache nested under AOT dir (#39718), split FX graph via codegen (#38657), Inductor pre-grad passes re-enabled for torch≥2.12 (#38944), strings in custom ops without compile regressions (#38123), MLA + group FP8 fusion (#38877), SiluMul activation+quant fusion refactor (#39684), `donate_graph_module=True` for `standalone_compile` (#39733), skip FX graph deserialization on loading (#40151), include Inductor & functorch configs in compile-cache key (#40627), respect `TORCH_COMPILE_DISABLE` at vLLM config level (#40715), disable Sequence Parallelism for piecewise compilation (#38373).\r\n* **Attention**: FA4 as default MLA prefill (#38819), head-dim 512 + paged-KV on sm90+FA4 (#38835), FA4 upstream sync (#38690), full CUDA graph for FlexAttention (#36298), FlexAttention non-causal support (#40394), unified 2D/3D triton_unified_attention (#40631), TRTLLM minimax_allreduce_rms ported (#37045), `concat_mla_q` half-types only (#37892), batch-invariance-aware backend auto-selection (#40193), avoid `seq_lens_cpu` GPU→CPU sync (#40654).\r\n* **Helion kernels**: torch.compile support for Helion kernels (#38592).\r\n* **HMA / KV offload**: GPU-side KV events for HMA (#37688), group block hashes/IDs tracked (#37109), unified memory layout for offloading workers (#37206), `shutdown()` on OffloadingConnector (#39182), request context passed through KV offload (#39185), sliding-window lookup (#36645), multi-group worker transfer (#38453), multi-KV-group lookup/load/store (#39401, #39402, #39403).\r\n* **Features**: NUMA binding for GPU workers (#38635), opt-in `VLLM_MEDIA_CACHE` media URL caching (#37123), safe request abort when FSM fails to advance (#38663), KV connector prioritized over internal registry (#38301), CUDAGraph memory profiling on by default (#38284), shared-expert overlap restored (#39222), `CONFIG_REGISTRY` config-class lookup fix when on-disk model_type differs (#39554), workspace-resize GPU memory leak fix (#39226), SWA/chunked-local runtime admission capped to startup pool-sizing bound (#40946).\r\n* **Pluggable layers**: Applied to llm_head / vocab embedding (#33465) and MoE layers (#33556).\r\n* **Mamba**: Stochastic rounding (#35753), different Conv state layouts (#37416), FlashInfer `selective_state_update` (#36162).\r\n* **Metrics & scheduling**: Labeled waiting-breakdown (capacity/deferred) metric (#38435), API server handshake simplified (#39364), mm-scheduler `get_num_embed` overhead reduced (#40143), `request_id` on `FinishedRequestStats` (#39710).\r\n* **Executor**: RayExecutorV2 introduced (#36836); unified engine process monitoring with Ray backend (#35862).\r\n\r\n### Hardware & Performance\r\n* **NVIDIA**: swapAB support for SM120 CUTLASS blockwise FP8 GEMM (#38325), MXFP4 W4A4 CUTLASS MoE for SM100 (#37463), TRTLLM GEN NVFP4 MoE with non-512-aligned hidden dims via weight padding (#39510), TRTLLM FP8 MoE with shuffled weights + BlockMajorK layout (#38993), fused qknorm+rope kernel on SM9.0 (#37376), tuned fused_moe config for RTX PRO 6000 Blackwell (#39183), ViT full CUDA graph for Qwen3-VL video (#38061), `--enable-vit-cuda-graph` for VLM examples (#40580), default `max_frames_per_batch` auto-infer for ViT CG video (#40445), fused FP8 output quantization into `merge_attn_states` (#36518), batched KV-cache swap via `cuMemcpyBatchAsync` (#38460), sm_110 (Jetson Thor) added to CUDA 13.0 build targets (#39233).\r\n* **AMD ROCm**: ZenCPU / AMD Zen CPU backend via zentorch (#39967), RDNA 3.5/4 device IDs (gfx1150/1151/1201) (#38455), gfx1102/gfx1103 added (#40037), MORI EP for unquantized MoE with AITER (#37529), MoRI build with AMD AINIC stack (#38371), MoRI-IO message format aligned with P2pNcclConnector and vllm-router (#39565), MORI prefill/decode API correction (#39835), AITER gemm w8a8 ptpc integration (#33773), TritonW4A16LinearKernel (#37352), asymmetric INT8 in `TritonInt8ScaledMMLinearKernel` (#38501), `fused_silu_mul_block_quant` enabled (#38817), KV-cache shuffle for `paged_attention_common` (#32914), MLA decode output zero-fill removed in AITER (#37539), MLA dual RMS norm fusion pass for DeepSeek/Kimi-K2 (#39242, with older-AITer guard #40386), AITER MLA + Eagle3 spec decode (#39616), DFlash on ROCm (#39703), wvSplitK FP8 path for RDNA (#37712), GPU↔NUMA-node detection (#40015), non-causal attention in `ROCM_ATTN` (#40176), engine-shutdown GPU memory leak fix (#38503), score-correction-bias dtype cast for DeepSeek/Kimi-K2 (#39999).\r\n* **Intel XPU**: torch 2.11 upgrade for XPU (#37947) — no longer pinned to 2.10, initial GDN attention for Qwen3-Next / Qwen3.5 (#33657), torch.compile for XPU GDN attention (#39466), XPU MXFP8 quant op (#38682), XPU MXFP4 quant op (#39857), per-channel FP8 linear (#38316), FP8 KV cache on XPU (#37731), `round_int8` for Intel Triton (#38825), MoE Triton in online FP8 quantization fix (#40109), `current_platform.supports_fp8()` updated for TritonExperts (#40132), NIXL import on XPU fix (#40430), fusion-pattern support disabled on XPU (#39789).\r\n* **CPU**: CPU draft-model speculative decoding (#32662), CPU int8 compute mode in AWQ (#35697), head_size 512 in `cpu_attn` (#38676), gelu in `cpu_fused_moe` (#38770), OMP replacement (#36487), BF16 GELU LUT on ARM (#37469), W4A16 Autoround on CPU (#38192), CPU affinity/memory mgmt refactor (#39781), IBM Z s390x torch 2.11 builds (#39910), faster exp routine for lower-precision dtypes (#38112), inter-node pipeline parallel fix (#40150), RISC-V multiple RVV VLEN targets (#39478), RISC-V platform detection fix (#40427), exp() input clamp to prevent NaN on CPU/RISC-V (#40428).\r\n* **TPU**: tpu-inference upgraded to 0.18.0 (#40395).\r\n* **DeepSeek / MLA / Indexer**: Persistent TopK scheduler for DSV3.2 DSA decode (#37421), DSV3.2 indexer fused weights projection (#38684), Triton MLA perf fixes (#33529), indexer WK upcast to BF16 for fusion (#38928), MLA indexer uniform-decode optimization for MTP>1 (#39458), DSA + MTP IMA fix (#40772).\r\n* **GDN / Mamba**: Kernel fusion in GDN (#37813), TMA aligned with upstream FLA (#38981), GPU↔CPU syncs eliminated in prefill and spec-decode paths (#38361, #38047).\r\n* **Other**: DeepGEMM integrated into the vLLM wheel via CMake (#37980), Lustre FS checkpoint prefetching enabled by default (#39422), Gemma4 fused routing Triton kernel (#39083), Gemma4 embed_input_ids GPU/CPU sync removed (#39234), Nemotron VL image/video preprocessing optimized (#40283), SiLU block-quant fusion v1 (#32996), bilinear_pos_embed Triton kernel for ViT (#37948), mean-pooling optimization (~5.9% throughput) (#38559), redundant-sync removal for pooling (~3.7% throughput) (#39113), H2D pageable-memory copy reduction (#38794), fused zero initializer for FP8 DeepGemm block-quant (#39547), batch-invariant fused-rms-norm 2.1% E2E latency improvement (#40413), `InductorPass.hash_source` cached (#39328), humming quantization kernel (#34556).\r\n\r\n### Large Scale Serving\r\n* **EPLB**: Alternative communication for EPLB weight exchange (#33176), nixl-based EPLB communicator (#36276), mapping optimization with router record for prefill (#36261), `TransferMetadata` consolidation (#37341), Async EPLB synchronization refactor (#37601), asyncio infrastructure removed from Async EPLB (#40730), replica-selection bias fix in fused_moe router (#40810), Async EPLB integration test added (#40168).\r\n* **WideEP**: Naive all2all replaced by allgather + reducescatter (#33728).\r\n* **KV Offload / Connector**: 3FS KVConnector (#37636), unified memory layout for offloading workers (#37206), cache_salt propagated through MP connector for per-user isolation (#39837), multi-connector metrics of same type (#40010), LMCache block-allocation event (#38856), LMCache MP save optimization with MLA (#38810), `num_lmcache_extra_cached_token` in KVTransferParams (#39843), offload all KV blocks during prefill in P/D (#40346), DP control bundle pinned to first GPU's node on Ray (#39167), FlashInfer NVLink MNNVL workspace sized to EP group (#40893).\r\n* **Disaggregated / NIXL / Mamba**: Full PD support for Mamba2-like models on Heterogeneous TP deployments (#37635), Nixl bumped to 0.10.1 (#39922), `TpKVTopology` + `HeteroTPTransferConfig` unified into `TransferTopology` (#39529), NIXL EP treated as batched experts in fused_moe (#40412).\r\n\r\n### Quantization\r\n* **New formats & methods**: TurboQuant 2-bit KV cache compression (#38479) with FA3/FA4 prefill (#40092), per-token-head INT8/FP8 KV cache quantization (#38378), fused FP8/NVFP4 output quantization in MLA attention (#35792), NVFP4 dense models on MI300/MI355X and Hopper via emulation (#35733), NVFP4 MoE emulation fallback for H100/MI300/MI350 (#35737), humming quantization kernel (#34556).\r\n* **Kernels**: MXFP8 in Marlin GEMM/MoE with Mxfp8LinearOp refactor (#34664), MXFP4 W4A4 CUTLASS MoE for SM100 (#37463), NVFP4 in `reshape_and_cache_flash` (#37332), batch-invariant NVFP4 linear (#39322), FlashInfer CuteDSL batched-experts backend for NVFP4 MoE (#38251), special `GptOssMxfp4MoeMethod` (#39604), W4A8_FP8 MoE TP>1 correctness fix (#40310), NVFP4 CUTLASS MoE OOB-read fix for non-multiple-of-4/16 expert counts (#40351), RMS norm + quant fusion fix on DeepGEMM UE8M0 path for B200 (#40552), Gemma4 quantized MoE (#39045).\r\n* **Compressed tensors**: W8A8 MXFP8 linear/MoE (`CompressedTensorsW8A8Mxfp8`) (#38815), CT W8A8 in Oracle structure (#39187), layerwise reloading of attention/KV quantized models (#38995), experts_int8 consolidated with FP8 online quant (#38463), MXFP8 online quant on the new frontend (#40152).\r\n* **Online quant**: Quantized model init failure fix with prefetch offloading (#40432), `current_platform.supports_fp8()` updated for TritonExperts on XPU/ROCm (#40132).\r\n* **XPU / CPU / AMD**: XPU MXFP4 (#39857), XPU MXFP8 GEMM + compressed-tensor schema (#38707), XPU FP8 per-channel linear (#38316), FP8 KV cache on XPU (#37731), CPU W4A16 Autoround (#38192), XPU W4A16 Autoround (#37986), asymmetric INT8 `TritonInt8ScaledMMLinearKernel` on ROCm (#38501), Quark W8A8 INT8 MoE inference (#36320).\r\n* **Deprecations**: Petit NVFP4 removed (#32694).\r\n\r\n### API & Frontend\r\n* **OpenAI / Anthropic API**: `presence_penalty` / `frequency_penalty` on Responses API (#38613), Responses API streaming migrated to unified parser (#38755), `tool_choice` / `tools` validation on Responses to match OpenAI (#40399), Mistral Grammar factory (#38150), multimodal support on `/inference/v1/generate` (#38405), `max_tokens_per_doc` in rerank (#38827), Generative Scoring (#34539), MaxSim re-enabled on GPU (#38620), `chat_template_kwargs` on Anthropic `/v1/messages` (#40125), auto-detection of `reasoning_config` when only `reasoning_parser` is set (#38214), reasoning parsers can access model config via `adjust_request` (#37848, #39027), effective chat-template kwargs passed to reasoning parsers (#40460), reasoning parsers expose `reasoning_start_str`/`reasoning_end_str` (#40566).\r\n* **Pooling ecosystem**: Pooling entrypoints overhauled across scoring (#28631), pooling (#39153), and cleanup (#39675); preprocessing/postprocessing offloaded to thread pool (#39763); async scheduling disabled by default for pooling (#39592); `logit_scale` added to PoolerConfig (#39435), then renamed `logit_bias`/`logit_scale` → `logit_mean`/`logit_sigma` for affine score calibration (#39530) — breaking. `LLM.reward` deprecated; use `LLM.encode` instead (#40688).\r\n* **gRPC / streaming**: Streaming on token-generation endpoint (#37171); gRPC periodic stats logging + servicer log forwarding (#38333); standard `grpc.health.v1` health check for Kubernetes-native probes (#38016).\r\n* **Tool / reasoning parsers**: Treat `<tool_call>` as implicit reasoning end in Qwen3 (#35687), `is_reasoning_end_streaming()` override for GptOssReasoningParser (#35745), Mistral tool parser HF-tokenizer fix (#39294), Mistral pre-v11 tool parser trailing-output fix (#40531), Gemma4 streaming HTML duplication / JSON corruption / null-as-string fixes (#38909, #38992, #39114, #39679), HF tokenizer concurrent-borrow fix in tool parsers (#40059), `HYV3ReasoningParser` no longer mutates `chat_template_kwargs` (#40713).\r\n* **Multimodal**: Externally processed `mm_kwargs` with cache injection (#39502), PyAV video backend for concurrent decoding (#39986), custom video metadata for pre-extracted frame sequences (#40133), image+video mixed inputs (per prompt) for VLM examples (#40335), deepstack buffer optimized for Qwen3 multimodal (#40145), readonly multimodal processor warmup during renderer startup (#40797), `mm_processor_kwargs` forwarded in offline `generate` APIs (#40251), normalize malformed dict prompts that carry token IDs in `prompt` (#40339), hotwords for FunASR (#39674), bundle `get_generation_prompt()` params into `SpeechToTextParams` (#36268).\r\n* **Frontend / vLLM Omni**: `--omni` delegates to vLLM Omni (#40744); avoid eager import of `mistral_common` (#40043).\r\n* **LLM / CLI**: Structured-output special tokens preserved in offline `LLM.chat` (#39352), `use_audio_in_video` passable at `vllm serve` for nemotron-nano-vl (#38538), deferred imports save ~2s CLI startup (#40056), improved MM-input-too-long error message (#39409), warning when FP8 KV cache misses prefill query quant (#39752), clearer DCP error message (#28443), `--model` deprecation warning updated (#39518), Mimo reasoning/tooling parsers mapped (#40089), human-readable `k/K/m/M…` suffix in JSON CLI args (#40473).\r\n\r\n### Spec Decode\r\n* Eagle3 for MiniMax-M2 (#37512), Eagle3 for Gemma4 (#39450), AITER MLA + Eagle3 on ROCm (#39616).\r\n* TurboQuant FA3/FA4 for prefill paths (#40092).\r\n* Mamba: default to `'align'` cache mode for Mamba-based models when speculative decoding is enabled (#40454).\r\n* Unified Synthetic Acceptance Rate for V1 and V2 (#40662); `SpecDecodeBaseProposer` moved out of `eagle.py` (#40732); DSA + MTP IMA fix (#40772).\r\n\r\n### Security\r\n* SSRF fix in batch runner `download_bytes_from_url` (#38482).\r\n\r\n### Dependencies\r\n* **PyTorch 2.11** for CUDA (#34644) and XPU (#37947) — XPU no longer pinned to 2.10.\r\n* **CUDA 13.0** default with updated architecture lists and cleaned build-args (#39878); CUDA bumped to 13.0.2 to match PyTorch 2.11.0 (#40669); sm_110 (Jetson Thor) added (#39233).\r\n* **Python 3.14** added to supported versions (#34770).\r\n* **Transformers v5** (#30566), with vision-encoder torch.compile bypass (#30518) and continued v4/v5 compat fixes.\r\n* **FlashAttention 4** upstream sync (#38690) and symlink-on-install behavior (#38814).\r\n* **FlashInfer** bumped to 0.6.8 (#39959).\r\n* **AITER** triton BUFFER_OPS fix + version updates (#38580), AITER reverted to v0.1.10.post3 (#39509); **Nixl** bumped to 0.10.1 (#39922) and pinned per CUDA major in CI (#39851); **DeepGEMM** integrated into the wheel via CMake (#37980); **fastsafetensors** added to NVIDIA Dockerfile (#38950); Helion bumped 0.3.2 → 0.3.3 (#38062).\r\n* **Removed / moved**: `resampy` dependency dropped (#39524), `librosa` direct dependency dropped (#39079), `pyav` and `soundfile` moved to common requirements (#39997).\r\n\r\n### Breaking Changes\r\n1. **PyTorch 2.11 + CUDA 13.0(.2) default** — environment dependency change, now applied to XPU as well.\r\n2. **Transformers v5** is the supported baseline (#30566).\r\n3. **Metrics rework**: `vllm:prompt_tokens_recomputed` removed (#38709); `num_cached_tokens` / `num_external_computed_tokens` replaced with `PrefillStats` (#37460).\r\n4. **Pooler config rename**: `logit_bias`/`logit_scale` → `logit_mean`/`logit_sigma` (#39530).\r\n5. **Async scheduling default OFF for pooling models** (#39592).\r\n6. **CUDAGraph memory profiling now ON by default** (#38284) — startup memory accounting changes.\r\n7. **Petit NVFP4 quantization removed** (#32694); `LLM.reward` deprecated, use `LLM.encode` (#40688); `cprofile` / `cprofile_context` deprecated (#39100); V0 `accept output buffer` deprecated (#39125).\r\n\r\n### V0 Deprecation\r\n* Petit NVFP4 (#32694), `accept output buffer` in attention (#39125), `cprofile` / `cprofile_context` (#39100), `LLM.reward` offline API (#40688).\r\n\r\n## New Contributors\r\n* @1096125073 made their first contribution in https://github.com/vllm-project/vllm/pull/38510\r\n* @2imi9 made their first contribution in https://github.com/vllm-project/vllm/pull/38970\r\n* @AAISSJ made their first contribution in https://github.com/vllm-project/vllm/pull/37831\r\n* @abatilo made their first contribution in https://github.com/vllm-project/vllm/pull/38987\r\n* @aditi-amd made their first contribution in https://github.com/vllm-project/vllm/pull/39953\r\n* @aeon-x made their first contribution in https://github.com/vllm-project/vllm/pull/39843\r\n* @Alchuang22-dev made their first contribution in https://github.com/vllm-project/vllm/pull/40339\r\n* @aleksandaryanakiev made their first contribution in https://github.com/vllm-project/vllm/pull/40125\r\n* @aliialsaeedii made their first contribution in https://github.com/vllm-project/vllm/pull/38253\r\n* @artem-spector made their first contribution in https://github.com/vllm-project/vllm/pull/40282\r\n* @bai made their first contribution in https://github.com/vllm-project/vllm/pull/39959\r\n* @bhargav-patel-29 made their first contribution in https://github.com/vllm-project/vllm/pull/38000\r\n* @bingshuailiu made their first contribution in https://github.com/vllm-project/vllm/pull/38788\r\n* @Bortlesboat made their first contribution in https://github.com/vllm-project/vllm/pull/39123\r\n* @BugenZhao made their first contribution in https://github.com/vllm-project/vllm/pull/40460\r\n* @carlyou made their first contribution in https://github.com/vllm-project/vllm/pull/36205\r\n* @Chinmay-Kulkarni-AMD made their first contribution in https://github.com/vllm-project/vllm/pull/39967\r\n* @crawfordxx made their first contribution in https://github.com/vllm-project/vllm/pull/38722\r\n* @daiyu1111 made their first contribution in https://github.com/vllm-project/vllm/pull/40011\r\n* @dalistarh made their first contribution in https://github.com/vllm-project/vllm/pull/40194\r\n* @daniebrill made their first contribution in https://github.com/vllm-project/vllm/pull/36934\r\n* @dhonnappa-amd made their first contribution in https://github.com/vllm-project/vllm/pull/38238\r\n* @dondetir made their first contribution in https://github.com/vllm-project/vllm/pull/38455\r\n* @efortin made their first contribution in https://github.com/vllm-project/vllm/pull/39183\r\n* @elenalil-aws made their first contribution in https://github.com/vllm-project/vllm/pull/38927\r\n* @elwhyjay made their first contribution in https://github.com/vllm-project/vllm/pull/39526\r\n* @EricccYang made their first contribution in https://github.com/vllm-project/vllm/pull/37376\r\n* @evezhier made their first contribution in https://github.com/vllm-project/vllm/pull/36540\r\n* @ezylopx5 made their first contribution in https://github.com/vllm-project/vllm/pull/37051\r\n* @fergusfinn made their first contribution in https://github.com/vllm-project/vllm/pull/35745\r\n* @foreverlms made their first contribution in https://github.com/vllm-project/vllm/pull/31113\r\n* @frgossen made their first contribution in https://github.com/vllm-project/vllm/pull/38944\r\n* @Galigator made their first contribution in https://github.com/vllm-project/vllm/pull/40161\r\n* @ganeshr10 made their first contribution in https://github.com/vllm-project/vllm/pull/32662\r\n* @hangy-amd made their first contribution in https://github.com/vllm-project/vllm/pull/39703\r\n* @heachary made their first contribution in https://github.com/vllm-project/vllm/pull/39999\r\n* @hhk7734 made their first contribution in https://github.com/vllm-project/vllm/pull/37171\r\n* @hnt2601 made their first contribution in https://github.com/vllm-project/vllm/pull/39892\r\n* @hospedales made their first contribution in https://github.com/vllm-project/vllm/pull/38847\r\n* @huangzhilin-hzl made their first contribution in https://github.com/vllm-project/vllm/pull/40092\r\n* @ianliuy made their first contribution in https://github.com/vllm-project/vllm/pull/39473\r\n* @ibifrost made their first contribution in https://github.com/vllm-project/vllm/pull/37636\r\n* @ibrahim1023 made their first contribution in https://github.com/vllm-project/vllm/pull/39169\r\n* @ichbinblau made their first contribution in https://github.com/vllm-project/vllm/pull/38371\r\n* @jackcfwang made their first contribution in https://github.com/vllm-project/vllm/pull/38794\r\n* @jaseelmohd2 made their first contribution in https://github.com/vllm-project/vllm/pull/39986\r\n* @jatseng-ai made their first contribution in https://github.com/vllm-project/vllm/pull/37352\r\n* @JeanPaulShapo made their first contribution in https://github.com/vllm-project/vllm/pull/35736\r\n* @jefp made their first contribution in https://github.com/vllm-project/vllm/pull/39435\r\n* @jesus-talavera-ibm made their first contribution in https://github.com/vllm-project/vllm/pull/38714\r\n* @jigangz made their first contribution in https://github.com/vllm-project/vllm/pull/39780\r\n* @JoursBleu made their first contribution in https://github.com/vllm-project/vllm/pull/36965\r\n* @khairulkabir1661 made their first contribution in https://github.com/vllm-project/vllm/pull/38388\r\n* @khushali9 made their first contribution in https://github.com/vllm-project/vllm/pull/40409\r\n* @kibitzing made their first contribution in https://github.com/vllm-project/vllm/pull/37501\r\n* @KimuGenie made their first contribution in https://github.com/vllm-project/vllm/pull/39679\r\n* @kkyyxhll made their first contribution in https://github.com/vllm-project/vllm/pull/38517\r\n* @kot-begemot-uk made their first contribution in https://github.com/vllm-project/vllm/pull/36487\r\n* @krishung5 made their first contribution in https://github.com/vllm-project/vllm/pull/39502\r\n* @KyleMylonakisProtopia made their first contribution in https://github.com/vllm-project/vllm/pull/38699\r\n* @lalit10 made their first contribution in https://github.com/vllm-project/vllm/pull/38955\r\n* @larryli2-amd made their first contribution in https://github.com/vllm-project/vllm/pull/39616\r\n* @lesj0610 made their first contribution in https://github.com/vllm-project/vllm/pull/40359\r\n* @liuchenbing2026 made their first contribution in https://github.com/vllm-project/vllm/pull/37512\r\n* @lyd1992 made their first contribution in https://github.com/vllm-project/vllm/pull/40428\r\n* @MekayelAnik made their first contribution in https://github.com/vllm-project/vllm/pull/39085\r\n* @menogrey made their first contribution in https://github.com/vllm-project/vllm/pull/37989\r\n* @mieshkiwrk made their first contribution in https://github.com/vllm-project/vllm/pull/38825\r\n* @misaAle made their first contribution in https://github.com/vllm-project/vllm/pull/39554\r\n* @Monishver11 made their first contribution in https://github.com/vllm-project/vllm/pull/32996\r\n* @mukesh-hai made their first contribution in https://github.com/vllm-project/vllm/pull/38435\r\n* @namgyu-youn made their first contribution in https://github.com/vllm-project/vllm/pull/38799\r\n* @nemanjaudovic made their first contribution in https://github.com/vllm-project/vllm/pull/38114\r\n* @nithinvc made their first contribution in https://github.com/vllm-project/vllm/pull/38405\r\n* @noobHappylife made their first contribution in https://github.com/vllm-project/vllm/pull/38519\r\n* @pedramr made their first contribution in https://github.com/vllm-project/vllm/pull/39650\r\n* @petern48 made their first contribution in https://github.com/vllm-project/vllm/pull/37247\r\n* @philip-essential made their first contribution in https://github.com/vllm-project/vllm/pull/39823\r\n* @pinsiangamd made their first contribution in https://github.com/vllm-project/vllm/pull/37529\r\n* @Prathmesh234 made their first contribution in https://github.com/vllm-project/vllm/pull/36466\r\n* @puririshi98 made their first contribution in https://github.com/vllm-project/vllm/pull/39206\r\n* @qiching made their first contribution in https://github.com/vllm-project/vllm/pull/39752\r\n* @qmx made their first contribution in https://github.com/vllm-project/vllm/pull/35687\r\n* @rbrugaro-amd made their first contribution in https://github.com/vllm-project/vllm/pull/39242\r\n* @rishaps made their first contribution in https://github.com/vllm-project/vllm/pull/39092\r\n* @Roy214 made their first contribution in https://github.com/vllm-project/vllm/pull/39575\r\n* @San-Nguyen made their first contribution in https://github.com/vllm-project/vllm/pull/40324\r\n* @SandishKumarHN made their first contribution in https://github.com/vllm-project/vllm/pull/35431\r\n* @SeraphimSerapis made their first contribution in https://github.com/vllm-project/vllm/pull/39861\r\n* @ShubyM made their first contribution in https://github.com/vllm-project/vllm/pull/38844\r\n* @shunting314 made their first contribution in https://github.com/vllm-project/vllm/pull/36298\r\n* @skavulya made their first contribution in https://github.com/vllm-project/vllm/pull/40430\r\n* @starkwj made their first contribution in https://github.com/vllm-project/vllm/pull/38726\r\n* @stevenkuang-tencent made their first contribution in https://github.com/vllm-project/vllm/pull/40681\r\n* @storyicon made their first contribution in https://github.com/vllm-project/vllm/pull/40133\r\n* @talorabr made their first contribution in https://github.com/vllm-project/vllm/pull/36029\r\n* @thomasmaindron made their first contribution in https://github.com/vllm-project/vllm/pull/39293\r\n* @TihoElek made their first contribution in https://github.com/vllm-project/vllm/pull/38849\r\n* @triangleXIV made their first contribution in https://github.com/vllm-project/vllm/pull/39102\r\n* @ultranationalism made their first contribution in https://github.com/vllm-project/vllm/pull/40191\r\n* @USTCKAY made their first contribution in https://github.com/vllm-project/vllm/pull/39181\r\n* @V2arK made their first contribution in https://github.com/vllm-project/vllm/pull/38016\r\n* @vedantjh2 made their first contribution in https://github.com/vllm-project/vllm/pull/34539\r\n* @velonica0 made their first contribution in https://github.com/vllm-project/vllm/pull/39478\r\n* @vibhavagarwal5 made their first contribution in https://github.com/vllm-project/vllm/pull/39064\r\n* @VinayakMishra95 made their first contribution in https://github.com/vllm-project/vllm/pull/40729\r\n* @Wangxiaoxiaoa made their first contribution in https://github.com/vllm-project/vllm/pull/40455\r\n* @wincent8 made their first contribution in https://github.com/vllm-project/vllm/pull/37841\r\n* @wojciech-wais made their first contribution in https://github.com/vllm-project/vllm/pull/34844\r\n* @wufann made their first contribution in https://github.com/vllm-project/vllm/pull/38615\r\n* @wuyingjun-lucky made their first contribution in https://github.com/vllm-project/vllm/pull/40251\r\n* @YifanLi3 made their first contribution in https://github.com/vllm-project/vllm/pull/40266\r\n* @yintong-lu made their first contribution in https://github.com/vllm-project/vllm/pull/35697\r\n* @YM2132 made their first contribution in https://github.com/vllm-project/vllm/pull/38427\r\n* @yoke233 made their first contribution in https://github.com/vllm-project/vllm/pull/38909\r\n* @yubofredwang made their first contribution in https://github.com/vllm-project/vllm/pull/39160\r\n* @yurun00 made their first contribution in https://github.com/vllm-project/vllm/pull/37766\r\n* @yuwenzho made their first contribution in https://github.com/vllm-project/vllm/pull/39466\r\n* @Yuyi-Ao made their first contribution in https://github.com/vllm-project/vllm/pull/38052\r\n* @z1ying made their first contribution in https://github.com/vllm-project/vllm/pull/39518\r\n* @zhangj1an made their first contribution in https://github.com/vllm-project/vllm/pull/40629\r\n* @Zhenzhong1 made their first contribution in https://github.com/vllm-project/vllm/pull/38192\r\n* @zxd1997066 made their first contribution in https://github.com/vllm-project/vllm/pull/38899\r\n","reactions":{"url":"https://api.github.com/repos/vllm-project/vllm/releases/312943348/reactions","total_count":140,"+1":60,"-1":0,"laugh":3,"hooray":15,"confused":0,"heart":10,"rocket":50,"eyes":2},"mentions_count":123},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/310648088","assets_url":"https://api.github.com/repos/vllm-project/vllm/releases/310648088/assets","upload_url":"https://uploads.github.com/repos/vllm-project/vllm/releases/310648088/assets{?name,label}","html_url":"https://github.com/vllm-project/vllm/releases/tag/v0.19.1","id":310648088,"author":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"node_id":"RE_kwDOI7xefs4ShB0Y","tag_name":"v0.19.1","target_commitish":"main","name":"v0.19.1","draft":false,"immutable":false,"prerelease":false,"created_at":"2026-04-18T00:57:52Z","updated_at":"2026-04-21T00:46:27Z","published_at":"2026-04-18T05:44:42Z","assets":[{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/399171685","id":399171685,"node_id":"RA_kwDOI7xefs4XyuBl","name":"vllm-0.19.1+cpu-cp38-abi3-manylinux_2_35_aarch64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":33518983,"digest":"sha256:daee2faece6401948dd4e9a57a4692e7b2c4af7374f520361daf7720dce2572a","download_count":17141,"created_at":"2026-04-18T05:44:46Z","updated_at":"2026-04-18T05:44:53Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.19.1/vllm-0.19.1%2Bcpu-cp38-abi3-manylinux_2_35_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/399171683","id":399171683,"node_id":"RA_kwDOI7xefs4XyuBj","name":"vllm-0.19.1+cpu-cp38-abi3-manylinux_2_35_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":71780104,"digest":"sha256:b082c1540bb015f880486e03a741419d30bf1ae9598f03ae40eb7d4323fad3e9","download_count":35282,"created_at":"2026-04-18T05:44:46Z","updated_at":"2026-04-18T05:44:59Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.19.1/vllm-0.19.1%2Bcpu-cp38-abi3-manylinux_2_35_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/399171686","id":399171686,"node_id":"RA_kwDOI7xefs4XyuBm","name":"vllm-0.19.1+cu130-cp38-abi3-manylinux_2_35_aarch64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":213520623,"digest":"sha256:3b9f46aecd13c6a6230db2006bb9d515dd352d95b37ebe8b43ead00d810927a8","download_count":3170,"created_at":"2026-04-18T05:44:46Z","updated_at":"2026-04-18T05:45:21Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.19.1/vllm-0.19.1%2Bcu130-cp38-abi3-manylinux_2_35_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/399171682","id":399171682,"node_id":"RA_kwDOI7xefs4XyuBi","name":"vllm-0.19.1+cu130-cp38-abi3-manylinux_2_35_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":227920553,"digest":"sha256:927fafd1e72c868109746d98dae0936bb367786bebd7debaaa3acd8885aa4a20","download_count":52326,"created_at":"2026-04-18T05:44:46Z","updated_at":"2026-04-18T05:45:26Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.19.1/vllm-0.19.1%2Bcu130-cp38-abi3-manylinux_2_35_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/399171684","id":399171684,"node_id":"RA_kwDOI7xefs4XyuBk","name":"vllm-0.19.1-cp38-abi3-manylinux_2_31_aarch64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":385252556,"digest":"sha256:c8dde3c9af20f00a644e64a50ebe43948f2921bab3ffd5407d634c15836cb181","download_count":342,"created_at":"2026-04-18T05:44:46Z","updated_at":"2026-04-18T05:45:35Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.19.1/vllm-0.19.1-cp38-abi3-manylinux_2_31_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/399171762","id":399171762,"node_id":"RA_kwDOI7xefs4XyuCy","name":"vllm-0.19.1-cp38-abi3-manylinux_2_31_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":433132101,"digest":"sha256:71a87f46cafab4489c69a5c5c83b870d0235e5694d8222303d460576293dc719","download_count":3643,"created_at":"2026-04-18T05:44:55Z","updated_at":"2026-04-18T05:45:38Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.19.1/vllm-0.19.1-cp38-abi3-manylinux_2_31_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/399171790","id":399171790,"node_id":"RA_kwDOI7xefs4XyuDO","name":"vllm-0.19.1.tar.gz","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/x-gtar","state":"uploaded","size":31105401,"digest":"sha256:9fb88ce6b50991eba41d183584f65f51d7f6015d86a42cdabf79c1c8bd5d66fa","download_count":2180,"created_at":"2026-04-18T05:45:01Z","updated_at":"2026-04-18T05:45:05Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.19.1/vllm-0.19.1.tar.gz"}],"tarball_url":"https://api.github.com/repos/vllm-project/vllm/tarball/v0.19.1","zipball_url":"https://api.github.com/repos/vllm-project/vllm/zipball/v0.19.1","body":"This is a patch release on top of `v0.19.0` with Transformers v5.5.3 upgrade and bug fixes for Gemma4:\r\n- Update to transformers v5 (#30566)\r\n- [Bugfix] Fix invalid JSON in Gemma 4 streaming tool calls by stripping partial delimiters (#38992)\r\n- [Bugfix][Frontend] Fix Gemma4 streaming HTML duplication after tool calls (#38909)\r\n- [Bugfix] Fix Gemma4 streaming tool call corruption for split boolean/number values (#39114)\r\n- [Tool] adjust_request to reasoning parser, and Gemma4 fixes (#39027)\r\n- [Gemma4] Support quantized MoE (#39045)\r\n- Add Gemma4 Eagle3 support (#39450)\r\n- [Gemma4][Bugfix]: Enable Gemma4ForCasualLM to load lora adapters correctly (#38844)\r\n- [Bugfix] Fix Gemma4 tool parser converting bare null to string \"null\" (#39679)\r\n- [Model] Fix Gemma 4 token repetition by dynamic BOS injection for PT models (#39842)\r\n- fix(kimi_k25): resolve media_placeholder_token_id from tokenizer (#39344)\r\n\r\n","reactions":{"url":"https://api.github.com/repos/vllm-project/vllm/releases/310648088/reactions","total_count":54,"+1":36,"-1":0,"laugh":0,"hooray":4,"confused":0,"heart":0,"rocket":14,"eyes":0}},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/304922777","assets_url":"https://api.github.com/repos/vllm-project/vllm/releases/304922777/assets","upload_url":"https://uploads.github.com/repos/vllm-project/vllm/releases/304922777/assets{?name,label}","html_url":"https://github.com/vllm-project/vllm/releases/tag/v0.19.0","id":304922777,"author":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"node_id":"RE_kwDOI7xefs4SLMCZ","tag_name":"v0.19.0","target_commitish":"main","name":"v0.19.0","draft":false,"immutable":false,"prerelease":false,"created_at":"2026-04-02T23:45:38Z","updated_at":"2026-04-04T20:16:12Z","published_at":"2026-04-03T02:19:12Z","assets":[{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/387525833","id":387525833,"node_id":"RA_kwDOI7xefs4XGSzJ","name":"vllm-0.19.0+cpu-cp38-abi3-manylinux_2_35_aarch64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":33505217,"digest":"sha256:e1dd943d486c9977af5e23966b178091034db24272865a5478621f76bdef9c4f","download_count":249,"created_at":"2026-04-03T02:19:16Z","updated_at":"2026-04-03T02:19:22Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.19.0/vllm-0.19.0%2Bcpu-cp38-abi3-manylinux_2_35_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/387525835","id":387525835,"node_id":"RA_kwDOI7xefs4XGSzL","name":"vllm-0.19.0+cpu-cp38-abi3-manylinux_2_35_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":71766406,"digest":"sha256:8cc674c5bf389535c699aa4ac3419fb2e384e44862ffba19a8aa48b89c4d3756","download_count":3386,"created_at":"2026-04-03T02:19:16Z","updated_at":"2026-04-03T02:19:29Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.19.0/vllm-0.19.0%2Bcpu-cp38-abi3-manylinux_2_35_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/387525838","id":387525838,"node_id":"RA_kwDOI7xefs4XGSzO","name":"vllm-0.19.0+cu130-cp38-abi3-manylinux_2_35_aarch64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":213217590,"digest":"sha256:05bb9756ca59067d338313771c0322f1d4ff6b151a62caaa47889dcc392d22e5","download_count":49860,"created_at":"2026-04-03T02:19:16Z","updated_at":"2026-04-03T02:19:52Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.19.0/vllm-0.19.0%2Bcu130-cp38-abi3-manylinux_2_35_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/387525822","id":387525822,"node_id":"RA_kwDOI7xefs4XGSy-","name":"vllm-0.19.0+cu130-cp38-abi3-manylinux_2_35_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":227555822,"digest":"sha256:84f26a82237ebbfdd892aec652c7dd890836edf0daf3c83bf9713beee82f3ab0","download_count":84750,"created_at":"2026-04-03T02:19:16Z","updated_at":"2026-04-03T02:19:55Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.19.0/vllm-0.19.0%2Bcu130-cp38-abi3-manylinux_2_35_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/387525834","id":387525834,"node_id":"RA_kwDOI7xefs4XGSzK","name":"vllm-0.19.0-cp38-abi3-manylinux_2_31_aarch64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":384650557,"digest":"sha256:6ab90ccca5d7ca3bd2c8f90133f0fac85e8f4af582a1c67c6cc3f63c615521e3","download_count":186,"created_at":"2026-04-03T02:19:16Z","updated_at":"2026-04-03T02:20:04Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.19.0/vllm-0.19.0-cp38-abi3-manylinux_2_31_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/387525923","id":387525923,"node_id":"RA_kwDOI7xefs4XGS0j","name":"vllm-0.19.0-cp38-abi3-manylinux_2_31_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":432281473,"digest":"sha256:2d0e5fae45367bdbf111fcad68f4c0f8fdddd2f2fb643e52f0f2daebef7b41cf","download_count":14381,"created_at":"2026-04-03T02:19:24Z","updated_at":"2026-04-03T02:20:08Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.19.0/vllm-0.19.0-cp38-abi3-manylinux_2_31_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/387526030","id":387526030,"node_id":"RA_kwDOI7xefs4XGS2O","name":"vllm-0.19.0.tar.gz","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/x-gtar","state":"uploaded","size":31071745,"digest":"sha256:81e59cf87175e7a62eb8d9acf5989484bbd17089d5eface353f89067bda282d9","download_count":2888,"created_at":"2026-04-03T02:19:31Z","updated_at":"2026-04-03T02:19:35Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.19.0/vllm-0.19.0.tar.gz"}],"tarball_url":"https://api.github.com/repos/vllm-project/vllm/tarball/v0.19.0","zipball_url":"https://api.github.com/repos/vllm-project/vllm/zipball/v0.19.0","body":"# vLLM v0.19.0\r\n\r\n## Highlights\r\nThis release features 448 commits from 197 contributors (54 new)!\r\n\r\n* **Gemma 4 support**: Full Google Gemma 4 architecture support including MoE, multimodal, reasoning, and tool-use capabilities (#38826, #38847). Requires `transformers>=5.5.0`. We recommend using pre-built docker image `vllm/vllm-openai:gemma4` for out of box usage.\r\n* **Zero-bubble async scheduling + speculative decoding**: Async scheduling now supports speculative decoding with zero-bubble overlap, significantly improving throughput (#32951).\r\n* **Model Runner V2 maturation**: MRV2 gains piecewise CUDA graphs for pipeline parallelism (#35162), spec decode rejection sampler with greedy/logprobs support (#37238, #37237), multi-modal embeddings for spec decode (#36097), streaming inputs (#37028), and EPLB support (#37488).\r\n* **ViT Full CUDA Graphs**: Vision encoders (ViT) now support full CUDA graph capture for reduced overhead (#35963).\r\n* **General CPU KV cache offloading**: A simple yet general CPU KV cache offloading mechanism for V1, with pluggable cache policy and block-level preemption handling (#37160, #37874, #34805, #36642, #37853).\r\n* **DBO (Dual-Batch Overlap) generalization**: The microbatch optimization (DBO) now works with general models, not just specific architectures (#37926).\r\n* **NVIDIA B300/GB300 (SM 10.3) support**: Allreduce fusion enabled by default with tuned all-reduce communicator (#37755, #37756).\r\n* **Transformers v5 compatibility**: Broad compatibility fixes across many models for HuggingFace Transformers v5 (#37681, #38127, #38090, #38247, #38410).\r\n\r\n### Model Support\r\n* New architectures: Gemma 4 (#38826), Cohere ASR (#35809), Cohere Transcribe (#38120), ColQwen3.5 4.5B (#36887), LFM2-ColBERT-350M (#37528), Granite 4.0 1B Speech (#38019), Qwen3-ForcedAligner (#35367).\r\n* Speculative decoding: Eagle3 for Pixtral (#37182), EagleMistralLarge3 fix (#37232).\r\n* LoRA expansion: H2OVL tower/connector LoRA (#31696), `--lora-target-modules` to restrict LoRA to specific modules (#34984), `language_model_only` respected (#37375), Mistral3 fix (#36928), Qwen3.5 fix (#36976), out-of-tree ops replacement (#37181).\r\n* Model fixes: NemotronH MTP + Chunked Prefill (#35447), Qwen3-VL video timestamps (#37439), Qwen3.5 GDN quantized models (#37448), Qwen3Next A_log FP32 (#37810), JAIS ALiBi (#37820), RoBERTa CUDA graph position IDs (#37873), AudioFlamingo3/MusicFlamingo (#37643), Music Flamingo loading (#35535), bge-m3 task selection (#37632), Nemotron Parse loading (#37407), GLM OCR patch merger (#37962), PaddleOCR checkpoint compat (#38232), DeepSeek v3.2 params (#33703), MiniMax NVFP4 weight loading (#37214), gated model HF token (#37920), Parakeet OOM on long audio (#36671).\r\n* Features: Temporal compression for Nemotron-3-VL videos (#36808), NemotronH Puzzle + MTP (#37803), torch.compile for InternVL vision encoder (#38049), multiple embedding types in single call (#35829).\r\n* Performance: GLM-4.xv ViT optimization (#37779).\r\n\r\n### Engine Core\r\n* **Zero-bubble async scheduling + speculative decoding** (#32951).\r\n* **Model Runner V2**: PP CUDA graphs (#35162), spec decode rejection sampler greedy (#37238) + logprobs (#37237), multimodal embeddings for spec decode (#36097), streaming inputs (#37028), configurable acceptance rate (#38045), FP32 draft logits (#37526), FP64 Gumbel noise (#37798), warmup with spec decode (#37812).\r\n* **ViT Full CUDA Graph** capture (#35963).\r\n* **General CPU KV cache offloading** with pluggable CachePolicy (#37160, #37874), block-level preemption (#34805), multiple KV groups (#36642), hybrid model support (#37853).\r\n* **DBO for general models**: Microbatch optimization generalized beyond specific architectures (#37926).\r\n* **Compilation**: Mega AOT artifact for torch 2.12+ (#37198), lazy graph module to defer recompile (#37609), remove model tag requirement for compile cache (#37345), Triton autotuning disk cache enabled by default (#37188), inductor runtime asserts disabled by default (#37485).\r\n* **FlexAttention**: Custom mask modification support (#37692).\r\n* **Attention**: Distinguish short extends vs decodes (#37303), allow qk_nope_head_dim=192 in FlashInfer MLA (#37475), skip sliding window attention layers with FP8 KV cache (#33695).\r\n* **Scheduling**: Schedule requests based on full input sequence length (#37307).\r\n* **Spec decode**: Per-draft-model MoE backend via `--speculative-config` (#37880), Eagle3 drafter quant_config propagation (#37280), Eagle3 norm_before_fc propagation (#38111).\r\n* **Extensibility**: PluggableLayer for CustomQwen2Decoder (#37293), tensor IPC transfer for multimodal data (#32104).\r\n* **Performance**: Optimize top-k in Triton sampler (#37225), optimize token_embed for pooling models with 1% improvement (#37347), fix slow hasattr in CUDAGraphWrapper (#37425), NFS prefetch auto-enabled with RAM guard (#37673), pybase64 replacement (#37290), optimize swap_states for hybrid models (#34733).\r\n* **Bugfixes**: Fix gibberish from FP8 MLA KV scale inconsistency (#37054), Mamba state corruption (#37728), deadlock with pause/resume (#37024), FlashInfer MNNVL socket collisions (#36674), multimodal prefix cache key collisions (#36708), DP coordinator ZMQ TOCTOU (#37452), CUDA graph memory double-counting (#37426), pooling non-determinism (#37775), AllReduce Fusion shutdown crash (#36955), FlashInfer allreduce workspace (#37461), async spec decoding with hybrid models (#38556), MLA sparse indexer prefill chunking (#36178), KV offloading + MLA (#37536), async scheduling extra CUDA context (#37449), DP MTP dummy run (#35243), offloading+prefetch for GLM-4.7-FP8 (#37178), max memory for multiple KV-cache groups (#36030).\r\n\r\n### Hardware & Performance\r\n* **NVIDIA**:\r\n  * B300/GB300 (SM 10.3): Allreduce fusion enabled by default (#37755), tuned all-reduce communicator (#37756).\r\n  * Blackwell: Optimized SM120 CUTLASS blockwise FP8 GEMM (#37970), fix NVFP4 NaN on desktop Blackwell (#37725), fix DeepGEMM E8M0 accuracy for Qwen3.5 FP8 (#38083), restore FP8 FlashMLA CUDA graph persistent buffers (#35175), DGX Spark fix (#38126).\r\n  * FlashInfer sparse MLA as default for FP8 KV cache (#37252).\r\n  * Tuned prefill configs for FP8 FA3 (#36265), tuned Triton MoE config for Qwen3.5 on H200 with 9.9% E2E improvement (#37340), H800 MoE configs (#31201).\r\n  * GPT-OSS: Router GEMM kernel (#37205), eliminate padding with FlashInfer MXFP4/MXFP8 MoE (#30647), reduce redundant SparseMatrix creation (#37683).\r\n  * NVFP4 CUTLASS MoE non-gated support (#37320), fuse pack topk in TRTLLM MoE via torch.compile (#37695).\r\n  * Non-contiguous KV cache in TRTLLM FP8 dequant kernel (#36867), Qwen3 dual stream input projection (#36795).\r\n* **AMD ROCm**:\r\n  * ROCm 7.2.1, torch 2.10, triton 3.6 (#38252).\r\n  * DeepEP as all2all backend (#34692).\r\n  * Persistent MLA kernel from AITER (#36574), FP8xFP8 attention in AITER (#36927).\r\n  * AWQ Marlin support (#36505), wvSplitK skinny GEMM for RDNA4/gfx1x (#34709).\r\n  * Nightly Docker image and wheel releases (#37283).\r\n  * Bugfixes: Sleep mode memory leak (#37533), hybrid model stride (#37228), qwen3_next crash (#36795).\r\n* **Intel XPU**: MLA model support (#37143), CompressedTensor W4A8 (#37207), auto-detect XPU build platform (#37634).\r\n* **TPU**: Async scheduling interface (#36924), Qwen3.5 FP8 weight loading fix (#37348).\r\n* **CPU**: Enable tcmalloc by default (#37607), graceful degradation without tcmalloc/libiomp (#37561), 48.9% throughput improvement for pooling models (#38139), OpenMP thread fix for torch.compile (#37538), structured output crash fix (#37706), KV cache block zeroing crash fix (#37550), slot mapping kernel (#37987), W4A16 compressed tensors (#38219).\r\n* **Performance fixes**: FP8 DeepGEMM batch invariance (#37718), Triton autotuning for Qwen3.5 (#37338), TRTLLM NVFP4 routing precision (#36725).\r\n\r\n### Large Scale Serving\r\n* **Disaggregated serving**: PD kv_transfer_params for Anthropic Messages (#37535) and Responses API (#37424), Mooncake heterogeneous TP (#36869), Mamba N-1 prefill for P/D (#37310).\r\n* **EPLB**: MRV2 support (#37488), improved responsiveness (#36271), EP weight filter fix (#37322).\r\n* **Elastic EP**: Fix repeated scale up/down cycles (#37131), fix stateless group port races (#36330).\r\n* **DBO**: Generalized to work with all models (#37926).\r\n* **Multi-node**: Fix allreduce fusion (#38136).\r\n* **KV connector**: Plugin-overridable metadata build (#37336).\r\n* **Constraints**: Cap API servers to 1 with Elastic EP (#37466).\r\n\r\n### Quantization\r\n* **Online MXFP8** quantization for MoE and dense models (#35448).\r\n* **FP8**: WoQ kernel abstraction (#32929), Marlin FP8 for compressed tensors fix (#38092).\r\n* **NVFP4**: Rescale weight scales to fix BF16 dequant underflow (#34577), fix Marlin NaN/Inf with float16 (#33972).\r\n* **QeRL**: Online quantization composed with quantized reloading for RLHF (#38032).\r\n* **CPU**: W4A16 compressed tensors (#38219).\r\n* **XPU**: CompressedTensor W4A8 (#37207).\r\n* **ROCm**: AWQ Marlin support (#36505).\r\n* **MXFP8 + DeepGEMM**: Fix crash when both are active (#37358).\r\n* **Removals**: Per-tensor-per-channel FP8 removed (#32700), Sparse24 integration and kernels removed (#36799).\r\n\r\n### API & Frontend\r\n* **New endpoints**: `/v1/chat/completions/batch` for batched chat completions (#38011).\r\n* **Features**: Limit thinking tokens (hard limit) (#20859), multiple embedding types in single call (#35829), numpy array embeddings for multimodal (#38119), `--lora-target-modules` (#34984), `-sc` shorthand for `--speculative-config` (#38380).\r\n* **Tool parsing**: GigaChat 3.1 parser (#36664), Kimi-K2.5 reasoning/tool parser (#37438), Gemma 4 tool parser (#38847), tools passed to parser constructor (#38029), fix Mistral parser (#37209), fix DeepSeek v3.2 streaming (#36056), fix GLM-4.7 parsing (#37386), fix Hermes streaming (#38168), fix OpenAI tool parser IndexError (#37958), fix Anthropic streaming (#37510).\r\n* **Responses API**: Fix crash with tool_choice=required exceeding max_output_tokens (#37258), fix TTFT recording (#37498), fix Anthropic serving template kwargs (#37899).\r\n* **Performance**: Offload blocking tokenizer ops to thread pool (#34789).\r\n* **Deprecations**: `--calculate-kv-scales` (#37201), `score` task (#37537), pooling multi-task support (#37956), `reasoning_content` message field removed (#37480).\r\n* **Bugfixes**: Embed/classify task routing (#37573), Cohere embed task instruction (#38362), renderer workers restricted to 1 with MM cache (#38418).\r\n* **UX**: Log once per node by default (#37568), torch profiler with stack enabled (#37571).\r\n\r\n### Security\r\n* Add `VLLM_MAX_N_SEQUENCES` environment variable to enforce sequence limits (#37952).\r\n* Enforce frame limit in VideoMediaIO to prevent resource exhaustion (#38636).\r\n\r\n### Dependencies\r\n* Transformers v5 compatibility across many models (#37681, #38127, #38247, #38410, #38090).\r\n* ROCm 7.2.1, torch 2.10, triton 3.6 for ROCm builds (#38252).\r\n* compressed-tensors bumped to 0.14.0.1 (#36988).\r\n* Python OpenAI package bumped (#32316).\r\n* flashinfer-cubin added as default CUDA dependency (#37233).\r\n* librosa removed from audio dependencies (#37058).\r\n\r\n### V0 Deprecation\r\n* Deprecate virtual engine (#37195).\r\n* Deprecate `--disable-frontend-multiprocessing` (#37612).\r\n* Refactor KV cache from list to element (#37487).\r\n\r\n### New Contributors\r\n* @aaab8b made their first contribution in #37533\r\n* @aasgaonkar made their first contribution in #35386\r\n* @allgather made their first contribution in #38410\r\n* @avinashsingh77 made their first contribution in #37100\r\n* @b-mu made their first contribution in #35963\r\n* @bongwoobak made their first contribution in #37424\r\n* @brandonpelfrey made their first contribution in #32104\r\n* @ccrhx4 made their first contribution in #37634\r\n* @cdpath made their first contribution in #37510\r\n* @cemigo114 made their first contribution in #37064\r\n* @cnyvfang made their first contribution in #37439\r\n* @DanBlanaru made their first contribution in #37307\r\n* @DorBernsohn made their first contribution in #37438\r\n* @dsingal0 made their first contribution in #37923\r\n* @fxdawnn made their first contribution in #36038\r\n* @grYe99 made their first contribution in #38074\r\n* @guillaumeguy made their first contribution in #38119\r\n* @gxd3 made their first contribution in #36924\r\n* @he-yufeng made their first contribution in #37301\r\n* @javierdejesusda made their first contribution in #37920\r\n* @jetxa made their first contribution in #37899\r\n* @jhsmith409 made their first contribution in #37448\r\n* @jrplatin made their first contribution in #37348\r\n* @kjiang249 made their first contribution in #37475\r\n* @laudney made their first contribution in #34709\r\n* @lcskrishna made their first contribution in #34692\r\n* @li-liwen made their first contribution in #38108\r\n* @Liangyx2 made their first contribution in #37523\r\n* @MatejRojec made their first contribution in #38011\r\n* @Nekofish-L made their first contribution in #37970\r\n* @pjo256 made their first contribution in #34733\r\n* @r266-tech made their first contribution in #37820\r\n* @RobTand made their first contribution in #37725\r\n* @scyyh11 made their first contribution in #34789\r\n* @SherryC41 made their first contribution in #37519\r\n* @shwetha-s-poojary made their first contribution in #31696\r\n* @siewcapital made their first contribution in #36955\r\n* @SKPsanjeevi made their first contribution in #36574\r\n* @thillai-c made their first contribution in #37231\r\n* @tianrengao made their first contribution in #34389\r\n* @tmm77 made their first contribution in #37694\r\n* @utsumi-fj made their first contribution in #38328\r\n* @vineetatiwari27 made their first contribution in #37998\r\n* @Wangbei25 made their first contribution in #37293\r\n* @WindChimeRan made their first contribution in #35007\r\n* @wjhrdy made their first contribution in #37706\r\n* @XLiu-2000 made their first contribution in #37371\r\n* @xueliangyang-oeuler made their first contribution in #37536\r\n* @yanghui1-arch made their first contribution in #37873\r\n* @yassha made their first contribution in #37369\r\n* @yeahdongcn made their first contribution in #37840\r\n* @Young-Leo made their first contribution in #37565\r\n* @ZeldaHuang made their first contribution in #37425\r\n* @zhejiangxiaomai made their first contribution in #37259\r\n\r\n","reactions":{"url":"https://api.github.com/repos/vllm-project/vllm/releases/304922777/reactions","total_count":49,"+1":15,"-1":0,"laugh":0,"hooray":30,"confused":0,"heart":0,"rocket":4,"eyes":0},"mentions_count":53},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/303469345","assets_url":"https://api.github.com/repos/vllm-project/vllm/releases/303469345/assets","upload_url":"https://uploads.github.com/repos/vllm-project/vllm/releases/303469345/assets{?name,label}","html_url":"https://github.com/vllm-project/vllm/releases/tag/v0.18.1","id":303469345,"author":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"node_id":"RE_kwDOI7xefs4SFpMh","tag_name":"v0.18.1","target_commitish":"main","name":"v0.18.1","draft":false,"immutable":false,"prerelease":false,"created_at":"2026-03-30T19:42:26Z","updated_at":"2026-03-31T05:59:49Z","published_at":"2026-03-31T00:53:26Z","assets":[{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/385170144","id":385170144,"node_id":"RA_kwDOI7xefs4W9Trg","name":"vllm-0.18.1+cpu-cp38-abi3-manylinux_2_35_aarch64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":33245934,"digest":"sha256:3725291eef5df7b98ecd900a80da16af35a7b92c2ed02a556349401ab3f5585d","download_count":86,"created_at":"2026-03-31T00:53:30Z","updated_at":"2026-03-31T00:53:36Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.18.1/vllm-0.18.1%2Bcpu-cp38-abi3-manylinux_2_35_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/385170142","id":385170142,"node_id":"RA_kwDOI7xefs4W9Tre","name":"vllm-0.18.1+cpu-cp38-abi3-manylinux_2_35_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":71445384,"digest":"sha256:47e7b2e7f3d0716f828baba053d1e2b05a0d28ac13f4c6d6f15d0a5331c0a539","download_count":3377,"created_at":"2026-03-31T00:53:30Z","updated_at":"2026-03-31T00:53:44Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.18.1/vllm-0.18.1%2Bcpu-cp38-abi3-manylinux_2_35_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/385170140","id":385170140,"node_id":"RA_kwDOI7xefs4W9Trc","name":"vllm-0.18.1+cu130-cp38-abi3-manylinux_2_35_aarch64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":213984259,"digest":"sha256:7674d8f7af20c8d8733a4b3cffdfccad5ca1c1dfb0eeb44c2fd177ccfffae369","download_count":1330,"created_at":"2026-03-31T00:53:30Z","updated_at":"2026-03-31T00:54:06Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.18.1/vllm-0.18.1%2Bcu130-cp38-abi3-manylinux_2_35_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/385170143","id":385170143,"node_id":"RA_kwDOI7xefs4W9Trf","name":"vllm-0.18.1+cu130-cp38-abi3-manylinux_2_35_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":228315650,"digest":"sha256:365a34b96031261173eaa24f8618303f2266e6c8dc904ce0440a08e37ecda0c0","download_count":5115,"created_at":"2026-03-31T00:53:30Z","updated_at":"2026-03-31T00:54:06Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.18.1/vllm-0.18.1%2Bcu130-cp38-abi3-manylinux_2_35_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/385170139","id":385170139,"node_id":"RA_kwDOI7xefs4W9Trb","name":"vllm-0.18.1-cp38-abi3-manylinux_2_31_aarch64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":385590570,"digest":"sha256:7e9e16639114760919ca86b74e8ac4be78e2593cd44ff387477e000a25ab5569","download_count":101,"created_at":"2026-03-31T00:53:30Z","updated_at":"2026-03-31T00:54:19Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.18.1/vllm-0.18.1-cp38-abi3-manylinux_2_31_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/385170286","id":385170286,"node_id":"RA_kwDOI7xefs4W9Ttu","name":"vllm-0.18.1-cp38-abi3-manylinux_2_31_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":433216393,"digest":"sha256:33d6d81299dedd45dde43fef261fbad2d513ed80e7cf7e64fb5880e8cf8ea105","download_count":568,"created_at":"2026-03-31T00:53:38Z","updated_at":"2026-03-31T00:54:19Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.18.1/vllm-0.18.1-cp38-abi3-manylinux_2_31_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/385170329","id":385170329,"node_id":"RA_kwDOI7xefs4W9TuZ","name":"vllm-0.18.1.tar.gz","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/x-gtar","state":"uploaded","size":30814301,"digest":"sha256:8d18eff1c4ed21eb8cf7a45a1d1b34752d5ec287e79186e451d5c2f797cacdb7","download_count":242,"created_at":"2026-03-31T00:53:46Z","updated_at":"2026-03-31T00:53:49Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.18.1/vllm-0.18.1.tar.gz"}],"tarball_url":"https://api.github.com/repos/vllm-project/vllm/tarball/v0.18.1","zipball_url":"https://api.github.com/repos/vllm-project/vllm/zipball/v0.18.1","body":"This is a patch release on top of v0.18.0 to address a few issues:\r\n- Change default SM100 MLA prefill backend back to TRT-LLM (#38562)\r\n- Fix mock.patch resolution failure for standalone_compile.FakeTensorMode on Python <= 3.10 (#37158)\r\n- Disable monolithic TRTLLM MoE for Renormalize routing #37605\r\n- Pre-download missing FlashInfer headers in Docker build #38391\r\n- Fix DeepGemm E8M0 accuracy degradation for Qwen3.5 FP8 on Blackwell (#38083)\r\n","reactions":{"url":"https://api.github.com/repos/vllm-project/vllm/releases/303469345/reactions","total_count":14,"+1":14,"-1":0,"laugh":0,"hooray":0,"confused":0,"heart":0,"rocket":0,"eyes":0}},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/299609267","assets_url":"https://api.github.com/repos/vllm-project/vllm/releases/299609267/assets","upload_url":"https://uploads.github.com/repos/vllm-project/vllm/releases/299609267/assets{?name,label}","html_url":"https://github.com/vllm-project/vllm/releases/tag/v0.18.0","id":299609267,"author":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"node_id":"RE_kwDOI7xefs4R26yz","tag_name":"v0.18.0","target_commitish":"main","name":"v0.18.0","draft":false,"immutable":false,"prerelease":false,"created_at":"2026-03-19T22:06:38Z","updated_at":"2026-03-20T22:19:03Z","published_at":"2026-03-20T21:31:36Z","assets":[{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/378187206","id":378187206,"node_id":"RA_kwDOI7xefs4Wiq3G","name":"vllm-0.18.0+cpu-cp38-abi3-manylinux_2_35_aarch64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":33245362,"digest":"sha256:102e0f66edeb6174d5e72ba9e1f085105aa65b6a92201c8d0ffa53d88511be89","download_count":95,"created_at":"2026-03-20T21:31:48Z","updated_at":"2026-03-20T21:31:54Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.18.0/vllm-0.18.0%2Bcpu-cp38-abi3-manylinux_2_35_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/378187205","id":378187205,"node_id":"RA_kwDOI7xefs4Wiq3F","name":"vllm-0.18.0+cpu-cp38-abi3-manylinux_2_35_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":71444812,"digest":"sha256:481ea5eb01971a492bf7aff9eaf921dc4004eb1a370bc625b53ecadea3305adb","download_count":2013,"created_at":"2026-03-20T21:31:48Z","updated_at":"2026-03-20T21:32:00Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.18.0/vllm-0.18.0%2Bcpu-cp38-abi3-manylinux_2_35_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/378187202","id":378187202,"node_id":"RA_kwDOI7xefs4Wiq3C","name":"vllm-0.18.0+cu130-cp38-abi3-manylinux_2_35_aarch64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":213983591,"digest":"sha256:2befefb9b24b9302d7c7fbc94769b0a0ab0f034a2722d07cb6a957f8ca3a00d1","download_count":953,"created_at":"2026-03-20T21:31:48Z","updated_at":"2026-03-20T21:32:23Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.18.0/vllm-0.18.0%2Bcu130-cp38-abi3-manylinux_2_35_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/378187207","id":378187207,"node_id":"RA_kwDOI7xefs4Wiq3H","name":"vllm-0.18.0+cu130-cp38-abi3-manylinux_2_35_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":228316984,"digest":"sha256:a2db9c6cd2e1166876ba00b0b586b975a6be395fd53e4447e2bd00d518d358fb","download_count":3893,"created_at":"2026-03-20T21:31:48Z","updated_at":"2026-03-20T21:32:25Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.18.0/vllm-0.18.0%2Bcu130-cp38-abi3-manylinux_2_35_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/378187204","id":378187204,"node_id":"RA_kwDOI7xefs4Wiq3E","name":"vllm-0.18.0-cp38-abi3-manylinux_2_31_aarch64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":385589729,"digest":"sha256:66a2c5bcf1bdf8de3e63b9fee067754068108cd510c65ffba70ff4368c33cba8","download_count":76,"created_at":"2026-03-20T21:31:48Z","updated_at":"2026-03-20T21:32:37Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.18.0/vllm-0.18.0-cp38-abi3-manylinux_2_31_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/378187274","id":378187274,"node_id":"RA_kwDOI7xefs4Wiq4K","name":"vllm-0.18.0-cp38-abi3-manylinux_2_31_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":433215727,"digest":"sha256:0bc51491598f4bcd161b693b27cbe2864082d6c49fa9065965d94b371f6ae8ef","download_count":2115,"created_at":"2026-03-20T21:31:56Z","updated_at":"2026-03-20T21:32:38Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.18.0/vllm-0.18.0-cp38-abi3-manylinux_2_31_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/378187332","id":378187332,"node_id":"RA_kwDOI7xefs4Wiq5E","name":"vllm-0.18.0.tar.gz","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/x-gtar","state":"uploaded","size":30812817,"digest":"sha256:9a1bee091db8dbb4664a2a09cd9c61912e9912a44af1ce12b8593a231d05971c","download_count":202,"created_at":"2026-03-20T21:32:01Z","updated_at":"2026-03-20T21:32:06Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.18.0/vllm-0.18.0.tar.gz"}],"tarball_url":"https://api.github.com/repos/vllm-project/vllm/tarball/v0.18.0","zipball_url":"https://api.github.com/repos/vllm-project/vllm/zipball/v0.18.0","body":"# vLLM v0.18.0\r\n\r\n## Known issues\r\n- Degraded accuracy when serving Qwen3.5 with FP8 KV cache on B200 (#37618)\r\n- If you previously ran into `CUBLAS_STATUS_INVALID_VALUE` and had to use a workaround in `v0.17.0`, you can reinstall `torch 2.10.0`. PyTorch published an updated wheel that addresses this bug.\r\n\r\n## Highlights\r\n\r\nThis release features 445 commits from 213 contributors (61 new)!\r\n\r\n* **gRPC Serving Support**: vLLM now supports gRPC serving via the new `--grpc` flag (#36169), enabling high-performance RPC-based serving alongside the existing HTTP/REST interface.\r\n* **GPU-less Render Serving**: New `vllm launch render` command (#36166, #34551) enables GPU-less preprocessing and rendering, allowing separation of multimodal preprocessing from GPU inference.\r\n* **NGram GPU Speculative Decoding**: NGram speculative decoding now runs on GPU and is compatible with the async scheduler (#29184), significantly reducing spec decode overhead.\r\n* **KV Cache Offloading Improvements**: Smart CPU offloading that stores only frequently-reused blocks (#35342), plus FlexKV as a new offloading backend (#34328) and support for multiple KV groups in offloading spec (#36610).\r\n* **Elastic Expert Parallelism Milestone 2**: NIXL-EP integration (#35627) enables dynamic GPU scaling for MoE experts, with new `--enable-ep-weight-filter` CLI option (#37351) for faster EP model loading.\r\n* **FlashInfer 0.6.6**: Updated FlashInfer dependency (#36768) with numerous performance and correctness improvements.\r\n* **Responses API Streaming Tool Calls**: The OpenAI Responses API now supports tool/function calling with streaming (#29947).\r\n* **Online Beam Search for ASR**: Beam search support for encoder/decoder models both offline (#36153) and online transcriptions (#36160).\r\n* **Ray No Longer a Default Dependency**: Ray has been removed as a default dependency (#36170) — install it explicitly if needed.\r\n\r\n### Model Support\r\n* **New architectures**: Sarvam MoE (#33942), OLMo Hybrid (#32550), HyperCLOVAX-SEED-Think-32B VLM (#31471), HyperCLOVAX-SEED-Think-14B (#37107), Kimi-Audio-7B-Instruct (#36127), ColPali late-interaction retrieval (#36818), ERNIE pooling models (#36385).\r\n* **Speculative decoding**: Eagle3 for Qwen3.5 (#36658), Eagle3 for Kimi K2.5 MLA (#36361), Eagle for Mistral Large 3 with dense layers (#36163).\r\n* **LoRA**: Whisper LoRA (#29856), FP8 LoRA dense kernel (#35242).\r\n* **Multimodal**: Online use_audio_in_video (#36319), audio extraction from MP4 for Nemotron Nano VL (#35539), audio transcription for MP4/M4A/WebM (#35109), expose media_io_kwargs at runtime (#34778), fast media preprocessing for Nano Nemotron VL (#35657).\r\n* **Compatibility**: Gemma/Gemma2 inputs_embeds (#36787), SigLIP/CLIP Transformers v5 (#37200), fused expert weights in Transformers backend (#36997).\r\n* **Performance**: Qwen3 Next fused GDN kernel (#35777), LFM2 tuned H100 MoE configs (#36699).\r\n* **Fixes**: DeepSeek-V3.2 tokenizer space stripping (#37004), Qwen3.5 tool calling (#36774), Qwen3-VL timestamp mismatch (#36136), Qwen3-Next TP>1 weight sharding (#36242), Qwen3-ASR torch.compile (#35869), MiniCPM-V audio inference (#36751), MiniCPM-O 4.5 ViT attention (#34127), routed experts for hybrid models (#35744), Qwen2.5-Omni/Qwen3-Omni multi-video audio_in_video (#37147), DeepSeek-OCR empty images crash (#36670).\r\n\r\n### Engine Core\r\n* **Model Runner V2**: Probabilistic rejection sampling for spec decode (#35461), pooling models (#36019), extensible CUDA graph dispatch (#35959), WhisperModelState (#35790), XD-RoPE (#36817), model_state CUDA graph capture (#36544).\r\n* **KV cache offloading**: Reuse-frequency-gated CPU stores (#35342), FlexKV offloading backend (#34328), multiple KV groups (#36610), async scheduling fix (#33881).\r\n* **Speculative decoding**: NGram GPU implementation with async scheduler (#29184), fused EAGLE step slot mapping (#33503).\r\n* **Performance**: Remove busy loop from idle buffer readers (#28053), 2.7% E2E throughput for pooling via worker-side maxsim (#36159), 3.2% via batched maxsim (#36710), CUDA graph memory accounting during profiling (#30515), checkpoint prefetch to OS page cache (#36012), InstantTensor weight loader (#36139), sporadic stall fix via pin_memory removal (#37006).\r\n* **Stability**: VLM concurrent throughput degradation fix (#36557), DP deadlock fix (#35194), DeepSeek V3.2 OOM during CG profiling (#36691), Ray DP startup crash (#36665), NCCL rank calculation fix (#36940), zero-init MLA output buffers for NaN prevention (#37442), CUDA OOM fix (#35594).\r\n* **Defaults**: Cascade attention disabled by default (#36318).\r\n* **Extensibility**: OOT linear method registration (#35981), custom collective ops registration for non-CUDA platforms (#34760).\r\n\r\n### Kernel\r\n* **FA4 for MLA prefill** (#34732).\r\n* **FlashInfer Sparse MLA**: FP8 KV cache support (#35891), CUDA graphs on ROCm (#35719), MTP lens > 1 on ROCm (#36681).\r\n* **TRTLLM FP8 MoE modular kernel** (#36307).\r\n* **FP8 KV cache for Triton MLA decode** (#34597).\r\n* **FlashInfer MoE A2A kernel** (#36022).\r\n* **Remove chunking from FusedMoE** for full batch processing (#34086).\r\n* **CustomOp FusedRMSNormGated** for torch.compile compatibility (#35877).\r\n* **Mamba2 SSD prefill Triton kernel** optimization (#35397).\r\n* **DeepSeek-V3.2**: Vectorized MLA query concat kernel (#34917), optimized FP8 KV cache gather for context parallel (#35290).\r\n* **320-dimension MLA head size** support (#36161).\r\n* **Packed recurrent fast path** for decode (#36596).\r\n* **EP scatter race condition** fix (#34991).\r\n\r\n### Hardware & Performance\r\n* **NVIDIA**: FA4 for MLA prefill (#34732), DeepSeek-V3.2 MLA kernel optimizations (#34917, #35290).\r\n* **AMD ROCm**: Sparse MLA CUDA graphs (#35719), MTP lens > 1 in Sparse MLA (#36681), MLA with nhead<16 + FP8 KV for TP=8 (#35850), RoPE+KV cache fusion for AITER FA (#35786), AITER MLA CPU sync avoidance (#35765), Quark W4A8 MXFP4/FP8 (#35316), gfx1152/gfx1153 Krackan support (#36499), fused_topk_bias AITER optimization (#36253), skinny GEMM improvements (#34304), DeepEP in ROCm Dockerfile (#36086), startup OOM fix (#36720).\r\n* **Intel XPU**: Model Runner V2 enabled (#36078), MLA Sparse backend for DeepSeek V3.2 (#33230), LoRA via torch.compile (#36962), block FP8 MoE fallback (#36458), deepseek_scaling_rope fused kernel (#36612).\r\n* **CPU**: aarch64 int8 matmul via OneDNN upgrade (#36147), AMD Zen CPU backend via zentorch (#35970).\r\n* **RISC-V**: CPU backend support (#36578).\r\n* **Performance**: 5% E2E improvement for PD disaggregation scheduling (#35781), packed recurrent decode fast path (#36596), pooling model maxsim 2.7%+3.2% throughput (#36159, #36710).\r\n* **torch.compile**: FakeTensors instead of real GPU tensors for single-size compilation (#36093), non-contiguous fused RMSNorm + group quant (#36551), stop lazy compiling (#35472).\r\n\r\n### Large Scale Serving\r\n* **Elastic EP Milestone 2**: NIXL-EP integration (#35627), `--enable-ep-weight-filter` for faster EP loading (#37351).\r\n* **PD Disaggregation**: ~5% scheduler overhead reduction (#35781), KV transfer fix with spec decode (#35158), P/D for hybrid SSM-FA models via NIXL (#36687), PP for multimodal models on Transformers backend (#37057).\r\n* **KV Connectors**: HMA + NIXL connector (#35758), FlexKV offloading (#34328), worker→scheduler metadata (#31964), All-to-All DCP backend (#34883).\r\n* **LMCache**: Fault tolerance mechanism (#36586), memory leak fix (#35931), race condition fix (#35831), TP size for MLA multi-reader locking (#36129).\r\n* **EP loading**: Skip non-local expert weights (#37136).\r\n\r\n### Quantization\r\n* **ModelOpt MXFP8 MoE** support (#35986).\r\n* **MXFP4 MoE routing simulation** override for accuracy (#33595).\r\n* **FP8 LoRA dense kernel** (#35242).\r\n* **ROCm**: Quark W4A8 MXFP4/FP8 for LinearLayer (#35316), compressed-tensors fix for DeepSeek-R1 on MI300x (#36247).\r\n* **Fixes**: MLA crash with AWQ/GPTQ quantized models (#34695), score layer quantization for reranker models (#35849), GLM-4.1V non-default quantization (#36321), FP8 k_scale/v_scale loading for Qwen3-MoE (#35656).\r\n\r\n### API & Frontend\r\n* **gRPC**: New `--grpc` flag for gRPC serving (#36169).\r\n* **GPU-less serving**: `vllm launch render` for preprocessing-only serving (#36166), `vllm launch` for GPU-less preprocessing (#34551).\r\n* **Responses API**: Streaming tool/function calling (#29947), reasoning item fixes (#34499, #36516).\r\n* **Anthropic API**: Accept redacted thinking blocks (#36992).\r\n* **ASR**: Online beam search transcriptions (#36160), offline beam search (#36153), audio transcription for MP4/M4A/WebM (#35109), realtime endpoint metrics (#35500).\r\n* **Tool calling**: Granite4 tool parser (#36827), Qwen3Coder anyOf double encoding fix (#36032).\r\n* **New options**: `--distributed-timeout-seconds` (#36047), `--attention-backend auto` (#35738), `reasoning_effort=none` (#36238), PyTorch profiler schedule (#35240).\r\n* **Cohere Embed v2 API** support (#37074).\r\n* **Azure Blob Storage** support for RunAI Model Streamer (#34614).\r\n* **Graceful shutdown** timeout for in-flight requests (#36666).\r\n* **Fixes**: tool_choice=required exceeding max_tokens crash (#36841), negative max_tokens with long prompts (#36789), concurrent classify/token_classify race (#36614), Anthropic billing header prefix cache miss (#36829), render endpoint crash for multimodal requests (#35684), xgrammar dtype mismatch on macOS CPU (#32384), minimax_m2 tool parser with stream interval > 1 (#35895).\r\n\r\n### Security\r\n* Respect user `trust_remote_code` setting in NemotronVL and KimiK25 (#36192).\r\n* Upgrade xgrammar for security fix (#36168).\r\n* Guard RLHF weight sync deserialization behind insecure serialization flag (#35928).\r\n\r\n### Dependencies\r\n* **FlashInfer 0.6.6** (#36768).\r\n* **Ray removed from default dependencies** (#36170).\r\n* `kaldi_native_fbank` made optional (#35996).\r\n* OpenAI dependency bounded to 2.24.0 (#36471).\r\n* Deprecated items from v0.18 removed (#36470, #36006).\r\n* Mistral common v10 (#36971).\r\n\r\n### Breaking Changes\r\n1. **Ray no longer a default dependency** — install explicitly if needed (#36170).\r\n2. **Deprecated items removed** — items deprecated in v0.18 have been removed (#36470, #36006).\r\n3. **Cascade attention disabled by default** (#36318).\r\n4. **swap_space parameter removed** (V0 deprecation, #36216).\r\n5. **Monolithic TRTLLM MoE disabled for renormalize routing** — late fix cherry-picked (#37591).\r\n\r\n\r\n## New Contributors 🎉\r\n\r\n* @11happy made their first contribution in https://github.com/vllm-project/vllm/pull/35481\r\n* @12010486 made their first contribution in https://github.com/vllm-project/vllm/pull/36782\r\n* @abhishkh made their first contribution in https://github.com/vllm-project/vllm/pull/32454\r\n* @AjAnubolu made their first contribution in https://github.com/vllm-project/vllm/pull/35976\r\n* @alvinttang made their first contribution in https://github.com/vllm-project/vllm/pull/36397\r\n* @amd-asalykov made their first contribution in https://github.com/vllm-project/vllm/pull/35093\r\n* @amd-lalithnc made their first contribution in https://github.com/vllm-project/vllm/pull/35970\r\n* @arlo-scitix made their first contribution in https://github.com/vllm-project/vllm/pull/36139\r\n* @benenzhu made their first contribution in https://github.com/vllm-project/vllm/pull/36253\r\n* @ChuanLi1101 made their first contribution in https://github.com/vllm-project/vllm/pull/35893\r\n* @cluster2600 made their first contribution in https://github.com/vllm-project/vllm/pull/34882\r\n* @cong-or made their first contribution in https://github.com/vllm-project/vllm/pull/36164\r\n* @daje0601 made their first contribution in https://github.com/vllm-project/vllm/pull/29856\r\n* @davzaman made their first contribution in https://github.com/vllm-project/vllm/pull/32441\r\n* @eellison made their first contribution in https://github.com/vllm-project/vllm/pull/35877\r\n* @fangyuchu made their first contribution in https://github.com/vllm-project/vllm/pull/35194\r\n* @feiqiangs made their first contribution in https://github.com/vllm-project/vllm/pull/34328\r\n* @fenypatel99 made their first contribution in https://github.com/vllm-project/vllm/pull/35240\r\n* @gambletan made their first contribution in https://github.com/vllm-project/vllm/pull/36402\r\n* @giulio-leone made their first contribution in https://github.com/vllm-project/vllm/pull/36937\r\n* @gkswns0531 made their first contribution in https://github.com/vllm-project/vllm/pull/35849\r\n* @grimulkan made their first contribution in https://github.com/vllm-project/vllm/pull/34597\r\n* @hai-meh-cs made their first contribution in https://github.com/vllm-project/vllm/pull/36684\r\n* @hasethuraman made their first contribution in https://github.com/vllm-project/vllm/pull/34614\r\n* @Hongbin10 made their first contribution in https://github.com/vllm-project/vllm/pull/36713\r\n* @jeonsworld made their first contribution in https://github.com/vllm-project/vllm/pull/34499\r\n* @jjmiao1 made their first contribution in https://github.com/vllm-project/vllm/pull/35994\r\n* @Kaonael made their first contribution in https://github.com/vllm-project/vllm/pull/36818\r\n* @ketyi made their first contribution in https://github.com/vllm-project/vllm/pull/36670\r\n* @KevinZonda made their first contribution in https://github.com/vllm-project/vllm/pull/36209\r\n* @leo-cf-tian made their first contribution in https://github.com/vllm-project/vllm/pull/36022\r\n* @lisperz made their first contribution in https://github.com/vllm-project/vllm/pull/34531\r\n* @mitre88 made their first contribution in https://github.com/vllm-project/vllm/pull/35933\r\n* @nkm-meta made their first contribution in https://github.com/vllm-project/vllm/pull/34760\r\n* @nvnbagrov made their first contribution in https://github.com/vllm-project/vllm/pull/35657\r\n* @rahul-sarvam made their first contribution in https://github.com/vllm-project/vllm/pull/33942\r\n* @royyhuang made their first contribution in https://github.com/vllm-project/vllm/pull/35931\r\n* @sbeurnier made their first contribution in https://github.com/vllm-project/vllm/pull/37006\r\n* @seanmamasde made their first contribution in https://github.com/vllm-project/vllm/pull/35109\r\n* @sergey-zinchenko made their first contribution in https://github.com/vllm-project/vllm/pull/35684\r\n* @shaunkotek made their first contribution in https://github.com/vllm-project/vllm/pull/36149\r\n* @shubhra made their first contribution in https://github.com/vllm-project/vllm/pull/36545\r\n* @simone-dotolo made their first contribution in https://github.com/vllm-project/vllm/pull/36000\r\n* @sladyn98 made their first contribution in https://github.com/vllm-project/vllm/pull/33503\r\n* @slin1237 made their first contribution in https://github.com/vllm-project/vllm/pull/36938\r\n* @SoluMilken made their first contribution in https://github.com/vllm-project/vllm/pull/36511\r\n* @Srinivasoo7 made their first contribution in https://github.com/vllm-project/vllm/pull/35342\r\n* @stecasta made their first contribution in https://github.com/vllm-project/vllm/pull/35871\r\n* @sungsooha made their first contribution in https://github.com/vllm-project/vllm/pull/34883\r\n* @SunMarc made their first contribution in https://github.com/vllm-project/vllm/pull/36896\r\n* @TQCB made their first contribution in https://github.com/vllm-project/vllm/pull/36165\r\n* @tunglinwood made their first contribution in https://github.com/vllm-project/vllm/pull/36127\r\n* @tusharshetty61 made their first contribution in https://github.com/vllm-project/vllm/pull/36243\r\n* @typer-J made their first contribution in https://github.com/vllm-project/vllm/pull/36578\r\n* @weiguangli-io made their first contribution in https://github.com/vllm-project/vllm/pull/35815\r\n* @wuxun-zhang made their first contribution in https://github.com/vllm-project/vllm/pull/33230\r\n* @XingLiu1 made their first contribution in https://github.com/vllm-project/vllm/pull/35197\r\n* @yanhong-lbh made their first contribution in https://github.com/vllm-project/vllm/pull/32550\r\n* @yitingw1 made their first contribution in https://github.com/vllm-project/vllm/pull/36612\r\n* @yuanheng-zhao made their first contribution in https://github.com/vllm-project/vllm/pull/36106\r\n* @zihaoanllm made their first contribution in https://github.com/vllm-project/vllm/pull/35973\r\n","reactions":{"url":"https://api.github.com/repos/vllm-project/vllm/releases/299609267/reactions","total_count":30,"+1":18,"-1":0,"laugh":0,"hooray":0,"confused":0,"heart":0,"rocket":12,"eyes":0},"mentions_count":61},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/295541535","assets_url":"https://api.github.com/repos/vllm-project/vllm/releases/295541535/assets","upload_url":"https://uploads.github.com/repos/vllm-project/vllm/releases/295541535/assets{?name,label}","html_url":"https://github.com/vllm-project/vllm/releases/tag/v0.17.1","id":295541535,"author":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"node_id":"RE_kwDOI7xefs4RnZsf","tag_name":"v0.17.1","target_commitish":"main","name":"v0.17.1","draft":false,"immutable":false,"prerelease":false,"created_at":"2026-03-11T09:51:18Z","updated_at":"2026-03-12T01:47:57Z","published_at":"2026-03-11T10:24:34Z","assets":[{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/371454539","id":371454539,"node_id":"RA_kwDOI7xefs4WI_JL","name":"vllm-0.17.1+cpu-cp38-abi3-manylinux_2_35_aarch64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":32945580,"digest":"sha256:574b14476bd5f761e7a9b6123d15e8bbce5990d608be3aac02acaf67cc530aee","download_count":58,"created_at":"2026-03-11T10:24:48Z","updated_at":"2026-03-11T10:24:56Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.17.1/vllm-0.17.1%2Bcpu-cp38-abi3-manylinux_2_35_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/371454548","id":371454548,"node_id":"RA_kwDOI7xefs4WI_JU","name":"vllm-0.17.1+cpu-cp38-abi3-manylinux_2_35_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":54386342,"digest":"sha256:53ada53812d798193f1fc9f29d915bd93e28632ff86ed0e97543c9b6435cbfe0","download_count":1821,"created_at":"2026-03-11T10:24:48Z","updated_at":"2026-03-11T10:24:57Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.17.1/vllm-0.17.1%2Bcpu-cp38-abi3-manylinux_2_35_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/371454547","id":371454547,"node_id":"RA_kwDOI7xefs4WI_JT","name":"vllm-0.17.1+cu130-cp38-abi3-manylinux_2_35_aarch64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":213773119,"digest":"sha256:b8957d4d5165a418cf4f853886d4d3bf9f3c2edaef2940a8ab11f5c58ebf37cb","download_count":25465,"created_at":"2026-03-11T10:24:48Z","updated_at":"2026-03-11T10:25:22Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.17.1/vllm-0.17.1%2Bcu130-cp38-abi3-manylinux_2_35_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/371454546","id":371454546,"node_id":"RA_kwDOI7xefs4WI_JS","name":"vllm-0.17.1+cu130-cp38-abi3-manylinux_2_35_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":228100955,"digest":"sha256:24ed093635dcde88351dc13147a7bcaf19228a50ef5b0ecd7631f7c7f7dc2d5d","download_count":36312,"created_at":"2026-03-11T10:24:48Z","updated_at":"2026-03-11T10:25:22Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.17.1/vllm-0.17.1%2Bcu130-cp38-abi3-manylinux_2_35_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/371454540","id":371454540,"node_id":"RA_kwDOI7xefs4WI_JM","name":"vllm-0.17.1-cp38-abi3-manylinux_2_31_aarch64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":385333057,"digest":"sha256:f04d63a94d0415b2323b0a0d3ab89a8d4d9bd346251ff60d47a7df679f7b3ff8","download_count":37,"created_at":"2026-03-11T10:24:48Z","updated_at":"2026-03-11T10:25:34Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.17.1/vllm-0.17.1-cp38-abi3-manylinux_2_31_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/371454633","id":371454633,"node_id":"RA_kwDOI7xefs4WI_Kp","name":"vllm-0.17.1-cp38-abi3-manylinux_2_31_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":432931666,"digest":"sha256:c52e892309532b4e51cb94d022c5e3c0087300cdb56e4645708601443299d871","download_count":551,"created_at":"2026-03-11T10:24:57Z","updated_at":"2026-03-11T10:25:35Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.17.1/vllm-0.17.1-cp38-abi3-manylinux_2_31_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/371454641","id":371454641,"node_id":"RA_kwDOI7xefs4WI_Kx","name":"vllm-0.17.1.tar.gz","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/x-gtar","state":"uploaded","size":30547577,"digest":"sha256:d26a95dcb92e2ff78ed4b48bff247d845b0c768edf6c0acf2401376a56c57b61","download_count":2617,"created_at":"2026-03-11T10:24:59Z","updated_at":"2026-03-11T10:25:04Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.17.1/vllm-0.17.1.tar.gz"}],"tarball_url":"https://api.github.com/repos/vllm-project/vllm/tarball/v0.17.1","zipball_url":"https://api.github.com/repos/vllm-project/vllm/zipball/v0.17.1","body":"This is a patch release on top of `v0.17.0` to address a few issues:\r\n- New Model: Nemotron 3 Super\r\n- Fix passing of activation_type to trtllm fused MoE NVFP4 and FP8 (#36017)\r\n- Fix/resupport nongated fused moe triton (#36412)\r\n- Re-enable EP for trtllm MoE FP8 backend (#36494)\r\n- [Mamba][Qwen3.5] Zero freed SSM cache blocks on GPU (#35219)\r\n- Fix TRTLLM Block FP8 MoE Monolithic (#36296)\r\n- [DSV3.2][MTP] Optimize Indexer MTP handling (#36723)","reactions":{"url":"https://api.github.com/repos/vllm-project/vllm/releases/295541535/reactions","total_count":49,"+1":37,"-1":0,"laugh":0,"hooray":12,"confused":0,"heart":0,"rocket":0,"eyes":0}},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/294153909","assets_url":"https://api.github.com/repos/vllm-project/vllm/releases/294153909/assets","upload_url":"https://uploads.github.com/repos/vllm-project/vllm/releases/294153909/assets{?name,label}","html_url":"https://github.com/vllm-project/vllm/releases/tag/v0.17.0","id":294153909,"author":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"node_id":"RE_kwDOI7xefs4RiG61","tag_name":"v0.17.0","target_commitish":"main","name":"v0.17.0","draft":false,"immutable":false,"prerelease":false,"created_at":"2026-03-06T21:04:15Z","updated_at":"2026-04-09T17:14:30Z","published_at":"2026-03-07T00:46:41Z","assets":[{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/368705313","id":368705313,"node_id":"RA_kwDOI7xefs4V-f8h","name":"vllm-0.17.0+cpu-cp38-abi3-manylinux_2_35_aarch64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":32941923,"digest":"sha256:8a406d55497bb89d9d97a605c454072d443e706e7e91b3232a54691ae3fe4521","download_count":76,"created_at":"2026-03-07T03:51:01Z","updated_at":"2026-03-07T03:51:07Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.17.0/vllm-0.17.0%2Bcpu-cp38-abi3-manylinux_2_35_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/368705315","id":368705315,"node_id":"RA_kwDOI7xefs4V-f8j","name":"vllm-0.17.0+cpu-cp38-abi3-manylinux_2_35_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":54382687,"digest":"sha256:59fea6107e0c75dc8907e3113abd60c4eab30d8551ba2e3e7dfe799289a6c163","download_count":1520,"created_at":"2026-03-07T03:51:01Z","updated_at":"2026-03-07T03:51:12Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.17.0/vllm-0.17.0%2Bcpu-cp38-abi3-manylinux_2_35_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/368705318","id":368705318,"node_id":"RA_kwDOI7xefs4V-f8m","name":"vllm-0.17.0+cu130-cp38-abi3-manylinux_2_35_aarch64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":213769448,"digest":"sha256:fb3ec08ba43c11f19b247fcf193d44cbf89a9a2e514d05cd60f6f43944fb1eff","download_count":11333,"created_at":"2026-03-07T03:51:02Z","updated_at":"2026-03-07T03:51:36Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.17.0/vllm-0.17.0%2Bcu130-cp38-abi3-manylinux_2_35_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/368705317","id":368705317,"node_id":"RA_kwDOI7xefs4V-f8l","name":"vllm-0.17.0+cu130-cp38-abi3-manylinux_2_35_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":228095156,"digest":"sha256:a59de5d7bde73b49b0c28a574ec0ab3ecd984ab76e097a15e2afa6639b75e517","download_count":6072,"created_at":"2026-03-07T03:51:02Z","updated_at":"2026-03-07T03:51:37Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.17.0/vllm-0.17.0%2Bcu130-cp38-abi3-manylinux_2_35_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/368705316","id":368705316,"node_id":"RA_kwDOI7xefs4V-f8k","name":"vllm-0.17.0-cp38-abi3-manylinux_2_31_aarch64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":385329399,"digest":"sha256:310fb82fe061ed75dceeb4aeb803cd8ee0d590337ec720f7abfb03a69314d710","download_count":26525,"created_at":"2026-03-07T03:51:01Z","updated_at":"2026-03-07T03:51:48Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.17.0/vllm-0.17.0-cp38-abi3-manylinux_2_31_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/368705341","id":368705341,"node_id":"RA_kwDOI7xefs4V-f89","name":"vllm-0.17.0-cp38-abi3-manylinux_2_31_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":432927988,"digest":"sha256:0296670a09d392ee43455d9bebf590d05a9bc2ebce5e25e2919222fc815158da","download_count":9877,"created_at":"2026-03-07T03:51:09Z","updated_at":"2026-03-07T03:51:45Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.17.0/vllm-0.17.0-cp38-abi3-manylinux_2_31_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/368705370","id":368705370,"node_id":"RA_kwDOI7xefs4V-f9a","name":"vllm-0.17.0.tar.gz","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/x-gtar","state":"uploaded","size":30541311,"digest":"sha256:b0b62e58ef4eb633ef371f2726976372cf6dfcb7ff2ea9ddf7194c1930d5629a","download_count":163,"created_at":"2026-03-07T03:51:13Z","updated_at":"2026-03-07T03:51:18Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.17.0/vllm-0.17.0.tar.gz"}],"tarball_url":"https://api.github.com/repos/vllm-project/vllm/tarball/v0.17.0","zipball_url":"https://api.github.com/repos/vllm-project/vllm/zipball/v0.17.0","body":"# vLLM v0.17.0\r\n\r\n**Known Issue**: If you are on CUDA 12.9+ and encounter a `CUBLAS_STATUS_INVALID_VALUE` error, this is caused by a CUDA library mismatch. To resolve, try one of the following:\r\n1. Remove the path to system CUDA shared library files (e.g. `/usr/local/cuda`) from `LD_LIBRARY_PATH`, or simply `unset LD_LIBRARY_PATH`.\r\n2. Install vLLM with `uv pip install vllm --torch-backend=auto`.\r\n3. Install vLLM with `pip install vllm --extra-index-url https://download.pytorch.org/whl/cu129` (change the CUDA version to match your system).\r\n\r\n## Highlights\r\n\r\nThis release features 699 commits from 272 contributors (48 new)!\r\n\r\n* **PyTorch 2.10 Upgrade**: This release upgrades to **PyTorch 2.10.0**, which is a breaking change for environment dependencies.\r\n* **FlashAttention 4 Integration**: vLLM now supports the **FlashAttention 4** backend (#32974), bringing next-generation attention performance.\r\n* **Model Runner V2 Maturation**: Model Runner V2 has reached a major milestone with **Pipeline Parallel** (#33960), **Decode Context Parallel** (#34179), **Eagle3 speculative decoding with CUDA graphs** (#35029, #35040), **pooling model support** (#35120), piecewise & mixed CUDA graph capture (#32771), DP+EP for spec decoding (#35294), and a new ModelState architecture. Design docs are now available (#35819).\r\n* **Qwen3.5 Model Family**: Full support for the **Qwen3.5** model family (#34110) featuring GDN (Gated Delta Networks), with FP8 quantization, MTP speculative decoding, and reasoning parser support.\r\n* **New `--performance-mode` Flag**: A new `--performance-mode {balanced, interactivity, throughput}` flag (#34936) simplifies performance tuning for common deployment scenarios.\r\n* **Anthropic API Compatibility**: Added support for **Anthropic thinking blocks** (#33671), **`count_tokens` API** (#35588), `tool_choice=none` (#35835), and streaming/image handling fixes.\r\n* **Weight Offloading V2 with Prefetching**: The weight offloader now **hides onloading latency via prefetching** (#29941), plus **selective CPU weight offloading** (#34535) and CPU offloading without pinned memory doubling (#32993).\r\n* **Elastic Expert Parallelism Milestone 2**: Initial support for elastic expert parallelism enabling dynamic GPU scaling for MoE models (#34861).\r\n* **Quantized LoRA Adapters**: Users can now load **quantized LoRA adapters** (e.g. QLoRA) directly (#30286).\r\n* **Transformers v5 Compatibility**: Extensive work to ensure compatibility with HuggingFace Transformers v5 across models and utilities.\r\n* **CPU release supports AVX2, AVX-512, VNNI, AVX512BF16, and AMX** (#35466). The multi-ISA CPU dispatcher was originally implemented by @MekayelAnik (https://github.com/dtrifiro/vllm/pull/9, merged December 22, 2025) in collaboration with Willy Hardy, and later reimplemented in C++ in #35466.\r\n\r\n### Model Support\r\n* **New architectures**: Qwen3.5 (#34110), COLQwen3 (#34398), ColModernVBERT (#34558), Ring 2.5 (#35102), skt/A.X-K1 (#32407), Ovis 2.6 (#34426), nvidia/llama-nemotron-embed-vl-1b-v2 (#35297), nvidia/llama-nemotron-rerank-vl-1b-v2 (#35735), nvidia/nemotron-colembed (#34574).\r\n* **ASR models**: FunASR (#33247), FireRedASR2 (#35727), Qwen3-ASR realtime streaming (#34613).\r\n* **Multimodal**: OpenPangu-VL video input (#34134), audio chunking for offline LLM (#34628), Parakeet audio encoder for nemotron-nano-vl (#35100), MiniCPM-o flagos (#34126).\r\n* **LoRA**: LFM2 (#34921), Llama 4 Vision tower/connector (#35147), max vocab size increased to 258048 (#34773), quantized LoRA adapters (#30286).\r\n* **Task expansion**: ColBERT extended to non-standard BERT backbones (#34170), multimodal scoring for late-interaction models (#34574).\r\n* **Performance**: Qwen3.5 GDN projector fusion (#34697), FlashInfer cuDNN backend for Qwen3 VL ViT (#34580), Step3.5-Flash NVFP4 (#34478), Qwen3MoE tuned configs for H200 (#35457).\r\n* **Fixes**: DeepSeek-VL V2 simplified loading (#35203), Qwen3/Qwen3.5 reasoning parser (#34779), Qwen2.5-Omni/Qwen3-Omni mixed-modality (#35368), Ernie4.5-VL garbled output (#35587), Qwen-VL tokenizer (#36140), Qwen-Omni audio cache (#35994), Nemotron-3-Nano NVFP4 accuracy with TP>1 (#34476).\r\n\r\n### Engine Core\r\n* **Model Runner V2**: Pipeline Parallel (#33960), Decode Context Parallel (#34179), piecewise & mixed CUDA graphs (#32771), Eagle3 with CUDA graphs (#35029, #35040), pooling models (#35120), DP+EP for spec decoding (#35294), bad_words sampling (#33433), ModelState architecture (#35350, #35383, #35564, #35621, #35774), design docs (#35819).\r\n* **Weight offloading**: V2 prefetching to hide latency (#29941), selective CPU weight offloading (#34535), CPU offloading without pinned memory doubling (#32993).\r\n* **Sleep level 0** mode with enqueue/wait pattern (#33195), pause/resume moved into engine (#34125).\r\n* **Fixes**: allreduce_rms_fusion disabled by default with PP > 1 (#35424), DCP + FA3 crash (#35082), prefix caching for Mamba \"all\" mode (#34874), num_active_loras fix (#34119), async TP reduce-scatter reduction fix (#33088).\r\n* Repetitive token pattern detection flags (#35451).\r\n\r\n### Kernel\r\n* **FlashAttention 4** integration (#32974).\r\n* **FlashInfer Sparse MLA** backend (#33451).\r\n* **Triton-based top-k and top-p** sampler kernels (#33538).\r\n* Faster topKperRow decode kernel for DeepSeek-V3.2 sparse attention (#33680).\r\n* Optimized grouped topk kernel (#34206).\r\n* TRTLLM DSV3 Router GEMM kernel, **6% batch-1 speedup** (#34302).\r\n* FA3 swizzle optimization (#34043).\r\n* 256-bit LDG/STG activation kernels (#33022).\r\n* TMA support for fused_moe_lora kernel (#32195).\r\n* **Helion kernel framework**: silu_mul_fp8 kernel (#33373), autotuning infrastructure (#34025), num_tokens autotuning (#34185), fx tracing via HOP (#34390), GPU variant canonicalization (#34928).\r\n* FlashInfer TRTLLM fused MoE non-gated FP8 & NVFP4 (#33506).\r\n* Optimized sample_recovered_tokens kernel (#34974).\r\n* KV cache update ops extraction from FlashInfer forward (#35422) and MLA backends (#34627).\r\n\r\n### Hardware & Performance\r\n* **NVIDIA**: SM100 FMHA FP8 prefill for MLA (#31195), SM100 MXFP8 blockscaled grouped MM and quant kernels (#34448), SM100 Oink RMSNorm path (#31828), SM120 FP8 GEMM optimization (#34424), FlashInfer DeepGEMM swapAB on SM90 by default (#34924), DeepSeek R1 BF16 min latency QKV GEMM 0.5% E2E speedup (#34758), Cublas BF16 gate with FP32 output (#35121), FlashInfer All Reduce default to TRTLLM backend (#35793).\r\n* **AMD ROCm**: AITER fused RoPE+KVCache (#33443), MXFP4 MoE weight pre-shuffling on gfx950 (#34192), bitsandbytes quantization (#34688), CK backend for MoE quantization (#34301), dynamic MXFP4 for DeepSeek V2 (#34157), GPT-OSS Quark format (#29008), GPT-OSS WMXFP4_AFP8 static scales (#30357), encoder/encoder-decoder on AITER (#35334), device capability derivation without CUDA init (#35069), `aiter` package renamed to `amd-aiter` (#35198).\r\n* **Intel XPU**: CUDA graph support (#34482), GPUDirect RDMA via NIXL (#35270), TORCH_SDPA/TRITON_ATTN as ViT backend (#35010), vllm-xpu-kernels v0.1.3 (#35984).\r\n* **CPU**: ARM BF16 cross-compilation (#33079), FP16 for s390x (#34116), KleidiAI INT8_W4A8 for all input dtypes (#34890), s390x vector intrinsics for attention (#34434), prefix caching for ppc64le (#35081), CPU release supports both AVX2 and AVX512 (#35466).\r\n* **Performance**: Pipeline Parallel async send/recv 2.9% E2E throughput (#33368), pooling maxsim **13.9% throughput improvement** (#35330), Triton ViT attention backend (#32183), Mamba1 kernel-level chunk alignment for prefix caching (#34798), detokenizer optimization (#32975), pooling model copy optimization 1.8% throughput (#35127).\r\n\r\n### Large Scale Serving\r\n* **Pipeline Parallel** async send/recv, **2.9% throughput improvement** (#33368).\r\n* **Elastic EP Milestone 2** (#34861).\r\n* **EPLB**: Async rebalance algorithm (#30888), sync enforcement for NCCL backend (#35212).\r\n* **Native weight syncing API** via IPC for RL workflows (#34171).\r\n* Decode Context Parallel in Model Runner V2 (#34179).\r\n* Ray env var propagation to workers (#34383).\r\n* **Breaking**: KV load failure policy default changed from \"recompute\" to \"fail\" (#34896).\r\n* Cross-node data parallelism message queue fix (#35429).\r\n* NIXL: Token-based IPC API (#34175), version bound (#35495), NUMA core binding (#32365).\r\n\r\n### Speculative Decoding\r\n* **Nemotron-H MTP** and Mamba speculative decoding (#33726).\r\n* **Eagle3** on Model Runner V2 with CUDA graphs (#35029, #35040), Eagle3 + disaggregated serving (#34529).\r\n* Hidden states extraction system (#33736).\r\n* `min_tokens` support with speculative decoding (#32642).\r\n* Reduced TP communication for draft generation (#34049).\r\n* MTP num_speculative_tokens > 1 with sparse MLA (#34552).\r\n* Sparse MLA + MTP with full CUDA graphs (#34457).\r\n* Spec decoding in Mamba cache align mode (#33705).\r\n* DP+EP for spec decoding in Model Runner V2 (#35294).\r\n\r\n### MoE Refactor\r\n* **MoERunner abstraction** (#32344) with modular kernel architecture.\r\n* MXFP4 Cutlass Experts to modular kernel (#34542), MXFP4 Marlin to modular kernel format (#34588), TRTLLM Kernels MK (#32564).\r\n* MoEActivation enum (#33843).\r\n* Improved default Triton fused MoE configs (#34846).\r\n* Fused MoE + LoRA shared expert dual stream, **1.07x throughput** (#34933).\r\n* DSV3 QKVAProj GEMM custom op for torch.compile (#35751).\r\n* Fix routing for models without expert groups (MiniMax-M2.1) (#34673).\r\n\r\n### torch.compile\r\n* **AOT compile** with PyTorch 2.10 (#34155).\r\n* **AR+RMSNorm fusion** by default at -O2 (#34299).\r\n* **SiLU+FP4 quant fusion** by default at O1+ (#34718).\r\n* Sequence parallelism threshold compile ranges (#28672).\r\n* Various compile fixes: recursive pre_grad_passes (#34092), FakeTensorProp elimination (#34093), time discrepancy logging (#34912), artifact load errors (#35115), atomic artifact saving (#35117), pytree slice caching (#35308), fast_moe_cold_start undo for torch>=2.11 (#35475).\r\n\r\n### Quantization\r\n* **Quantized LoRA adapters** (#30286).\r\n* **Per-head KV cache scales** in attention selector (#34281).\r\n* FP8 MoE bias for GPT-OSS (#34906).\r\n* SM100 MXFP8 blockscaled grouped MM and quant kernels (#34448).\r\n* Mixed precision support for ModelOpt (#35047).\r\n* Llama-4 attention quantization (int8, fp8) (#34243).\r\n* Sparse24 compressed tensors fix (#33446).\r\n* KV scale loading fix for MLA models (#35430).\r\n* Compressed tensors as ground-truth for quant strategies (#34254).\r\n* **AMD**: CK backend for MoE (#34301), dynamic MXFP4 for DeepSeek V2 (#34157), bitsandbytes on ROCm (#34688), GPT-OSS Quark format (#29008).\r\n* **CPU**: KleidiAI INT8_W4A8 for all input dtypes (#34890).\r\n* **Qwen3.5**: FP8 weight loading fix (#35289), mlp.gate not quantizable (#35156).\r\n* int4_w4a16 fused_moe benchmark and tuning (#34130).\r\n* FlashInfer integrate mm_mxfp8 in ModelOpt MXFP8 (#35053).\r\n\r\n### API & Frontend\r\n* **Anthropic API**: Thinking blocks (#33671), count_tokens (#35588), tool_choice=none (#35835), tool call streaming fix (#34887), base64 image handling (#35557).\r\n* **Responses API**: Structured outputs (#33709), reasoning_tokens fix (#33513), reasoning_part streaming events (#35184).\r\n* **UX**: `--performance-mode {balanced, interactivity, throughput}` (#34936), `--moe-backend` for explicit kernel selection (#33807), `--language-model-only` for hybrid models (#34120), `--enforce-eager` clarification (#34523).\r\n* Whisper automatic language detection (#34342).\r\n* MFU Prometheus counters (#30950).\r\n* Unrecognized environment variable warnings (#33581).\r\n* `generation_config` max_tokens treated as default not ceiling (#34063).\r\n* Structured output bugfix for completions (#35237).\r\n* Structured output JSON feature validation (#33233).\r\n* Validate non-text content in system messages (#34072).\r\n* Explicit validation error for tool calls (#34438).\r\n* IO Processor plugin simplification (#34236).\r\n* Sparse embedding IO process plugin (#34214).\r\n* Pooling entrypoint improvements (#35604).\r\n\r\n### Security\r\n* Fix SSRF bypass via backslash-@ URL parsing inconsistency (#34743).\r\n\r\n### Dependencies\r\n* **PyTorch 2.10.0 upgrade** — breaking change requiring environment updates. ROCm torch also updated to official 2.10 release (#34387).\r\n* OpenTelemetry libraries included by default (#34466).\r\n* Bound NIXL upper bound version (#35495).\r\n* mooncake-transfer-engine added to kv_connectors requirements (#34826).\r\n* openai bounded to under 2.25.0.\r\n* lm-eval bumped for Transformers v5 compatibility (#33994).\r\n* mamba-ssm bumped for Transformers v5 (#34233).\r\n* PyPI source distribution (sdist) now included (#35136).\r\n* amd-quark package added for ROCm (#35658).\r\n\r\n### V0 Deprecation\r\n* Removed per-request logits processors (#34400).\r\n* Removed unused MM placeholders in request output (#34944).\r\n* Removed Swin model (#35821).\r\n* Scheduled v0.17 deprecations applied (#35441).\r\n\r\n### Transformers v5 Compatibility\r\n* Model fixes: Qwen3VL (#34262), JAIS (#34264), MiniCPM-V, GLM-ASR, Qwen3.5.\r\n* Xet high-performance mode (#35098).\r\n* Custom processor import fixes (#35101, #35107).\r\n* padding_index removal for compatibility (#35189).\r\n* lm-eval (#33994) and mamba-ssm (#34233) version bumps.\r\n\r\n## New Contributors 🎉\r\n\r\n* @2ez4bz made their first contribution in https://github.com/vllm-project/vllm/pull/33607\r\n* @Alibaba-HZY made their first contribution in https://github.com/vllm-project/vllm/pull/35289\r\n* @aykoppol made their first contribution in https://github.com/vllm-project/vllm/pull/35451\r\n* @bhoomit made their first contribution in https://github.com/vllm-project/vllm/pull/34773\r\n* @charlesashby made their first contribution in https://github.com/vllm-project/vllm/pull/34169\r\n* @chengyinie made their first contribution in https://github.com/vllm-project/vllm/pull/35457\r\n* @EdalatiAli made their first contribution in https://github.com/vllm-project/vllm/pull/34448\r\n* @ehfd made their first contribution in https://github.com/vllm-project/vllm/pull/33992\r\n* @flutist made their first contribution in https://github.com/vllm-project/vllm/pull/35838\r\n* @fort726 made their first contribution in https://github.com/vllm-project/vllm/pull/32407\r\n* @fynnsu made their first contribution in https://github.com/vllm-project/vllm/pull/33736\r\n* @gante made their first contribution in https://github.com/vllm-project/vllm/pull/35281\r\n* @hallerite made their first contribution in https://github.com/vllm-project/vllm/pull/35834\r\n* @hujia177 made their first contribution in https://github.com/vllm-project/vllm/pull/34982\r\n* @itayalroy made their first contribution in https://github.com/vllm-project/vllm/pull/34861\r\n* @jasonozuzu-cohere made their first contribution in https://github.com/vllm-project/vllm/pull/34715\r\n* @jcaip made their first contribution in https://github.com/vllm-project/vllm/pull/35327\r\n* @jhaotingc made their first contribution in https://github.com/vllm-project/vllm/pull/34933\r\n* @jjmiao1 made their first contribution in https://github.com/vllm-project/vllm/pull/35994\r\n* @jonoillar made their first contribution in https://github.com/vllm-project/vllm/pull/34513\r\n* @koush made their first contribution in https://github.com/vllm-project/vllm/pull/33646\r\n* @lailoo made their first contribution in https://github.com/vllm-project/vllm/pull/35616\r\n* @Laurawly made their first contribution in https://github.com/vllm-project/vllm/pull/31828\r\n* @Li-Yongwen made their first contribution in https://github.com/vllm-project/vllm/pull/34336\r\n* @lichuang made their first contribution in https://github.com/vllm-project/vllm/pull/34679\r\n* @lin-shh made their first contribution in https://github.com/vllm-project/vllm/pull/35645\r\n* @majian4work made their first contribution in https://github.com/vllm-project/vllm/pull/35466\r\n* @ojhaanshika made their first contribution in https://github.com/vllm-project/vllm/pull/34986\r\n* @PatrykWo made their first contribution in https://github.com/vllm-project/vllm/pull/35307\r\n* @pi314ever made their first contribution in https://github.com/vllm-project/vllm/pull/35434\r\n* @pkousha made their first contribution in https://github.com/vllm-project/vllm/pull/33839\r\n* @pks made their first contribution in https://github.com/vllm-project/vllm/pull/35237\r\n* @qianlihuang made their first contribution in https://github.com/vllm-project/vllm/pull/32642\r\n* @simonreginis made their first contribution in https://github.com/vllm-project/vllm/pull/31025\r\n* @stakeswky made their first contribution in https://github.com/vllm-project/vllm/pull/35230\r\n* @SteadfastAsArt made their first contribution in https://github.com/vllm-project/vllm/pull/34888\r\n* @stingoChen made their first contribution in https://github.com/vllm-project/vllm/pull/35352\r\n* @sychen52 made their first contribution in https://github.com/vllm-project/vllm/pull/35047\r\n* @thepushkarp made their first contribution in https://github.com/vllm-project/vllm/pull/32114\r\n* @Tib-Gridello made their first contribution in https://github.com/vllm-project/vllm/pull/35423\r\n* @umut-polat made their first contribution in https://github.com/vllm-project/vllm/pull/35510\r\n* @voipmonitor made their first contribution in https://github.com/vllm-project/vllm/pull/35615\r\n* @wangxingran222 made their first contribution in https://github.com/vllm-project/vllm/pull/33088\r\n* @wenshuai-xiaomi made their first contribution in https://github.com/vllm-project/vllm/pull/34424\r\n* @wjabbour made their first contribution in https://github.com/vllm-project/vllm/pull/35672\r\n* @yashwantbezawada made their first contribution in https://github.com/vllm-project/vllm/pull/31057\r\n* @yoonsnowdev made their first contribution in https://github.com/vllm-project/vllm/pull/35382\r\n* @ZhongsJie made their first contribution in https://github.com/vllm-project/vllm/pull/35835\r\n* @MekayelAnik made their first contribution in https://github.com/vllm-project/vllm/pull/35466","reactions":{"url":"https://api.github.com/repos/vllm-project/vllm/releases/294153909/reactions","total_count":64,"+1":18,"-1":0,"laugh":0,"hooray":22,"confused":0,"heart":8,"rocket":16,"eyes":0},"mentions_count":49},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/285918914","assets_url":"https://api.github.com/repos/vllm-project/vllm/releases/285918914/assets","upload_url":"https://uploads.github.com/repos/vllm-project/vllm/releases/285918914/assets{?name,label}","html_url":"https://github.com/vllm-project/vllm/releases/tag/v0.16.0","id":285918914,"author":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"node_id":"RE_kwDOI7xefs4RCsbC","tag_name":"v0.16.0","target_commitish":"main","name":"v0.16.0","draft":false,"immutable":false,"prerelease":false,"created_at":"2026-02-25T04:30:22Z","updated_at":"2026-02-26T04:35:50Z","published_at":"2026-02-25T19:58:49Z","assets":[{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/362343599","id":362343599,"node_id":"RA_kwDOI7xefs4VmOyv","name":"vllm-0.16.0+cpu-cp38-abi3-manylinux_2_35_aarch64.whl","label":"","uploader":{"login":"ywang96","id":136131678,"node_id":"U_kgDOCB00Xg","avatar_url":"https://avatars.githubusercontent.com/u/136131678?v=4","gravatar_id":"","url":"https://api.github.com/users/ywang96","html_url":"https://github.com/ywang96","followers_url":"https://api.github.com/users/ywang96/followers","following_url":"https://api.github.com/users/ywang96/following{/other_user}","gists_url":"https://api.github.com/users/ywang96/gists{/gist_id}","starred_url":"https://api.github.com/users/ywang96/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/ywang96/subscriptions","organizations_url":"https://api.github.com/users/ywang96/orgs","repos_url":"https://api.github.com/users/ywang96/repos","events_url":"https://api.github.com/users/ywang96/events{/privacy}","received_events_url":"https://api.github.com/users/ywang96/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":32564365,"digest":"sha256:5a5e70c94f6a928c0b5843efc080507dea663a642311675cd1195c25774bcd5d","download_count":738,"created_at":"2026-02-25T19:58:56Z","updated_at":"2026-02-25T19:59:02Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.16.0/vllm-0.16.0%2Bcpu-cp38-abi3-manylinux_2_35_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/362343598","id":362343598,"node_id":"RA_kwDOI7xefs4VmOyu","name":"vllm-0.16.0+cpu-cp38-abi3-manylinux_2_35_x86_64.whl","label":"","uploader":{"login":"ywang96","id":136131678,"node_id":"U_kgDOCB00Xg","avatar_url":"https://avatars.githubusercontent.com/u/136131678?v=4","gravatar_id":"","url":"https://api.github.com/users/ywang96","html_url":"https://github.com/ywang96","followers_url":"https://api.github.com/users/ywang96/followers","following_url":"https://api.github.com/users/ywang96/following{/other_user}","gists_url":"https://api.github.com/users/ywang96/gists{/gist_id}","starred_url":"https://api.github.com/users/ywang96/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/ywang96/subscriptions","organizations_url":"https://api.github.com/users/ywang96/orgs","repos_url":"https://api.github.com/users/ywang96/repos","events_url":"https://api.github.com/users/ywang96/events{/privacy}","received_events_url":"https://api.github.com/users/ywang96/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":34396253,"digest":"sha256:a2aea8a95ff6b797f5d543f71a7ff48738b9351f76aab0f9825cb6dddc05cb43","download_count":1066,"created_at":"2026-02-25T19:58:56Z","updated_at":"2026-02-25T19:59:02Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.16.0/vllm-0.16.0%2Bcpu-cp38-abi3-manylinux_2_35_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/362343594","id":362343594,"node_id":"RA_kwDOI7xefs4VmOyq","name":"vllm-0.16.0+cu130-cp38-abi3-manylinux_2_35_aarch64.whl","label":"","uploader":{"login":"ywang96","id":136131678,"node_id":"U_kgDOCB00Xg","avatar_url":"https://avatars.githubusercontent.com/u/136131678?v=4","gravatar_id":"","url":"https://api.github.com/users/ywang96","html_url":"https://github.com/ywang96","followers_url":"https://api.github.com/users/ywang96/followers","following_url":"https://api.github.com/users/ywang96/following{/other_user}","gists_url":"https://api.github.com/users/ywang96/gists{/gist_id}","starred_url":"https://api.github.com/users/ywang96/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/ywang96/subscriptions","organizations_url":"https://api.github.com/users/ywang96/orgs","repos_url":"https://api.github.com/users/ywang96/repos","events_url":"https://api.github.com/users/ywang96/events{/privacy}","received_events_url":"https://api.github.com/users/ywang96/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":265448018,"digest":"sha256:8354ec3264fb8811741d9f9d48e7ba6250d7cfbb7bdeb5d9c4b9632ad8cde885","download_count":2820,"created_at":"2026-02-25T19:58:56Z","updated_at":"2026-02-25T19:59:33Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.16.0/vllm-0.16.0%2Bcu130-cp38-abi3-manylinux_2_35_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/362343597","id":362343597,"node_id":"RA_kwDOI7xefs4VmOyt","name":"vllm-0.16.0+cu130-cp38-abi3-manylinux_2_35_x86_64.whl","label":"","uploader":{"login":"ywang96","id":136131678,"node_id":"U_kgDOCB00Xg","avatar_url":"https://avatars.githubusercontent.com/u/136131678?v=4","gravatar_id":"","url":"https://api.github.com/users/ywang96","html_url":"https://github.com/ywang96","followers_url":"https://api.github.com/users/ywang96/followers","following_url":"https://api.github.com/users/ywang96/following{/other_user}","gists_url":"https://api.github.com/users/ywang96/gists{/gist_id}","starred_url":"https://api.github.com/users/ywang96/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/ywang96/subscriptions","organizations_url":"https://api.github.com/users/ywang96/orgs","repos_url":"https://api.github.com/users/ywang96/repos","events_url":"https://api.github.com/users/ywang96/events{/privacy}","received_events_url":"https://api.github.com/users/ywang96/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":280962334,"digest":"sha256:bda6ff19ead743fb30c6271cdeb7daf62d5bd5f7a53cb6c2e7d987d53ea3d49f","download_count":8557,"created_at":"2026-02-25T19:58:56Z","updated_at":"2026-02-25T19:59:32Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.16.0/vllm-0.16.0%2Bcu130-cp38-abi3-manylinux_2_35_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/362343593","id":362343593,"node_id":"RA_kwDOI7xefs4VmOyp","name":"vllm-0.16.0-cp38-abi3-manylinux_2_31_aarch64.whl","label":"","uploader":{"login":"ywang96","id":136131678,"node_id":"U_kgDOCB00Xg","avatar_url":"https://avatars.githubusercontent.com/u/136131678?v=4","gravatar_id":"","url":"https://api.github.com/users/ywang96","html_url":"https://github.com/ywang96","followers_url":"https://api.github.com/users/ywang96/followers","following_url":"https://api.github.com/users/ywang96/following{/other_user}","gists_url":"https://api.github.com/users/ywang96/gists{/gist_id}","starred_url":"https://api.github.com/users/ywang96/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/ywang96/subscriptions","organizations_url":"https://api.github.com/users/ywang96/orgs","repos_url":"https://api.github.com/users/ywang96/repos","events_url":"https://api.github.com/users/ywang96/events{/privacy}","received_events_url":"https://api.github.com/users/ywang96/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":460494130,"digest":"sha256:dfaa14846608fd229dda9d372e2ad3f13854fd09147c2ba36b40579cf3c03804","download_count":242,"created_at":"2026-02-25T19:58:56Z","updated_at":"2026-02-25T19:59:50Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.16.0/vllm-0.16.0-cp38-abi3-manylinux_2_31_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/362343670","id":362343670,"node_id":"RA_kwDOI7xefs4VmOz2","name":"vllm-0.16.0-cp38-abi3-manylinux_2_31_x86_64.whl","label":"","uploader":{"login":"ywang96","id":136131678,"node_id":"U_kgDOCB00Xg","avatar_url":"https://avatars.githubusercontent.com/u/136131678?v=4","gravatar_id":"","url":"https://api.github.com/users/ywang96","html_url":"https://github.com/ywang96","followers_url":"https://api.github.com/users/ywang96/followers","following_url":"https://api.github.com/users/ywang96/following{/other_user}","gists_url":"https://api.github.com/users/ywang96/gists{/gist_id}","starred_url":"https://api.github.com/users/ywang96/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/ywang96/subscriptions","organizations_url":"https://api.github.com/users/ywang96/orgs","repos_url":"https://api.github.com/users/ywang96/repos","events_url":"https://api.github.com/users/ywang96/events{/privacy}","received_events_url":"https://api.github.com/users/ywang96/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":508337437,"digest":"sha256:f066b2a2f8597a4a3ada8fbbfd122b59086864b2260ca42dc81bf9fb57af0c42","download_count":3640,"created_at":"2026-02-25T19:59:04Z","updated_at":"2026-02-25T19:59:51Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.16.0/vllm-0.16.0-cp38-abi3-manylinux_2_31_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/362343677","id":362343677,"node_id":"RA_kwDOI7xefs4VmOz9","name":"vllm-0.16.0.tar.gz","label":"","uploader":{"login":"ywang96","id":136131678,"node_id":"U_kgDOCB00Xg","avatar_url":"https://avatars.githubusercontent.com/u/136131678?v=4","gravatar_id":"","url":"https://api.github.com/users/ywang96","html_url":"https://github.com/ywang96","followers_url":"https://api.github.com/users/ywang96/followers","following_url":"https://api.github.com/users/ywang96/following{/other_user}","gists_url":"https://api.github.com/users/ywang96/gists{/gist_id}","starred_url":"https://api.github.com/users/ywang96/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/ywang96/subscriptions","organizations_url":"https://api.github.com/users/ywang96/orgs","repos_url":"https://api.github.com/users/ywang96/repos","events_url":"https://api.github.com/users/ywang96/events{/privacy}","received_events_url":"https://api.github.com/users/ywang96/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/x-gtar","state":"uploaded","size":29198913,"digest":"sha256:754d94c95528f0b6279a3f458cd9be92fa573cbe2092f8db768139f87578b365","download_count":276,"created_at":"2026-02-25T19:59:04Z","updated_at":"2026-02-25T19:59:09Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.16.0/vllm-0.16.0.tar.gz"}],"tarball_url":"https://api.github.com/repos/vllm-project/vllm/tarball/v0.16.0","zipball_url":"https://api.github.com/repos/vllm-project/vllm/zipball/v0.16.0","body":"# vLLM v0.16.0\r\nPlease note that this release was branch cut on Feb 8, so any features added to vLLM after that date is not included.\r\n\r\n## Highlights\r\n\r\nThis release features 440 commits from 203 contributors (7 new)!\r\n\r\n* **Async scheduling + Pipeline Parallelism** is now fully supported, delivering **30.8% E2E throughput improvement** and **31.8% TPOT improvement** (#32618).\r\n* **Realtime API**: A new WebSocket-based Realtime API enables streaming audio interactions (#33187), building on the Voxtral realtime infrastructure.\r\n* **RLHF workflow improvements**: Native NCCL-based weight syncing API (#31943), layerwise weight reloading for QeRL (#32133), and engine pause/resume with request preservation (#32351).\r\n* **Unified Parallel Drafting** for speculative decoding (#32887), plus spec decode now works with structured outputs (#33374) and penalty application in Model Runner V2 (#33251).\r\n* **Major XPU platform overhaul**: Deprecated IPEX in favor of vllm-xpu-kernels (#33379), adding MoE (#33659), MXFP4 MoE (#33679), WNA16 (#33973), scaled_mm (#34117), and FP8 MoE (#34202) support.\r\n\r\n### Model Support\r\n* New architectures: GLM-OCR with MTP (#33005), Qwen3-ASR (#33312), DeepSeek-OCR-2 (#33165), Intern-S1-Pro (#33636), MiniCPM-o 4.5 (#33431), openPangu7B-VL (#32449), NemotronHPuzzle heterogeneous (#32549), MusicFlamingo (#32696), FunAudioChat (#2), ColBERT late interaction (#33686), voyage-4-nano (#33720), GLM-5 (#34124).\r\n* Speculative decoding: EAGLE3 for Hunyuan/HunyuanVL (#33035), AFMoE (#33111), Mistral3 (#33939).\r\n* LoRA expansion: Gemma3 vision components (#32764), Nemotron-H MTP models (#32265), Qwen3 output embedding (#29816). Optimized fused MoE-LoRA kernel indexing (#32770, #32774), unpermute-aware fused MoE LoRA path (#32655), reduced kernel overhead for fewer active LoRAs with multiple CUDA graphs (#32005).\r\n* Features: Qwen3-Omni transcription (#29828), Mistral Large 3 with FlashInfer MoE (#33174), LFM2 SigLIP2 intermediate encoder layers (#33370), Qwen3-Omni/GLM-4.xV MRoPE positioning fixes (#33010, #33039), embedding input for disabled modalities (#32493).\r\n* Performance: GLM-4.7-GPTQ decode and MTP acceptance rate regression fix (#33771), DeepSeek V3.2 fast detokenization (#33855), DeepSeek V3.2 tokenizer fix (#33832), GLM-5 MTP accuracy fix (#34385).\r\n\r\n### Engine Core\r\n* Async scheduling + Pipeline Parallelism: Full support with 30.8% throughput improvement (#32618), optimized spec decode + async scheduling with 1.5% throughput improvement (#33612), deadlock fix for torchrun PP broadcast (#33701).\r\n* Speculative decoding: Unified Parallel Drafting (#32887), structured output support (#33374), penalty application in MRV2 (#33251), skip softmax for all-greedy rejection sampling (#32852), correctness fix for spec tokens with prefill chunks (#33652).\r\n* RLHF: Native NCCL weight syncing API (#31943), layerwise reloading for QeRL (#32133), engine pause/resume with request preservation (#32351).\r\n* Helion kernel framework: ConfigManager (#32740), kernel wrapper (#32964), kernel registry (#33203).\r\n* PluggableLayer: Applied to linear layers (#33152) and Mamba layers (#33660).\r\n* Batch invariance: Disable Cascade Attention (#32561), enable Triton attention (#33688).\r\n* Performance: Grammar bitmask H2D copy on separate stream (#33059), zero-copy GQA for multimodal and CPU (#33732), early-reject oversized MM requests (#33502), CPU memory leak fix from Request reference cycle in prefix caching (#34183).\r\n\r\n### Hardware & Performance\r\n* **NVIDIA**: FlashInfer TRTLLM BF16 MoE integration (#32954), SM100 INT4 W4A16 kernel (#32437), SM121 (DGX Spark) CUTLASS support (#33517), MNNVL protocol for GB series (#33540), FlashInfer MLA concat optimization (#31171), GDN attention layout optimization (#33291), DeepGEMM FP8 MLA performance (#33568), wvSplitK_fp8 performance (#33527, #33493), B200 MoE configs for Nemotron Nano (#32804), Super B200 TP2 (#33510), GLM 4.6 (#32958), Mamba selective scan tuning for B200 (#32873). Fix: DeepSeek R1 CUTLASS MLA on B200 (#33637), QK Norm+RoPE fusion on B200+FP8 (#33967), CUTLASS FP8 blockwise on SM103a (#32224).\r\n* **AMD ROCm**: QWEN3-NEXT FP8 tunings (#32042), AITER attention backend for Qwen3-Next (#32492), fused_add_rmsnorm_pad for GPT-OSS (#30976), Qwen3-Omni startup fix (#33077).\r\n* **Intel XPU**: Platform overhaul - deprecated IPEX, switched to vllm-xpu-kernels (#33379). New: unquantized MoE (#33659), MXFP4 MoE (#33679), WNA16 kernel (#33973), scaled_mm kernel (#34117), FP8 MoE (#34202).\r\n* **ARM CPU**: KleidiAI INT4 dynamic quant with BF16 activations (#33122), NEON BFMMLA BF16 paged attention (#32263), vectorization backend optimization (#30329), attention dispatch by head_dim alignment (#32161).\r\n* **IBM Z**: BF16 kernel type for s390x (#33788).\r\n* **torch.compile**: Stop compiling identical artifacts (#34003), MoE cold start optimization option (#33735), fix 32-bit indexing assumption (#33113), attention fusion pass fix (#33945).\r\n* **Performance**: Chat completion streaming optimization (#33782), ORJSONResponse for faster API responses (#33548), MoE permute optimization for CUTLASS FP8 (#32892), shared/routed overlap for latent MoE on Nemotron-H (#32790), FlashInfer autotune control flag (#34006).\r\n\r\n### Large Scale Serving\r\n* Disaggregated serving: Mooncake connector rework with bootstrap server (#31034), cross-layer KV cache layout at NIXL Connector V2 (#33339), delay freeing blocks for aborted async loads (#32255), async double-free fix (#33377), Ray multi-replica single-instance fix (#33604).\r\n* EPLB: Capture logical experts with router replay (#33013), DP metadata fix for dense models (#32739).\r\n* Metrics: KV offloading connector metrics (#27942), labeled prompt token metrics for P/D disaggregation (#33290).\r\n\r\n### Quantization\r\n* New: FP8 block quant for CompressedTensorsW8A16Fp8 (#33280), ModelOpt MXFP8 for dense models (#33786), NVFP4/FP8 on Turing GPUs (#33076), TP > 4 for FP4 Gemm (#31099).\r\n* Bugfixes: FP8 online quantization memory fix (#31914), asymmetric W4A16 (ConchLinear) for CT (#33200), DeepSeek V3.2 NVFP4 (#33932), LoRA FP8 (#33879), quantized Falcon-H1 model loading (#32728), quantized Mamba TP with n_groups=1 (#33257), CPU W8A8 with bias (#33582), CPU W8A8 3D input support (#33727).\r\n* **Deprecation**: Removed BitBlas (#32683) and Marlin 24 (#32688).\r\n\r\n### API & Frontend\r\n* **Realtime API**: WebSocket-based streaming API (#33187) with Voxtral realtime support.\r\n* **Responses API**: Sampling parameters (#32609), return token IDs (#33212), return prompt token IDs (#33378), parser implementation (#32712).\r\n* Pooling API: Request schema consensus for ScoreRequest (#33060) and final standardization (#31127).\r\n* Tool calling: Fix multi-turn tool call ID preservation (#32768), fix indexing double-counting (#33141), GLM-4 incremental string streaming (#33218), DSV3.2 fast detokenization fix (#33964), MCP tools non-streaming fix (#32762).\r\n* Structured outputs: Performance optimization with reasoning (#33557), guidance vocab size fix (#33509).\r\n* CLI: `--disable-access-log-for-endpoints` option (#30011).\r\n* UX: Nested configs in YAML files (#33193), GGUF `repo_id:quant_type` syntax (#33371), DeepSeek ReasoningParser with thinking enabled by default (#33221), remove noisy CT warning (#33273), early tokenization validation (#31366), reasoning_content backward compatibility (#33635), only include Authorization header when OPENAI_API_KEY is set (#33488).\r\n* Features: run_batch transcription/translation support (#33934), /server_info collect_env (#33246), OTEL tracing during model loading (#31162), clear MM and encoder cache (#33452), HF Hub LoRA resolver (#20320).\r\n* Scoring: Fix multi-document scoring returning single result (#33837).\r\n\r\n### Security\r\n* Patch protobuf for CVE-2026-0994 (#34253).\r\n\r\n### Dependencies\r\n* huggingface-hub updates for Transformers v5 preparation (#33473).\r\n* Transformers v5 compatibility fixes across multiple models (#33977, #33683).\r\n\r\n### Deprecation & Breaking Changes\r\n* Removed BitBlas quantization (#32683) and Marlin 24 (#32688).\r\n* Removed deprecated `reasoning_content` message field (#33402).\r\n* Removed deprecated pooling items (#33477).\r\n* Removed deprecated `VLLM_ALL2ALL_BACKEND` environment variable (#33535).\r\n* Deprecated IPEX for XPU, switched to vllm-xpu-kernels (#33379).\r\n\r\n---\r\n## New Contributors 🎉\r\n\r\n* @aabbccddwasd made their first contribution in https://github.com/vllm-project/vllm/pull/33771\r\n* @Code4me2 made their first contribution in https://github.com/vllm-project/vllm/pull/33517\r\n* @ikchifo made their first contribution in https://github.com/vllm-project/vllm/pull/33967\r\n* @jiangwu300 made their first contribution in https://github.com/vllm-project/vllm/pull/33604\r\n* @pjs102793 made their first contribution in https://github.com/vllm-project/vllm/pull/33963\r\n* @sleepcoo made their first contribution in https://github.com/vllm-project/vllm/pull/33978\r\n* @TundeAtSN made their first contribution in https://github.com/vllm-project/vllm/pull/33939","reactions":{"url":"https://api.github.com/repos/vllm-project/vllm/releases/285918914/reactions","total_count":97,"+1":51,"-1":0,"laugh":0,"hooray":5,"confused":0,"heart":0,"rocket":32,"eyes":9},"mentions_count":7},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/283129012","assets_url":"https://api.github.com/repos/vllm-project/vllm/releases/283129012/assets","upload_url":"https://uploads.github.com/repos/vllm-project/vllm/releases/283129012/assets{?name,label}","html_url":"https://github.com/vllm-project/vllm/releases/tag/v0.15.1","id":283129012,"author":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"node_id":"RE_kwDOI7xefs4Q4DS0","tag_name":"v0.15.1","target_commitish":"main","name":"v0.15.1","draft":false,"immutable":false,"prerelease":false,"created_at":"2026-02-04T01:28:32Z","updated_at":"2026-02-05T01:01:39Z","published_at":"2026-02-04T20:48:08Z","assets":[{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/350768857","id":350768857,"node_id":"RA_kwDOI7xefs4U6E7Z","name":"vllm-0.15.1+cpu-cp38-abi3-manylinux_2_35_aarch64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":30951900,"digest":"sha256:068ad7f8ada077c3c9ef95d69f347e062c094ed104195070741e95c568c0f473","download_count":252,"created_at":"2026-02-04T20:48:21Z","updated_at":"2026-02-04T20:48:23Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.15.1/vllm-0.15.1%2Bcpu-cp38-abi3-manylinux_2_35_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/350768858","id":350768858,"node_id":"RA_kwDOI7xefs4U6E7a","name":"vllm-0.15.1+cpu-cp38-abi3-manylinux_2_35_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":34542588,"digest":"sha256:64f33bb646027d8017d53a6d5d5548610e9b79994d4be98b9ff4b4eb49379a39","download_count":6605,"created_at":"2026-02-04T20:48:21Z","updated_at":"2026-02-04T20:48:23Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.15.1/vllm-0.15.1%2Bcpu-cp38-abi3-manylinux_2_35_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/350768863","id":350768863,"node_id":"RA_kwDOI7xefs4U6E7f","name":"vllm-0.15.1+cu130-cp38-abi3-manylinux_2_35_aarch64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":265822577,"digest":"sha256:90060db4d1a65b8bd26d87d7f28e1d3e37ea95e732e078126349c945cab4f644","download_count":1497,"created_at":"2026-02-04T20:48:21Z","updated_at":"2026-02-04T20:48:36Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.15.1/vllm-0.15.1%2Bcu130-cp38-abi3-manylinux_2_35_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/350768856","id":350768856,"node_id":"RA_kwDOI7xefs4U6E7Y","name":"vllm-0.15.1+cu130-cp38-abi3-manylinux_2_35_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":281337829,"digest":"sha256:a8bd539be4c3efaef3c7e990caa4029072b8f06ad046e87b359cbf75de8b5260","download_count":5905,"created_at":"2026-02-04T20:48:21Z","updated_at":"2026-02-04T20:48:38Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.15.1/vllm-0.15.1%2Bcu130-cp38-abi3-manylinux_2_35_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/350768861","id":350768861,"node_id":"RA_kwDOI7xefs4U6E7d","name":"vllm-0.15.1-cp38-abi3-manylinux_2_31_aarch64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":461362624,"digest":"sha256:97bfc79b0c29d242c57b0d395e48d2949a868957587b853deb813a985a41ed6e","download_count":72,"created_at":"2026-02-04T20:48:21Z","updated_at":"2026-02-04T20:48:43Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.15.1/vllm-0.15.1-cp38-abi3-manylinux_2_31_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/350768891","id":350768891,"node_id":"RA_kwDOI7xefs4U6E77","name":"vllm-0.15.1-cp38-abi3-manylinux_2_31_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":509219874,"digest":"sha256:d3810299d331fc1031c8a2a9886f1f0e0cc2f14ddad284d337174324b1c83e92","download_count":4969,"created_at":"2026-02-04T20:48:23Z","updated_at":"2026-02-04T20:48:40Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.15.1/vllm-0.15.1-cp38-abi3-manylinux_2_31_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/350768895","id":350768895,"node_id":"RA_kwDOI7xefs4U6E7_","name":"vllm-0.15.1.tar.gz","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/x-gtar","state":"uploaded","size":20016024,"digest":"sha256:a8fd1306f84590dbf83e8f88c941a6252ed0e6cc82bf4c3db1ab7b2f553e2ab4","download_count":344,"created_at":"2026-02-04T20:48:23Z","updated_at":"2026-02-04T20:48:25Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.15.1/vllm-0.15.1.tar.gz"}],"tarball_url":"https://api.github.com/repos/vllm-project/vllm/tarball/v0.15.1","zipball_url":"https://api.github.com/repos/vllm-project/vllm/zipball/v0.15.1","body":"v0.15.1 is a patch release with security fixes, RTX Blackwell GPU fixes support, and bug fixes.\r\n\r\n## Security\r\n\r\n- **CVE-2025-69223**: Updated aiohttp dependency (#33621)\r\n- **CVE-2026-0994**: Updated Protobuf dependency (#33619)\r\n\r\n## Highlights\r\n\r\n### Bugfix Hardware Support\r\n- **RTX Blackwell (SM120)**: Fixed NVFP4 MoE kernel support for RTX Blackwell workstation GPUs. Previously, NVFP4 MoE models would fail to load on these GPUs (#33417)\r\n- **FP8 kernel selection**: Fixed FP8 CUTLASS group GEMM to properly fall back to Triton kernels on SM120 GPUs (#33285)\r\n\r\n### Model Support\r\n- **Step-3.5-Flash**: New model support (#33523)\r\n\r\n### Bugfix Model Support\r\n- **Qwen3-VL-Reranker**: Fixed model loading (#33298)\r\n- **Whisper**: Fixed FlashAttention2 with full CUDA graphs (#33360)\r\n\r\n### Performance\r\n- **torch.compile cold-start**: Fixed regression that increased cold-start compilation time (Llama3-70B: ~88s → ~22s) (#33441)\r\n- **MoE forward pass**: Optimized by caching layer name computation (#33184)\r\n\r\n### Bug Fixes\r\n- Fixed prefix cache hit rate of 0% with GPT-OSS style hybrid attention models (#33524)\r\n- Enabled Triton MoE backend for FP8 per-tensor dynamic quantization (#33300)\r\n- Disabled unsupported Renormalize routing methods for TRTLLM per-tensor FP8 MoE (#33620)\r\n- Fixed speculative decoding metrics crash when no tokens generated (#33729)\r\n- Disabled fast MoE cold start optimization with speculative decoding (#33624)\r\n- Fixed ROCm skinny GEMM dispatch logic (#33366)\r\n\r\n### Dependencies\r\n- Pinned LMCache >= v0.3.9 for API compatibility (#33440)\r\n\r\n## New Contributors 🎉\r\n* @zaristei2 made their first contribution in https://github.com/vllm-project/vllm/pull/33621\r\n\r\n**Full Changelog**: https://github.com/vllm-project/vllm/compare/v0.15.0...v0.15.1","reactions":{"url":"https://api.github.com/repos/vllm-project/vllm/releases/283129012/reactions","total_count":20,"+1":0,"-1":0,"laugh":0,"hooray":0,"confused":0,"heart":0,"rocket":20,"eyes":0},"mentions_count":1},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/281026737","assets_url":"https://api.github.com/repos/vllm-project/vllm/releases/281026737/assets","upload_url":"https://uploads.github.com/repos/vllm-project/vllm/releases/281026737/assets{?name,label}","html_url":"https://github.com/vllm-project/vllm/releases/tag/v0.15.0","id":281026737,"author":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"node_id":"RE_kwDOI7xefs4QwCCx","tag_name":"v0.15.0","target_commitish":"main","name":"v0.15.0","draft":false,"immutable":false,"prerelease":false,"created_at":"2026-01-29T06:47:10Z","updated_at":"2026-01-29T10:21:01Z","published_at":"2026-01-29T10:21:01Z","assets":[{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/347611889","id":347611889,"node_id":"RA_kwDOI7xefs4UuCLx","name":"vllm-0.15.0+cpu-cp38-abi3-manylinux_2_35_aarch64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":30926362,"digest":"sha256:b822ececd033a2f4af6e1bb9902caedf1adbfd8ae77355554afffdbf927f5360","download_count":139,"created_at":"2026-01-29T10:15:11Z","updated_at":"2026-01-29T10:15:18Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.15.0/vllm-0.15.0%2Bcpu-cp38-abi3-manylinux_2_35_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/347611893","id":347611893,"node_id":"RA_kwDOI7xefs4UuCL1","name":"vllm-0.15.0+cpu-cp38-abi3-manylinux_2_35_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":34516852,"digest":"sha256:38c25efe50276e4473aa0d3cdcbd00532a1a241ec9f5063fdb3d755e97a49a1e","download_count":711,"created_at":"2026-01-29T10:15:11Z","updated_at":"2026-01-29T10:15:19Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.15.0/vllm-0.15.0%2Bcpu-cp38-abi3-manylinux_2_35_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/347611885","id":347611885,"node_id":"RA_kwDOI7xefs4UuCLt","name":"vllm-0.15.0+cu130-cp38-abi3-manylinux_2_35_aarch64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":265797024,"digest":"sha256:5b90fad30e0c6dc3c3a085c5a2d6626c28bd05f895a419e23178b33916779afc","download_count":897,"created_at":"2026-01-29T10:15:11Z","updated_at":"2026-01-29T10:15:31Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.15.0/vllm-0.15.0%2Bcu130-cp38-abi3-manylinux_2_35_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/347611892","id":347611892,"node_id":"RA_kwDOI7xefs4UuCL0","name":"vllm-0.15.0+cu130-cp38-abi3-manylinux_2_35_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":281312277,"digest":"sha256:43449e7350ee4c695693bc47561d34e35b722bdceb025b14e505262717677b3f","download_count":1514,"created_at":"2026-01-29T10:15:11Z","updated_at":"2026-01-29T10:15:29Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.15.0/vllm-0.15.0%2Bcu130-cp38-abi3-manylinux_2_35_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/347611886","id":347611886,"node_id":"RA_kwDOI7xefs4UuCLu","name":"vllm-0.15.0-cp38-abi3-manylinux_2_31_aarch64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":461337080,"digest":"sha256:faa6d7bc5f65b1b6f8b1e6a028759f78246113294fbc7e6633d25459d652d2a7","download_count":54,"created_at":"2026-01-29T10:15:11Z","updated_at":"2026-01-29T10:15:37Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.15.0/vllm-0.15.0-cp38-abi3-manylinux_2_31_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/347611944","id":347611944,"node_id":"RA_kwDOI7xefs4UuCMo","name":"vllm-0.15.0-cp38-abi3-manylinux_2_31_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":509194343,"digest":"sha256:fd03371687568c511be2f9204932b3d76fd432ccaf4c19632cd53e98b0eb53c8","download_count":222,"created_at":"2026-01-29T10:15:18Z","updated_at":"2026-01-29T10:15:40Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.15.0/vllm-0.15.0-cp38-abi3-manylinux_2_31_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/347611950","id":347611950,"node_id":"RA_kwDOI7xefs4UuCMu","name":"vllm-0.15.0.tar.gz","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/x-gtar","state":"uploaded","size":19991714,"digest":"sha256:8836f2639d70c911a38fecda6e9586adae9ea29049b3c00d007c761481e56026","download_count":153,"created_at":"2026-01-29T10:15:19Z","updated_at":"2026-01-29T10:15:21Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.15.0/vllm-0.15.0.tar.gz"}],"tarball_url":"https://api.github.com/repos/vllm-project/vllm/tarball/v0.15.0","zipball_url":"https://api.github.com/repos/vllm-project/vllm/zipball/v0.15.0","body":"## Highlights\r\n\r\nThis release features 335 commits from 158 contributors (39 new)!\r\n\r\n### Model Support\r\n* **New architectures**: Kimi-K2.5 (#33131), Molmo2 (#30997), Step3vl 10B (#32329), Step1 (#32511), GLM-Lite (#31386), Eagle2.5-8B VLM (#32456).\r\n* **LoRA expansion**: Nemotron-H (#30802), InternVL2 (#32397), MiniMax M2 (#32763).\r\n* **Speculative decoding**: EAGLE3 for Pixtral/LlavaForConditionalGeneration (#32542), Qwen3 VL MoE (#32048), draft model support (#24322).\r\n* **Embeddings**: BGE-M3 sparse embeddings and ColBERT embeddings (#14526).\r\n* **Model enhancements**: Voxtral streaming architecture (#32861), SharedFusedMoE for Qwen3MoE (#32082), dynamic resolution for Nemotron Nano VL (#32121), Molmo2 vision backbone quantization (#32385).\r\n\r\n### Engine Core\r\n* **Async scheduling + Pipeline Parallelism**: `--async-scheduling` now works with pipeline parallelism (#32359).\r\n* **Mamba prefix caching**: Block-aligned prefix caching for Mamba/hybrid models with `--enable-prefix-caching --mamba-cache-mode align`. Achieves ~2x speedup by caching Mamba states directly (#30877).\r\n* **Session-based streaming input**: New incremental input support for interactive workloads like ASR. Accepts async generators producing `StreamingInput` objects while maintaining KV cache alignment (#28973).\r\n* **Model Runner V2**: VLM support (#32546), architecture improvements.\r\n* **LoRA**: Inplace loading for memory efficiency (#31326).\r\n* **AOT compilation**: torch.compile inductor artifacts support (#25205).\r\n* **Performance**: KV cache offloading redundant load prevention (#29087), FlashAttn attention/cache update separation (#25954).\r\n\r\n### Hardware & Performance\r\n\r\n#### NVIDIA\r\n* **Blackwell defaults**: FlashInfer MLA is now the default MLA backend on Blackwell, with TRTLLM as default prefill (#32615).\r\n* **MoE performance**: 1.2-2% E2E throughput improvement via grouped topk kernel fusion (#32058), NVFP4 small-batch decoding improvement (#30885), faster cold start for MoEs with torch.compile (#32805).\r\n* **FP4 kernel optimization**: Up to 65% faster FP4 quantization on Blackwell (SM100F) using 256-bit loads, ~4% E2E throughput improvement (#32520).\r\n* **Kernel improvements**: topk_sigmoid kernel for MoE routing (#31246), atomics reduce counting for SplitK skinny GEMMs (#29843), fused cat+quant for FP8 KV cache in MLA (#32950).\r\n* **torch.compile**: SiluAndMul and QuantFP8 CustomOp compilation (#32806), Triton prefill attention performance (#32403).\r\n\r\n#### AMD ROCm\r\n* **MoRI EP**: High-performance all2all backend for Expert Parallel (#28664).\r\n* **Attention improvements**: Shuffle KV cache layout and assembly paged attention kernel for AiterFlashAttentionBackend (#29887).\r\n* **FP4 support**: MLA projection GEMMs with dynamic quantization (#32238).\r\n* **Consumer GPU support**: Flash Attention Triton backend on RDNA3/RDNA4 (#32944).\r\n\r\n#### Other Platforms\r\n* **TPU**: Pipeline parallelism support (#28506), backend option (#32438).\r\n* **Intel XPU**: AgRsAll2AllManager for distributed communication (#32654).\r\n* **CPU**: NUMA-aware acceleration for TP/DP inference on ARM (#32792), PyTorch 2.10 (#32869).\r\n* **Whisper**: torch.compile support (#30385).\r\n* **WSL**: Platform compatibility fix for Windows Subsystem for Linux (#32749).\r\n\r\n### Quantization\r\n* **MXFP4**: W4A16 support for compressed-tensors MoE models (#32285).\r\n* **Non-gated MoE**: Quantization support with Marlin, NVFP4 CUTLASS, FP8, INT8, and compressed-tensors (#32257).\r\n* **Intel**: Quantization Toolkit integration (#31716).\r\n* **FP8 KV cache**: Per-tensor and per-attention-head quantization via llmcompressor (#30141).\r\n\r\n### API & Frontend\r\n* **Responses API**: Partial message generation (#32100), `include_stop_str_in_output` tuning (#32383), `prompt_cache_key` support (#32824).\r\n* **OpenAI API**: `skip_special_tokens` configuration (#32345).\r\n* **Score endpoint**: Flexible input formats with `data_1`/`data_2` and `queries`/`documents` (#32577).\r\n* **Render endpoints**: New endpoints for prompt preprocessing (#32473).\r\n* **Whisper API**: `avg_logprob` and `compression_ratio` in verbose_json segments (#31059).\r\n* **Security**: FIPS 140-3 compliant hash option for enterprise/government users (#32386), `--ssl-ciphers` CLI argument (#30937).\r\n* **UX improvements**: Auto `api_server_count` based on `dp_size` (#32525), wheel variant auto-detection during install (#32948), custom profiler URI schemes (#32393).\r\n\r\n### Dependencies\r\n* FlashInfer v0.6.1 (#30993)\r\n* Transformers 4.57.5 (#32287)\r\n* PyTorch 2.10 for CPU backend (#32869)\r\n* DeepGEMM newer version (#32479)\r\n\r\n### Breaking Changes & Deprecations\r\n* **Metrics**: Removed deprecated `vllm:time_per_output_token_seconds` metric - use `vllm:inter_token_latency_seconds` instead (#32661).\r\n* **Environment variables**: Removed deprecated environment variables (#32812).\r\n* **Quantization**: DeepSpeedFp8 removed (#32679), RTN removed (#32697), HQQ deprecated (#32681).\r\n\r\n### Bug Fixes\r\n* **Speculative decoding**: Eagle draft_model_config fix (#31753).\r\n* **DeepSeek**: DeepSeek-V3.1 + DeepGEMM incompatible scale shapes fix (#32361).\r\n* **Distributed**: DP+MoE inference fix via CpuCommunicator (#31867), P/D with non-MoE DP fix (#33037).\r\n* **EPLB**: Possible deadlock fix (#32418).\r\n* **NIXL**: UCX memory leak fix by exporting UCX_MEM_MMAP_HOOK_MODE=none (#32181).\r\n* **Structured output**: Outlines byte fallback handling fix (#31391).\r\n\r\n---\r\n\r\n## New Contributors 🎉\r\n* @YunzhuLu made their first contribution in https://github.com/vllm-project/vllm/pull/32126\r\n* @emricksini-h made their first contribution in https://github.com/vllm-project/vllm/pull/30784\r\n* @dsfaccini made their first contribution in https://github.com/vllm-project/vllm/pull/32289\r\n* @ofirzaf made their first contribution in https://github.com/vllm-project/vllm/pull/32312\r\n* @seekskyworld made their first contribution in https://github.com/vllm-project/vllm/pull/32321\r\n* @brian033 made their first contribution in https://github.com/vllm-project/vllm/pull/31715\r\n* @TomerBN-Nvidia made their first contribution in https://github.com/vllm-project/vllm/pull/32257\r\n* @vanshilshah97 made their first contribution in https://github.com/vllm-project/vllm/pull/32448\r\n* @George-Polya made their first contribution in https://github.com/vllm-project/vllm/pull/32385\r\n* @T1mn made their first contribution in https://github.com/vllm-project/vllm/pull/32411\r\n* @mritunjaysharma394 made their first contribution in https://github.com/vllm-project/vllm/pull/31492\r\n* @randzero made their first contribution in https://github.com/vllm-project/vllm/pull/32511\r\n* @DemingCheng made their first contribution in https://github.com/vllm-project/vllm/pull/32556\r\n* @iboiko-habana made their first contribution in https://github.com/vllm-project/vllm/pull/32471\r\n* @honglyua-il made their first contribution in https://github.com/vllm-project/vllm/pull/32462\r\n* @hyeongyun0916 made their first contribution in https://github.com/vllm-project/vllm/pull/32473\r\n* @DanielMe made their first contribution in https://github.com/vllm-project/vllm/pull/32560\r\n* @netanel-haber made their first contribution in https://github.com/vllm-project/vllm/pull/32121\r\n* @longregen made their first contribution in https://github.com/vllm-project/vllm/pull/28784\r\n* @jasonyanwenl made their first contribution in https://github.com/vllm-project/vllm/pull/32749\r\n* @Wauplin made their first contribution in https://github.com/vllm-project/vllm/pull/32788\r\n* @ikaadil made their first contribution in https://github.com/vllm-project/vllm/pull/32775\r\n* @alexsun07 made their first contribution in https://github.com/vllm-project/vllm/pull/28664\r\n* @liranschour made their first contribution in https://github.com/vllm-project/vllm/pull/30207\r\n* @AuYang261 made their first contribution in https://github.com/vllm-project/vllm/pull/32844\r\n* @diviramon made their first contribution in https://github.com/vllm-project/vllm/pull/32393\r\n* @RishabhSaini made their first contribution in https://github.com/vllm-project/vllm/pull/32884\r\n* @MatteoFari made their first contribution in https://github.com/vllm-project/vllm/pull/32397\r\n* @peakcrosser7 made their first contribution in https://github.com/vllm-project/vllm/pull/30877\r\n* @orionr made their first contribution in https://github.com/vllm-project/vllm/pull/30443\r\n* @marksverdhei made their first contribution in https://github.com/vllm-project/vllm/pull/32614\r\n* @joninco made their first contribution in https://github.com/vllm-project/vllm/pull/32935\r\n* @monajafi-amd made their first contribution in https://github.com/vllm-project/vllm/pull/32944\r\n* @ruizcrp made their first contribution in https://github.com/vllm-project/vllm/pull/32988\r\n* @sjhddh made their first contribution in https://github.com/vllm-project/vllm/pull/32983\r\n* @HirokenOvo made their first contribution in https://github.com/vllm-project/vllm/pull/32646\r\n* @Chenhao-Guan made their first contribution in https://github.com/vllm-project/vllm/pull/32763\r\n* @joshuadeng made their first contribution in https://github.com/vllm-project/vllm/pull/28973\r\n* @ZhanqiuHu made their first contribution in https://github.com/vllm-project/vllm/pull/33016\r\n\r\n**Full Changelog**: https://github.com/vllm-project/vllm/compare/v0.14.1...v0.15.0","reactions":{"url":"https://api.github.com/repos/vllm-project/vllm/releases/281026737/reactions","total_count":43,"+1":13,"-1":0,"laugh":1,"hooray":0,"confused":0,"heart":8,"rocket":21,"eyes":0},"mentions_count":39},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/279660935","assets_url":"https://api.github.com/repos/vllm-project/vllm/releases/279660935/assets","upload_url":"https://uploads.github.com/repos/vllm-project/vllm/releases/279660935/assets{?name,label}","html_url":"https://github.com/vllm-project/vllm/releases/tag/v0.14.1","id":279660935,"author":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"node_id":"RE_kwDOI7xefs4Qq0mH","tag_name":"v0.14.1","target_commitish":"main","name":"v0.14.1","draft":false,"immutable":false,"prerelease":false,"created_at":"2026-01-23T22:22:49Z","updated_at":"2026-01-24T20:59:37Z","published_at":"2026-01-24T20:29:27Z","assets":[{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/345362311","id":345362311,"node_id":"RA_kwDOI7xefs4Ulc-H","name":"vllm-0.14.1+cpu-cp38-abi3-manylinux_2_35_aarch64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":29036749,"digest":"sha256:e87d6f7d47dd928d0f7b3c2bef544fee834a46cf06e6131a0c23f2b8dafb56b7","download_count":174,"created_at":"2026-01-24T20:29:44Z","updated_at":"2026-01-24T20:29:46Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.14.1/vllm-0.14.1%2Bcpu-cp38-abi3-manylinux_2_35_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/345362309","id":345362309,"node_id":"RA_kwDOI7xefs4Ulc-F","name":"vllm-0.14.1+cpu-cp38-abi3-manylinux_2_35_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":34314254,"digest":"sha256:2f4fc363c0115553814cb15d147fc3ea06dd8bf34693ba4615c48f5bfc955b1d","download_count":1630,"created_at":"2026-01-24T20:29:44Z","updated_at":"2026-01-24T20:29:46Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.14.1/vllm-0.14.1%2Bcpu-cp38-abi3-manylinux_2_35_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/345362310","id":345362310,"node_id":"RA_kwDOI7xefs4Ulc-G","name":"vllm-0.14.1+cu130-cp38-abi3-manylinux_2_35_aarch64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":259756884,"digest":"sha256:78ef3d111534abd835888aac456556a998434fe3aaa6fae7bb1a79aef7ac654a","download_count":14621,"created_at":"2026-01-24T20:29:44Z","updated_at":"2026-01-24T20:29:55Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.14.1/vllm-0.14.1%2Bcu130-cp38-abi3-manylinux_2_35_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/345362308","id":345362308,"node_id":"RA_kwDOI7xefs4Ulc-E","name":"vllm-0.14.1+cu130-cp38-abi3-manylinux_2_35_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":274641608,"digest":"sha256:8704d372ac9521915173870788373ec9601bb286b96b2b2fa26c77eb2deb8c5d","download_count":3589,"created_at":"2026-01-24T20:29:44Z","updated_at":"2026-01-24T20:29:56Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.14.1/vllm-0.14.1%2Bcu130-cp38-abi3-manylinux_2_35_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/345362307","id":345362307,"node_id":"RA_kwDOI7xefs4Ulc-D","name":"vllm-0.14.1-cp38-abi3-manylinux_2_31_aarch64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":448041824,"digest":"sha256:3b77b9daece63d8c9837ccc512c367ead1d142bf9965a3d4e1fbf158d7a56b85","download_count":66,"created_at":"2026-01-24T20:29:44Z","updated_at":"2026-01-24T20:30:03Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.14.1/vllm-0.14.1-cp38-abi3-manylinux_2_31_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/345362316","id":345362316,"node_id":"RA_kwDOI7xefs4Ulc-M","name":"vllm-0.14.1-cp38-abi3-manylinux_2_31_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":495378590,"digest":"sha256:435565a8299a5fcdbf5af4a798e342ddeec3f90df83c8227cc204bdebca19494","download_count":264,"created_at":"2026-01-24T20:29:46Z","updated_at":"2026-01-24T20:30:08Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.14.1/vllm-0.14.1-cp38-abi3-manylinux_2_31_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/345362317","id":345362317,"node_id":"RA_kwDOI7xefs4Ulc-N","name":"vllm-0.14.1.tar.gz","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/x-gtar","state":"uploaded","size":19698767,"digest":"sha256:c7c6d0a729d29e70ac10baacbc7186930da24f8ae34f60451330527ff0be84ac","download_count":2244,"created_at":"2026-01-24T20:29:47Z","updated_at":"2026-01-24T20:29:48Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.14.1/vllm-0.14.1.tar.gz"}],"tarball_url":"https://api.github.com/repos/vllm-project/vllm/tarball/v0.14.1","zipball_url":"https://api.github.com/repos/vllm-project/vllm/zipball/v0.14.1","body":"This is a patch release on top of `v0.14.0` to address a few security and memory leak fixes.","reactions":{"url":"https://api.github.com/repos/vllm-project/vllm/releases/279660935/reactions","total_count":19,"+1":14,"-1":0,"laugh":0,"hooray":0,"confused":0,"heart":0,"rocket":5,"eyes":0}},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/278179519","assets_url":"https://api.github.com/repos/vllm-project/vllm/releases/278179519/assets","upload_url":"https://uploads.github.com/repos/vllm-project/vllm/releases/278179519/assets{?name,label}","html_url":"https://github.com/vllm-project/vllm/releases/tag/v0.14.0","id":278179519,"author":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"node_id":"RE_kwDOI7xefs4QlK6_","tag_name":"v0.14.0","target_commitish":"main","name":"v0.14.0","draft":false,"immutable":false,"prerelease":false,"created_at":"2026-01-17T05:04:48Z","updated_at":"2026-01-20T21:08:29Z","published_at":"2026-01-20T09:20:31Z","assets":[{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/343200103","id":343200103,"node_id":"RA_kwDOI7xefs4UdNFn","name":"vllm-0.14.0+cpu-cp38-abi3-manylinux_2_35_aarch64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":29036491,"digest":"sha256:367fea77698bae8aaadd504bded4b6b60258a7397c667e15af2215e457149296","download_count":156,"created_at":"2026-01-20T09:22:35Z","updated_at":"2026-01-20T09:22:37Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.14.0/vllm-0.14.0%2Bcpu-cp38-abi3-manylinux_2_35_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/343200104","id":343200104,"node_id":"RA_kwDOI7xefs4UdNFo","name":"vllm-0.14.0+cpu-cp38-abi3-manylinux_2_35_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":34313994,"digest":"sha256:51a1cf04c7e55932da84cc95bea2ed8ab0a009c453112a9c91f5eef3ae28e990","download_count":2038,"created_at":"2026-01-20T09:22:35Z","updated_at":"2026-01-20T09:22:37Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.14.0/vllm-0.14.0%2Bcpu-cp38-abi3-manylinux_2_35_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/343200105","id":343200105,"node_id":"RA_kwDOI7xefs4UdNFp","name":"vllm-0.14.0+cu130-cp38-abi3-manylinux_2_35_aarch64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":259756623,"digest":"sha256:6803b33eb3211f1425ebbcf3cdbfa64350143e739ffef5a4a125b48a82ff98ac","download_count":1254,"created_at":"2026-01-20T09:22:35Z","updated_at":"2026-01-20T09:22:50Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.14.0/vllm-0.14.0%2Bcu130-cp38-abi3-manylinux_2_35_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/343200106","id":343200106,"node_id":"RA_kwDOI7xefs4UdNFq","name":"vllm-0.14.0+cu130-cp38-abi3-manylinux_2_35_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":274641401,"digest":"sha256:3ba4b04b34e7f924c93f9b4681064b9bb41c8ee353ce071c64cfb4083700cfd7","download_count":1068,"created_at":"2026-01-20T09:22:35Z","updated_at":"2026-01-20T09:22:49Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.14.0/vllm-0.14.0%2Bcu130-cp38-abi3-manylinux_2_35_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/343200100","id":343200100,"node_id":"RA_kwDOI7xefs4UdNFk","name":"vllm-0.14.0-cp38-abi3-manylinux_2_31_aarch64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":448040411,"digest":"sha256:f4684881e3677b69d731494c68ad0525588b02cad556c5d5aef4f7eba9368c86","download_count":72,"created_at":"2026-01-20T09:22:35Z","updated_at":"2026-01-20T09:22:57Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.14.0/vllm-0.14.0-cp38-abi3-manylinux_2_31_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/343200133","id":343200133,"node_id":"RA_kwDOI7xefs4UdNGF","name":"vllm-0.14.0-cp38-abi3-manylinux_2_31_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":495377847,"digest":"sha256:9362cadde2cd1329a0c4a0ea9eace233d79576b3d6f5648b854922b2a2554a01","download_count":508,"created_at":"2026-01-20T09:22:37Z","updated_at":"2026-01-20T09:23:00Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.14.0/vllm-0.14.0-cp38-abi3-manylinux_2_31_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/343200134","id":343200134,"node_id":"RA_kwDOI7xefs4UdNGG","name":"vllm-0.14.0.tar.gz","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/x-gtar","state":"uploaded","size":19690728,"digest":"sha256:ed86c6a9d22fbfd059648794493cad88444c71688f3ff6efcb26c5ca353f0e5d","download_count":166,"created_at":"2026-01-20T09:22:38Z","updated_at":"2026-01-20T09:22:39Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.14.0/vllm-0.14.0.tar.gz"}],"tarball_url":"https://api.github.com/repos/vllm-project/vllm/tarball/v0.14.0","zipball_url":"https://api.github.com/repos/vllm-project/vllm/zipball/v0.14.0","body":"## Highlights\r\n\r\nThis release features approximately 660 commits from 251 contributors (86 new contributors).\r\n\r\n**Breaking Changes:**\r\n- **Async scheduling is now enabled by default** - Users who experience issues can disable with `--no-async-scheduling`.\r\n   - Excludes some not-yet-supported configurations: pipeline parallel, CPU backend, non-MTP/Eagle spec decoding.\r\n- **PyTorch 2.9.1** is now required and the default wheel is compiled against cu129.\r\n- Deprecated quantization schemes have been removed (#31688, #31285).\r\n- When using speculative decoding, unsupported sampling parameters will fail rather than being silently ignored (#31982).\r\n\r\n**Key Improvements:**\r\n- **Async scheduling enabled by default** (#27614): Overlaps engine core scheduling with GPU execution, improving throughput without user configuration. Now also works with speculative decoding (#31998) and structured outputs (#29821).\r\n- **gRPC server entrypoint** (#30190): Alternative to REST API with binary protocol, HTTP/2 multiplexing.\r\n- **`--max-model-len auto`** (#29431): Automatically fits context length to available GPU memory, eliminating OOM startup failures.\r\n- **Model inspection view** (#29450): View the modules, attention backends, and quantization of your model in vLLM by specifying `VLLM_LOG_MODEL_INSPECTION=1` or by simply printing the `LLM` object.\r\n- **Model Runner V2 enhancements**: UVA block tables (#31965), M-RoPE (#32143), `logit_bias`/`allowed_token_ids`/`min_tokens` support (#32163).\r\n  - Please note that Model Runner V2 is still experimental and disabled by default.\r\n\r\n### Model Support\r\n\r\n**New Model Architectures:**\r\n- Grok-2 with tiktoken tokenizer (#31847)\r\n- LFM2-VL vision-language model (#31758)\r\n- MiMo-V2-Flash (#30836)\r\n- openPangu MoE (#28775)\r\n- IQuestCoder (#31575)\r\n- Nemotron Parse 1.1 (#30864)\r\n- GLM-ASR audio (#31436)\r\n- Isaac vision model v0.1/v0.2 (#28367, #31550)\r\n- Kanana-1.5-v-3b-instruct (#29384)\r\n- K-EXAONE-236B-A23B MoE (#31621)\r\n\r\n**LoRA Support Expansion:**\r\n- Multimodal tower/connector LoRA (#26674): LLaVA (#31513), BLIP2 (#31620), PaliGemma (#31656), Pixtral (#31724), DotsOCR (#31825), GLM4-V (#31652)\r\n- DeepSeek-OCR (#31569), Qwen3-Next (#31719), NemotronH (#31539), PLaMo 2/3 (#31322)\r\n- Vision LoRA mm_processor_cache support (#31927)\r\n- MoE expert base_layer loading (#31104)\r\n\r\n**Model Enhancements:**\r\n- Qwen3-VL as reranker (#31890)\r\n- DeepSeek v3.2 chat prefix completion (#31147)\r\n- GLM-4.5/GLM-4.7 `enable_thinking: false` (#31788)\r\n- Ernie4.5-VL video timestamps (#31274)\r\n- Score template expansion (#31335)\r\n- LLaMa4 vision encoder compilation (#30709)\r\n- NemotronH quantized attention (#31898)\r\n\r\n### Engine Core\r\n\r\n- **Async scheduling default** with spec decode (#27614, #31998) and structured outputs (#29821)\r\n- **Hybrid allocator + KV connector** (#30166) with multiple KV cache groups (#31707)\r\n- Triton attention: encoder-only/cross attention (#31406), cross-layer blocks (#30687)\r\n- Mamba2 prefix cache optimization (#28047)\r\n- Batch invariant LoRA (#30097)\r\n- LoRA name in BlockStored for KV-cache reconstruction (#27577)\r\n- Request ID collision prevention (#27987)\r\n- Dense model DP without overhead (#30739)\r\n- Async + spec decode penalties/bad_words (#30495)\r\n\r\n### Hardware & Performance\r\n\r\n**CUTLASS MoE Optimizations:**\r\n- 2.9% throughput + 10.8% TTFT via fill(0) optimization (#31754)\r\n- 5.3% throughput + 2.2% TTFT via problem size calculation (#31830)\r\n- Fused SiLU+Mul+Quant for NVFP4 (#31832)\r\n- NVFP4 stride fusion (#31837)\r\n\r\n**Other Performance:**\r\n- GDN attention decode speedup (Qwen3-Next) (#31722)\r\n- Fused RoPE + MLA KV-cache write (#25774)\r\n- Sliding window attention optimization (#31984)\r\n- FlashInfer DeepGEMM swapAB SM90 (#29213)\r\n- Unpermute-aware fused MoE + small-batch fallback (#29354)\r\n- GDN Attention blocking copy removal (#31167)\r\n- FusedMoE LoRA small rank performance (#32019)\r\n- EPLB numpy optimization (#29499)\r\n- FlashInfer rotary for DeepSeek (#30729)\r\n- Vectorized activations (#29512)\r\n- NUMA interleaved memory (#30800)\r\n- Async spec decode logprobs (#31336)\r\n\r\n**Hardware Configs:**\r\n- SM103 support (#30705, #31150)\r\n- B300 Blackwell MoE configs (#30629)\r\n- Qwen3-Next FP8 CUTLASS configs (#29553)\r\n- Qwen3Moe B200 Triton configs (#31448)\r\n- GLM-4.5/4.6 RTX Pro 6000 kernels (#31407)\r\n- MiniMax-M2/M2.1 QKNorm (#31493)\r\n- NVFP4 small batch tuning (#30897)\r\n\r\n**Platform:**\r\n- ROCm: AITER RMSNorm fusion (#26575), MTP for AITER MLA (#28624), moriio connector (#29304), xgrammar upstream (#31327)\r\n- XPU: FP8 streaming quant (#30944), custom workers (#30935)\r\n- CPU: Head sizes 80/112 (#31968), async disabled by default (#31525), LoRA MoE CPU pinning (#31317)\r\n- TPU: tpu-inference path (#30808), Sophgo docs (#30949)\r\n\r\n### Large Scale Serving\r\n\r\n- **XBO** (Extended Dual-Batch Overlap) (#30120)\r\n- **NIXL asymmetric TP** (P > D tensor-parallel-size) (#27274)\r\n- NIXL heterogeneous BlockSize/kv_layout (#30275)\r\n- Cross-layers KV layout for MultiConnector (#30761)\r\n- Mooncake protocol expansion (#30133)\r\n- LMCache KV cache registration (#31397)\r\n- EPLB default all2all backend (#30559)\r\n\r\n### Quantization\r\n\r\n- **Marlin for Turing (sm75)** (#29901, #31000)\r\n- **Quark int4-fp8 w4a8 MoE** (#30071)\r\n- **MXFP4 W4A16 dense models** (#31926)\r\n- **ModelOpt FP8 variants** (FP8_PER_CHANNEL_PER_TOKEN, FP8_PB_WO) (#30957)\r\n- ModelOpt KV cache quantization update (#31895)\r\n- NVFP4 Marlin for NVFP4A16 MoEs (#30881)\r\n- Static quant all group shapes (#30833)\r\n- Default MXFP4 LoRA backend: Marlin (#30598)\r\n- compressed-tensors 0.13.0 (#30799)\r\n\r\n### API & Frontend\r\n\r\n**New Features:**\r\n- gRPC server (#30190)\r\n- `--max-model-len auto` (#29431)\r\n- Model inspection view (#29450)\r\n- Offline FastAPI docs (#30184)\r\n- `attention_config` in LLM() (#30710)\r\n- MFU metrics (#30738)\r\n- Iteration logging + NVTX (#31193)\r\n- `reasoning_effort` parameter (#31956)\r\n\r\n**Tool Calling:**\r\n- FunctionGemma parser (#31218)\r\n- GLM-4.7 parser (#30876)\r\n- Kimi K2 update (#31207)\r\n\r\n**CLI:**\r\n- `-ep` for `--enable-expert-parallel` (#30890)\r\n- Complete help messages (#31226)\r\n- Bench serve auto-discovery + `--input-len` (#30816)\r\n- Spec decode acceptance stats (#31739)\r\n- `--enable-log-deltas` (renamed) (#32020)\r\n- `--default-chat-template-kwargs` (#31343)\r\n\r\n**API:**\r\n- `/server_info` env info (#31899)\r\n- MCP streaming in Responses API (#31761)\r\n- `/embeddings` `continue_final_message` (#31497)\r\n- Reranking score templates (#30550)\r\n- Chat template warmup (#30700)\r\n- Configurable handshake timeout (#27444)\r\n- Better 500 errors (#20610)\r\n- Worker init logging (#29493)\r\n- Bench error reporting (#31808)\r\n- Corrupted video recovery (#29197)\r\n- Spec-decode param validation (#31982)\r\n- Validation error metadata (#30134)\r\n\r\n### Security\r\n\r\n- Prevent token leaks in crash logs (#30751)\r\n- `weights_only=True` in torch.load (#32045)\r\n\r\n### Dependencies\r\n\r\n- **PyTorch 2.9.1** (#28495)\r\n- compressed-tensors 0.13.0 (#30799)\r\n- CUDA 13 LMCache/NIXL in Docker (#30913)\r\n- Configurable NVSHMEM version (#30732)\r\n\r\n### Bug Fixes (User-Facing)\r\n\r\n- Invalid UTF-8 tokens (#28874)\r\n- CPU RoPE gibberish with `--enforce-eager` (#31643)\r\n- Tool call streaming finish chunk (#31438)\r\n- Encoder cache leak CPU scheduling stuck (#31857)\r\n- Engine crash: tools + response_format (#32127)\r\n- Voxtral transcription API (#31388)\r\n- Safetensors download optimization (#30537)\r\n\r\n### Deprecations\r\n\r\n- Deprecated quantization schemes removed (#31688, #31285)\r\n- `seed_everything` deprecated (#31659)\r\n\r\n### Documentation\r\n\r\n- vllm-metal plugin docs (#31174)\r\n- Claude Code example (#31188)\r\n- CustomOp developer guide (#30886)\r\n\r\n## New Contributors 🎉\r\n* @penfree made their first contribution in https://github.com/vllm-project/vllm/pull/30237\r\n* @jiangkuaixue123 made their first contribution in https://github.com/vllm-project/vllm/pull/30120\r\n* @jr-shen made their first contribution in https://github.com/vllm-project/vllm/pull/29663\r\n* @grzegorz-k-karch made their first contribution in https://github.com/vllm-project/vllm/pull/30795\r\n* @shanjiaz made their first contribution in https://github.com/vllm-project/vllm/pull/30799\r\n* @Somoku made their first contribution in https://github.com/vllm-project/vllm/pull/29569\r\n* @baoqian426 made their first contribution in https://github.com/vllm-project/vllm/pull/30841\r\n* @SongDI911 made their first contribution in https://github.com/vllm-project/vllm/pull/30852\r\n* @www-spam made their first contribution in https://github.com/vllm-project/vllm/pull/30827\r\n* @Xunzhuo made their first contribution in https://github.com/vllm-project/vllm/pull/30844\r\n* @TheCodeWrangler made their first contribution in https://github.com/vllm-project/vllm/pull/30700\r\n* @SungMinCho made their first contribution in https://github.com/vllm-project/vllm/pull/30738\r\n* @sarathc-cerebras made their first contribution in https://github.com/vllm-project/vllm/pull/30188\r\n* @wzyrrr made their first contribution in https://github.com/vllm-project/vllm/pull/30949\r\n* @navmarri14 made their first contribution in https://github.com/vllm-project/vllm/pull/30629\r\n* @HaloWorld made their first contribution in https://github.com/vllm-project/vllm/pull/30867\r\n* @jeffreywang-anyscale made their first contribution in https://github.com/vllm-project/vllm/pull/31013\r\n* @AmeenP made their first contribution in https://github.com/vllm-project/vllm/pull/31093\r\n* @westers made their first contribution in https://github.com/vllm-project/vllm/pull/31071\r\n* @CedricHwong made their first contribution in https://github.com/vllm-project/vllm/pull/30957\r\n* @c0de128 made their first contribution in https://github.com/vllm-project/vllm/pull/31114\r\n* @Bounty-hunter made their first contribution in https://github.com/vllm-project/vllm/pull/30242\r\n* @jzakrzew made their first contribution in https://github.com/vllm-project/vllm/pull/30550\r\n* @1643661061leo made their first contribution in https://github.com/vllm-project/vllm/pull/30760\r\n* @NickCao made their first contribution in https://github.com/vllm-project/vllm/pull/30070\r\n* @amithkk made their first contribution in https://github.com/vllm-project/vllm/pull/31212\r\n* @gateremark made their first contribution in https://github.com/vllm-project/vllm/pull/31218\r\n* @Tiiiktak made their first contribution in https://github.com/vllm-project/vllm/pull/31274\r\n* @oscardev256 made their first contribution in https://github.com/vllm-project/vllm/pull/28367\r\n* @Jzz1943 made their first contribution in https://github.com/vllm-project/vllm/pull/31448\r\n* @mratsim made their first contribution in https://github.com/vllm-project/vllm/pull/31407\r\n* @twjww made their first contribution in https://github.com/vllm-project/vllm/pull/31445\r\n* @amittell made their first contribution in https://github.com/vllm-project/vllm/pull/31438\r\n* @ricky-chaoju made their first contribution in https://github.com/vllm-project/vllm/pull/30184\r\n* @effortprogrammer made their first contribution in https://github.com/vllm-project/vllm/pull/31343\r\n* @ZT-AIA made their first contribution in https://github.com/vllm-project/vllm/pull/31408\r\n* @rogerxfeng8 made their first contribution in https://github.com/vllm-project/vllm/pull/31522\r\n* @kevin-pw made their first contribution in https://github.com/vllm-project/vllm/pull/31497\r\n* @vintipandey made their first contribution in https://github.com/vllm-project/vllm/pull/31505\r\n* @SameerAsal made their first contribution in https://github.com/vllm-project/vllm/pull/31520\r\n* @Dylan1229 made their first contribution in https://github.com/vllm-project/vllm/pull/31546\r\n* @reaganjlee made their first contribution in https://github.com/vllm-project/vllm/pull/29105\r\n* @zhima771 made their first contribution in https://github.com/vllm-project/vllm/pull/31569\r\n* @jayhemnani9910 made their first contribution in https://github.com/vllm-project/vllm/pull/31513\r\n* @Tmn07 made their first contribution in https://github.com/vllm-project/vllm/pull/31572\r\n* @vsourirajan made their first contribution in https://github.com/vllm-project/vllm/pull/31549\r\n* @labAxiaoming made their first contribution in https://github.com/vllm-project/vllm/pull/31601\r\n* @massif-01 made their first contribution in https://github.com/vllm-project/vllm/pull/31604\r\n* @PHOEBEMOON0802 made their first contribution in https://github.com/vllm-project/vllm/pull/31147\r\n* @tpopp made their first contribution in https://github.com/vllm-project/vllm/pull/29993\r\n* @ppppqp made their first contribution in https://github.com/vllm-project/vllm/pull/31620\r\n* @zzzzwwjj made their first contribution in https://github.com/vllm-project/vllm/pull/31674\r\n* @Catacomba made their first contribution in https://github.com/vllm-project/vllm/pull/30322\r\n* @kunpengW-code made their first contribution in https://github.com/vllm-project/vllm/pull/31669\r\n* @johncalesp made their first contribution in https://github.com/vllm-project/vllm/pull/28874\r\n* @BlankRH made their first contribution in https://github.com/vllm-project/vllm/pull/31800\r\n* @guicho271828 made their first contribution in https://github.com/vllm-project/vllm/pull/20610\r\n* @ReinforcedKnowledge made their first contribution in https://github.com/vllm-project/vllm/pull/31055\r\n* @vSeamar made their first contribution in https://github.com/vllm-project/vllm/pull/29197\r\n* @A1c0r-Z made their first contribution in https://github.com/vllm-project/vllm/pull/31656\r\n* @MrIceCreamMan made their first contribution in https://github.com/vllm-project/vllm/pull/31465\r\n* @tianshu-Michael-yu made their first contribution in https://github.com/vllm-project/vllm/pull/31841\r\n* @weiyu0824 made their first contribution in https://github.com/vllm-project/vllm/pull/30808\r\n* @andyl98 made their first contribution in https://github.com/vllm-project/vllm/pull/31757\r\n* @JaredforReal made their first contribution in https://github.com/vllm-project/vllm/pull/31779\r\n* @katec846 made their first contribution in https://github.com/vllm-project/vllm/pull/29213\r\n* @kfirtoledo made their first contribution in https://github.com/vllm-project/vllm/pull/30761\r\n* @Ayobami-00 made their first contribution in https://github.com/vllm-project/vllm/pull/31868\r\n* @ShaanveerS made their first contribution in https://github.com/vllm-project/vllm/pull/31825\r\n* @Zyyeric made their first contribution in https://github.com/vllm-project/vllm/pull/31652\r\n* @wangshangsam made their first contribution in https://github.com/vllm-project/vllm/pull/31775\r\n* @devbyteai made their first contribution in https://github.com/vllm-project/vllm/pull/31536\r\n* @BJWang-ant made their first contribution in https://github.com/vllm-project/vllm/pull/31719\r\n* @dangoldbj made their first contribution in https://github.com/vllm-project/vllm/pull/31847\r\n* @maylikenoother made their first contribution in https://github.com/vllm-project/vllm/pull/31610\r\n* @yxing-bj made their first contribution in https://github.com/vllm-project/vllm/pull/31575\r\n* @xbfs made their first contribution in https://github.com/vllm-project/vllm/pull/31948\r\n* @RunkaiTao made their first contribution in https://github.com/vllm-project/vllm/pull/29354\r\n* @AkshatSh made their first contribution in https://github.com/vllm-project/vllm/pull/31550\r\n* @frelam made their first contribution in https://github.com/vllm-project/vllm/pull/31857\r\n* @shyeh25 made their first contribution in https://github.com/vllm-project/vllm/pull/31617\r\n* @andikarachman made their first contribution in https://github.com/vllm-project/vllm/pull/32092\r\n* @minimAluminiumalism made their first contribution in https://github.com/vllm-project/vllm/pull/32158\r\n* @andyzhangx made their first contribution in https://github.com/vllm-project/vllm/pull/32185\r\n* @sanghoon-yn made their first contribution in https://github.com/vllm-project/vllm/pull/31956\r\n* @potatosalad made their first contribution in https://github.com/vllm-project/vllm/pull/32212\r\n\r\n**Full Changelog**: https://github.com/vllm-project/vllm/compare/v0.13.0...v0.14.0","reactions":{"url":"https://api.github.com/repos/vllm-project/vllm/releases/278179519/reactions","total_count":75,"+1":7,"-1":0,"laugh":0,"hooray":43,"confused":0,"heart":20,"rocket":5,"eyes":0},"mentions_count":86},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/271634343","assets_url":"https://api.github.com/repos/vllm-project/vllm/releases/271634343/assets","upload_url":"https://uploads.github.com/repos/vllm-project/vllm/releases/271634343/assets{?name,label}","html_url":"https://github.com/vllm-project/vllm/releases/tag/v0.13.0","id":271634343,"author":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"node_id":"RE_kwDOI7xefs4QMM-n","tag_name":"v0.13.0","target_commitish":"main","name":"v0.13.0","draft":false,"immutable":false,"prerelease":false,"created_at":"2025-12-18T22:07:04Z","updated_at":"2025-12-22T10:25:17Z","published_at":"2025-12-19T03:02:22Z","assets":[{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/330489553","id":330489553,"node_id":"RA_kwDOI7xefs4Tst7R","name":"vllm-0.13.0+cpu-cp38-abi3-manylinux_2_35_aarch64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":26314465,"digest":"sha256:f7ba82dd22fe6246fba6c0c2d3136f1df5be0b65d185e707ac4fd69d5e314b5f","download_count":338,"created_at":"2025-12-19T03:03:04Z","updated_at":"2025-12-19T03:03:17Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.13.0/vllm-0.13.0%2Bcpu-cp38-abi3-manylinux_2_35_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/330489554","id":330489554,"node_id":"RA_kwDOI7xefs4Tst7S","name":"vllm-0.13.0+cpu-cp38-abi3-manylinux_2_35_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":30805855,"digest":"sha256:fa925006b68342cd7099d7291ff827f6e59c58dd4ccb5849a1419da4a15428c8","download_count":1426,"created_at":"2025-12-19T03:03:04Z","updated_at":"2025-12-19T03:03:30Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.13.0/vllm-0.13.0%2Bcpu-cp38-abi3-manylinux_2_35_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/330489557","id":330489557,"node_id":"RA_kwDOI7xefs4Tst7V","name":"vllm-0.13.0+cu130-cp38-abi3-manylinux_2_35_aarch64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":258218891,"digest":"sha256:e82666d57729e3c58b537f0082e592ce2d9b0fefd05dcb07571abf5165878074","download_count":3708,"created_at":"2025-12-19T03:03:04Z","updated_at":"2025-12-19T03:07:01Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.13.0/vllm-0.13.0%2Bcu130-cp38-abi3-manylinux_2_35_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/330489555","id":330489555,"node_id":"RA_kwDOI7xefs4Tst7T","name":"vllm-0.13.0+cu130-cp38-abi3-manylinux_2_35_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":267480120,"digest":"sha256:8b2993cbaaebe23eafc3449065138b43b6aba9aea955f18944f41585908db768","download_count":18208,"created_at":"2025-12-19T03:03:04Z","updated_at":"2025-12-19T03:07:05Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.13.0/vllm-0.13.0%2Bcu130-cp38-abi3-manylinux_2_35_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/330489556","id":330489556,"node_id":"RA_kwDOI7xefs4Tst7U","name":"vllm-0.13.0-cp38-abi3-manylinux_2_31_aarch64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":442031402,"digest":"sha256:464b722c5c5d67a39593ada4a228f7558e860a732cb74a3bfa61c1b442b57581","download_count":1809,"created_at":"2025-12-19T03:03:04Z","updated_at":"2025-12-19T03:07:42Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.13.0/vllm-0.13.0-cp38-abi3-manylinux_2_31_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/330489637","id":330489637,"node_id":"RA_kwDOI7xefs4Tst8l","name":"vllm-0.13.0-cp38-abi3-manylinux_2_31_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":474942618,"digest":"sha256:12b3d0a3b91c32a0091349de64b464f1c3d499a5b3a5d0ec387fef94ed5df6ee","download_count":24297,"created_at":"2025-12-19T03:03:17Z","updated_at":"2025-12-19T03:06:11Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.13.0/vllm-0.13.0-cp38-abi3-manylinux_2_31_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/330489670","id":330489670,"node_id":"RA_kwDOI7xefs4Tst9G","name":"vllm-0.13.0.tar.gz","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/x-gtar","state":"uploaded","size":17828199,"digest":"sha256:4ad43db45fef37114b550d03a4f423fb3fa3a31d8bc09ee810ef8b9cdcd4b5fe","download_count":4051,"created_at":"2025-12-19T03:03:31Z","updated_at":"2025-12-19T03:03:42Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.13.0/vllm-0.13.0.tar.gz"}],"tarball_url":"https://api.github.com/repos/vllm-project/vllm/tarball/v0.13.0","zipball_url":"https://api.github.com/repos/vllm-project/vllm/zipball/v0.13.0","body":"# vLLM v0.13.0 Release Notes Highlights\r\n\r\n## Highlights\r\n\r\nThis release features **442 commits from 207 contributors (61 new contributors)!**\r\n\r\n**Breaking Changes**: This release includes deprecation removals, PassConfig flag renames, and attention configuration changes from environment variables to CLI arguments. Please review the breaking changes section carefully before upgrading.\r\n\r\n### Model Support\r\n* **New models**: BAGEL (AR only) (#28439), AudioFlamingo3 (#30539), JAIS 2 (#30188), latent MoE architecture support (#30203).\r\n* **Tool parsers**: DeepSeek-V3.2 (#29848), Gigachat 3 (#29905), Holo2 reasoning (#30048).\r\n* **Model enhancements**: Qwen3-VL embeddings support (#30037), Qwen3-VL EVS (Efficient Video Sampling) (#29752), DeepSeek V3.2 proper `drop_thinking` logic (#30490), DeepSeek V3.2 top-k fix (#27568).\r\n* **Task expansion**: Automatic TokenClassification model conversion (#30666), Ultravox v0.7 transformer projector (#30089).\r\n* **Quantization**: BitsAndBytes for Qwen3-Omni-MoE (#29896).\r\n* **Speculative decoding**: Eagle/Eagle3 Transformers backend (#30340), Mamba `selective_state_update` spec decode (#29488).\r\n\r\n### Engine Core\r\n* **Compilation**: Conditional compilation via `compile_ranges` for selective kernel compilation (#24252).\r\n* **Prefix caching**: xxHash high-performance hash option (#29163).\r\n* **Attention**: PrefixLM support for FlexAttention (#27938) and TritonAttention (#30386), CUDA graphs for 3D Triton attention (#28306), `TRITON_MLA` without prefix-caching (#29125).\r\n* **Batch invariance**: FA2 and LoRA batch-invariant support (#30018).\r\n* **Pooling**: Chunked prefill for ALL pooling tasks (#27145), multi-vector retrieval API (#26686).\r\n* **Model Runner V2**: Min-p sampling (#30171), NaN detection in logits (#30187).\r\n* **Speculative decoding**: Medusa GPU-CPU sync avoidance (#29723), async spec-decode improvements (#29624).\r\n* **Whisper**: Major performance improvements - [V1 is now faster than V0](https://github.com/vllm-project/vllm/issues/24946#issuecomment-3680725754) (~3x speedup vs v0.12.0). Encoder batching (#29421), `FULL_DECODE_ONLY` CUDA graph (#30072), CPU backend support (#30062).\r\n* **Performance**: Fused blockwise quant RMS norm (#27883), MoE LoRA loading reduction (#30243), encoder cache optimization (#30475), CPU KV offloading streams (#29013).\r\n\r\n### Hardware & Performance\r\n* **NVIDIA Blackwell Ultra**: SM103 (GB300) support with CUDA 13 (#30484).\r\n* **DeepSeek optimizations** (benchmarked on DeepSeek-V3.1):\r\n  - DeepEP High-Throughput CUDA graph enabled by default: **5.3% throughput, 4.4% TTFT improvement** (#29558)\r\n  - DeepGEMM fused layout kernel: **4.3% throughput, 10.7% TTFT improvement** (#29546)\r\n  - DeepGEMM experts initialization: **3.9% TTFT improvement** (#30494)\r\n  - `group_topk` kernel: **1.9% throughput, 2.1% TPOT improvement** (#30159)\r\n  - Sparse prefill kernel for FP8 KV-cache in DeepSeek-V3.2 (#27532)\r\n  - MLA FP8 optimization with ReduceScatterSum (#29795), direct k_nope/k_pe copy (#29710)\r\n* **CPU**: Whisper support (#30062), Arm Optimized Routines vectorized exp (#30068), x86 CPU wheel pipeline (#28848).\r\n* **AMD ROCm**: Aiter quantization kernels (#25552), torch.compile layernorm/silu + FP8 quant (#25693), Triton ScaledMM fallback (#26668), MXFP4 w4a4 inference (#29775).\r\n* **Intel XPU**: wNa16 compressed tensors (#29484).\r\n* **Build**: CUDA 13 aarch64 wheels (#30341), Docker kernel build stage (#29452), Ascend NPU Docker (#30015).\r\n\r\n### Large Scale Serving & Disaggregated Prefill/Decode\r\n* **KV connectors**: Mooncake Transfer Engine (#24718), cache reset via `/reset_prefix_cache` (#27170), KV events (#28309), failure recovery config (#26813).\r\n* **NIXL**: Compatibility checking in handshake (#29503), large batch proxy support (#28782).\r\n* **EPLB**: NVFP4 support (#29804), algorithm abstraction (#26471).\r\n* **Multi-node**: External launcher mode (#29833).\r\n* **Hybrid allocator**: Optional KV connector integration (#29805).\r\n* **Performance**: silu_mul_per_token_group_quant_fp8 kernel for DP/EP (#29470).\r\n\r\n### Quantization\r\n* **New**: W4A8 grouped GEMM on Hopper (#29691), online FP8 with streaming post-processing (#29196), FP8 weight reloading for RLHF (#28480).\r\n* **MoE + LoRA**: AWQ Marlin (#30442) and GPTQ Marlin (#30254) support.\r\n* **GGUF**: MoE + GGUF restored for Qwen3 MoE (#30116), Qwen2 MoE (#30307), HF defaults override (#30118).\r\n* **Compatibility**: Transformers v5 RoPE support (#30046).\r\n\r\n### API & Frontend\r\n* **Responses API**: MCP type infrastructure (#30054), Browser/Container MCP tools (#29989), full MCP Python loop (#29798), extra body parameters (#30532).\r\n* **Configuration**: `AttentionConfig` replaces `VLLM_ATTENTION_BACKEND` env var (#26315).\r\n* **Chat templates**: DeepSeek-V3.2 (#29837), DeepSeek-V3.2 developer tools (#30040).\r\n* **Anthropic API**: Streaming fixes (#29971, #30266).\r\n* **Embeddings**: Binary format with `encoding_format=bytes_only` (#30249), multiple image/audio per request (#29988), tokenization_kwargs override (#29794).\r\n* **Metrics**: Prefill KV compute metric excluding cached tokens (#30189).\r\n* **Profiling**: Layer-wise NVTX (#29990), profiling CLI config (#29912).\r\n* **UX**: Better OOM errors (#28051), ModelConfig validation (#30213), distributed executor errors (#30140).\r\n\r\n### Security\r\n* Additional protection for CVE-2025-62164 (#30649).\r\n\r\n### Dependencies\r\n* NVSHMEM 3.3.24 + CUDA 13 fix (#30149).\r\n* TPU tpu-inference 0.12.0 (#30221).\r\n\r\n### Breaking Changes & Deprecations\r\n1. **PassConfig flags renamed** per RFC #27995 (#29646)\r\n2. **Attention env vars → CLI args**: `VLLM_ATTENTION_BACKEND` replaced with `--attention-backend` (#26315)\r\n3. **Removed `-O.xx` flag** (#29991)\r\n4. **Removed deprecated plugin/compilation fields** (#30396)\r\n5. **Removed deprecated task, seed, MM settings** (#30397)\r\n6. **Removed `embed_input_ids`/`embed_multimodal` fallbacks** (#30458)\r\n7. **Removed tokenizer setter** (#30400)\r\n8. **Deprecations**: `merge_by_field_config` (#30035, #30170), `--convert reward` → `--convert embed` (#30463)\r\n\r\n## New Contributors 🎉\r\n\r\n* @ajpqs made their first contribution in https://github.com/vllm-project/vllm/pull/29905\r\n* @amitz-nv made their first contribution in https://github.com/vllm-project/vllm/pull/29978\r\n* @amrmahdi made their first contribution in https://github.com/vllm-project/vllm/pull/29452\r\n* @andrewbriand made their first contribution in https://github.com/vllm-project/vllm/pull/29804\r\n* @anker-c2 made their first contribution in https://github.com/vllm-project/vllm/pull/30344\r\n* @AuruTus made their first contribution in https://github.com/vllm-project/vllm/pull/30182\r\n* @avigny made their first contribution in https://github.com/vllm-project/vllm/pull/19425\r\n* @Bhanu068 made their first contribution in https://github.com/vllm-project/vllm/pull/30254\r\n* @Copilot made their first contribution in https://github.com/vllm-project/vllm/pull/29025\r\n* @dbotwinick made their first contribution in https://github.com/vllm-project/vllm/pull/30583\r\n* @dependabot[bot] made their first contribution in https://github.com/vllm-project/vllm/pull/30234\r\n* @desertfire made their first contribution in https://github.com/vllm-project/vllm/pull/29919\r\n* @dmitry-tokarev-nv made their first contribution in https://github.com/vllm-project/vllm/pull/30149\r\n* @drslark made their first contribution in https://github.com/vllm-project/vllm/pull/30632\r\n* @dtcccc made their first contribution in https://github.com/vllm-project/vllm/pull/24718\r\n* @elizabetht made their first contribution in https://github.com/vllm-project/vllm/pull/28671\r\n* @Elm8116 made their first contribution in https://github.com/vllm-project/vllm/pull/30068\r\n* @gausah01 made their first contribution in https://github.com/vllm-project/vllm/pull/29604\r\n* @gh-wf made their first contribution in https://github.com/vllm-project/vllm/pull/30285\r\n* @hdlj-h made their first contribution in https://github.com/vllm-project/vllm/pull/30056\r\n* @HF-001 made their first contribution in https://github.com/vllm-project/vllm/pull/30051\r\n* @hzxuzhonghu made their first contribution in https://github.com/vllm-project/vllm/pull/29931\r\n* @JaviS-Rei made their first contribution in https://github.com/vllm-project/vllm/pull/29882\r\n* @johannesflommersfeld made their first contribution in https://github.com/vllm-project/vllm/pull/30390\r\n* @KevinMusgrave made their first contribution in https://github.com/vllm-project/vllm/pull/30529\r\n* @kitaekatt made their first contribution in https://github.com/vllm-project/vllm/pull/30408\r\n* @lashahub made their first contribution in https://github.com/vllm-project/vllm/pull/30539\r\n* @LuminolT made their first contribution in https://github.com/vllm-project/vllm/pull/29163\r\n* @majiayu000 made their first contribution in https://github.com/vllm-project/vllm/pull/30615\r\n* @MaoJianwei made their first contribution in https://github.com/vllm-project/vllm/pull/29797\r\n* @Mercykid-bash made their first contribution in https://github.com/vllm-project/vllm/pull/26471\r\n* @mgehre-amd made their first contribution in https://github.com/vllm-project/vllm/pull/30364\r\n* @mivehk made their first contribution in https://github.com/vllm-project/vllm/pull/30512\r\n* @mondaylord made their first contribution in https://github.com/vllm-project/vllm/pull/30671\r\n* @noa-neria made their first contribution in https://github.com/vllm-project/vllm/pull/29320\r\n* @PatrykSaffer made their first contribution in https://github.com/vllm-project/vllm/pull/30330\r\n* @Peng-YM made their first contribution in https://github.com/vllm-project/vllm/pull/29074\r\n* @realliujiaxu made their first contribution in https://github.com/vllm-project/vllm/pull/30059\r\n* @redwrasse made their first contribution in https://github.com/vllm-project/vllm/pull/29261\r\n* @Ri0S made their first contribution in https://github.com/vllm-project/vllm/pull/30532\r\n* @sarathc-cerebras made their first contribution in https://github.com/vllm-project/vllm/pull/30188\r\n* @scratch-ml made their first contribution in https://github.com/vllm-project/vllm/pull/30351\r\n* @seokhyunan made their first contribution in https://github.com/vllm-project/vllm/pull/30648\r\n* @shaharmor98 made their first contribution in https://github.com/vllm-project/vllm/pull/30203\r\n* @taoyun951753 made their first contribution in https://github.com/vllm-project/vllm/pull/30037\r\n* @tom-zju made their first contribution in https://github.com/vllm-project/vllm/pull/30057\r\n* @tomtomjhj made their first contribution in https://github.com/vllm-project/vllm/pull/29692\r\n* @vkuzo made their first contribution in https://github.com/vllm-project/vllm/pull/29196\r\n* @vladnosiv made their first contribution in https://github.com/vllm-project/vllm/pull/30490\r\n* @weiguihua2 made their first contribution in https://github.com/vllm-project/vllm/pull/30042\r\n* @wenqiglantz made their first contribution in https://github.com/vllm-project/vllm/pull/30649\r\n* @wkcn made their first contribution in https://github.com/vllm-project/vllm/pull/29879\r\n* @wu-kan made their first contribution in https://github.com/vllm-project/vllm/pull/21804\r\n* @wz1qqx made their first contribution in https://github.com/vllm-project/vllm/pull/30376\r\n* @xyDong0223 made their first contribution in https://github.com/vllm-project/vllm/pull/30455\r\n* @yifant-code made their first contribution in https://github.com/vllm-project/vllm/pull/30213\r\n* @yjc9696 made their first contribution in https://github.com/vllm-project/vllm/pull/30040\r\n* @yurekami made their first contribution in https://github.com/vllm-project/vllm/pull/30552\r\n* @yuttian1 made their first contribution in https://github.com/vllm-project/vllm/pull/30102\r\n* @ZhijianJiang made their first contribution in https://github.com/vllm-project/vllm/pull/30219\r\n* @ZhiweiYan-96 made their first contribution in https://github.com/vllm-project/vllm/pull/29773\r\n\r\n**Full Changelog**: https://github.com/vllm-project/vllm/compare/v0.12.0...v0.13.0","reactions":{"url":"https://api.github.com/repos/vllm-project/vllm/releases/271634343/reactions","total_count":84,"+1":16,"-1":0,"laugh":2,"hooray":40,"confused":0,"heart":11,"rocket":13,"eyes":2},"mentions_count":60},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/267016722","assets_url":"https://api.github.com/repos/vllm-project/vllm/releases/267016722/assets","upload_url":"https://uploads.github.com/repos/vllm-project/vllm/releases/267016722/assets{?name,label}","html_url":"https://github.com/vllm-project/vllm/releases/tag/v0.12.0","id":267016722,"author":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"node_id":"RE_kwDOI7xefs4P6loS","tag_name":"v0.12.0","target_commitish":"main","name":"v0.12.0","draft":false,"immutable":false,"prerelease":false,"created_at":"2025-12-03T04:38:43Z","updated_at":"2025-12-06T01:58:43Z","published_at":"2025-12-03T09:36:17Z","assets":[{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/323798615","id":323798615,"node_id":"RA_kwDOI7xefs4TTMZX","name":"vllm-0.12.0+cu130-cp38-abi3-manylinux_2_31_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":264860903,"digest":"sha256:d502464ebd285911cbd5403d00be2f8202bda0aa2105a1e535a274df09b1554e","download_count":3445,"created_at":"2025-12-03T09:38:32Z","updated_at":"2025-12-03T09:38:38Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.12.0/vllm-0.12.0%2Bcu130-cp38-abi3-manylinux_2_31_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/323798433","id":323798433,"node_id":"RA_kwDOI7xefs4TTMWh","name":"vllm-0.12.0-cp38-abi3-manylinux_2_31_aarch64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":435774675,"digest":"sha256:44684361b22b12898fb356074cc3e8786036f9f1e17ea1105bd27d1ffa63cbe9","download_count":212,"created_at":"2025-12-03T09:38:15Z","updated_at":"2025-12-03T09:38:26Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.12.0/vllm-0.12.0-cp38-abi3-manylinux_2_31_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/323798145","id":323798145,"node_id":"RA_kwDOI7xefs4TTMSB","name":"vllm-0.12.0-cp38-abi3-manylinux_2_31_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":466530617,"digest":"sha256:596673edcac5ea8ab8cb2d9b0200052978cd91d78a054926d9d0c6895d34237f","download_count":2785,"created_at":"2025-12-03T09:37:42Z","updated_at":"2025-12-03T09:37:52Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.12.0/vllm-0.12.0-cp38-abi3-manylinux_2_31_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/323799078","id":323799078,"node_id":"RA_kwDOI7xefs4TTMgm","name":"vllm-0.12.0.tar.gz","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/x-gtar","state":"uploaded","size":17594403,"digest":"sha256:dc2863be80260f378f1e72f9b82b4009610e434d05fdd7067383b231b89c4f4a","download_count":385,"created_at":"2025-12-03T09:39:22Z","updated_at":"2025-12-03T09:39:23Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.12.0/vllm-0.12.0.tar.gz"}],"tarball_url":"https://api.github.com/repos/vllm-project/vllm/tarball/v0.12.0","zipball_url":"https://api.github.com/repos/vllm-project/vllm/zipball/v0.12.0","body":"# vLLM v0.12.0 Release Notes Highlights\r\n\r\n## Highlights\r\n\r\nThis release features 474 commits from 213 contributors (57 new)！\r\n\r\n**Breaking Changes**: This release includes PyTorch 2.9.0 upgrade (CUDA 12.9), V0 deprecations including `xformers` backend, and scheduled removals - please review the changelog carefully.\r\n\r\n**Major Features**:\r\n* **EAGLE Speculative Decoding Improvements**: Multi-step CUDA graph support (#29559), DP>1 support (#26086), and multimodal support with Qwen3VL (#29594).\r\n* **Significant Performance Optimizations**: 18.1% throughput improvement from batch invariant BMM (#29345), 2.2% throughput improvement from shared experts overlap (#28879).\r\n* **AMD ROCm Expansion**: DeepSeek v3.2 + SparseMLA support (#26670), FP8 MLA decode (#28032), AITER attention backend (#28701).\r\n\r\n### Model Support\r\n\r\n* **New model families**: PLaMo-3 (#28834), OpenCUA-7B (#29068), HunyuanOCR (#29327), Mistral Large 3 and Ministral 3 (#29757).\r\n* **Format support**: Gemma3 GGUF multimodal support (#27772).\r\n* **Multimodal enhancements**: Qwen3 Omni audio-in-video support (#27721), Eagle3 multimodal support for Qwen3VL (#29594).\r\n* **Performance**: QwenVL cos/sin cache optimization (#28798).\r\n\r\n### Engine Core\r\n\r\n* **GPU Model Runner V2 (Experimental)** (#25266): Complete refactoring of model execution pipeline:\r\n  - No \"reordering\" or complex bookkeeping with persistent batch removal\r\n  - GPU-persistent block tables for better scalability with `max_model_len` and `num_kv_groups`\r\n  - Triton-native sampler: no -1 temperature hack, efficient per-request seeds, memory-efficient prompt logprobs\r\n  - Simplified DP and CUDA graph implementations\r\n  - Efficient structured outputs support\r\n\r\n* **Prefill Context Parallel (PCP) (Preparatory)** (#28718): Partitions the sequence dimension during prefill for improved long-sequence inference. Complements existing Decode Context Parallel (DCP). See RFC #25749 for details.\r\n\r\n* **RLHF Support**: Pause and Resume Generation for Asynchronous RL Training (#28037).\r\n\r\n* **KV Cache Enhancements**: Cross-layer KV blocks support (#27743), KV cache residency metrics (#27793).\r\n\r\n* **Audio support**: Audio embeddings support in chat completions (#29059).\r\n\r\n* **Speculative Decoding**:\r\n  - Multi-step Eagle with CUDA graph (#29559)\r\n  - EAGLE DP>1 support (#26086)\r\n  - EAGLE3 heads without `use_aux_hidden_states` (#27688)\r\n  - Eagle multimodal CUDA graphs with MRoPE (#28896)\r\n  - Logprobs support with spec decode + async scheduling (#29223)\r\n\r\n* **Configuration**: Flexible `inputs_embeds_size` separate from `hidden_size` (#29741), `--fully-sharded-loras` for fused_moe (#28761).\r\n\r\n### Hardware & Performance\r\n\r\n* **NVIDIA Performance**:\r\n  - **Batch invariant BMM optimization**: 18.1% throughput improvement, 10.7% TTFT improvement on DeepSeek-V3.1 (#29345)\r\n  - **Shared Experts Overlap with FlashInfer DeepGEMM**: 2.2% throughput improvement, 3.6% TTFT improvement at batch size 32 (#28879)\r\n  - DeepGEMM N dim restriction reduced from 128 to 64 multiplier (#28687)\r\n  - DeepEP low-latency with round-robin expert placement (#28449)\r\n  - NVFP4 MoE CUTLASS support for SM120 (#29242)\r\n  - H200 Fused MoE Config improvements (#28992)\r\n\r\n* **AMD ROCm**:\r\n  - DeepSeek v3.2 and SparseMLA support (#26670)\r\n  - FP8 MLA decode support (#28032)\r\n  - AITER sampling ops integration (#26084)\r\n  - AITER triton attention backend (#28701)\r\n  - Bitsandbytes quantization on AMD GPUs with warp size 32 (#27307)\r\n  - Fastsafetensors support (#28225)\r\n  - Sliding window support for AiterFlashAttentionBackend (#29234)\r\n  - Whisper v1 with Aiter Unified/Flash Attention (#28376)\r\n\r\n* **CPU**:\r\n  - Paged attention GEMM acceleration on ARM CPUs with NEON (#29193)\r\n  - Parallelize over tokens in int4 MoE (#29600)\r\n  - CPU all reduce optimization for async_scheduling + DP>1 (#29311)\r\n\r\n* **Attention**: FlashAttention ViT support, now default backend (#28763).\r\n\r\n* **Long Context**: Optimized `gather_and_maybe_dequant_cache` kernel for extremely long sequences (#28029).\r\n\r\n* **Multi-NUMA**: Enhanced NUMA functionality for systems with multiple NUMA nodes per socket (#25559).\r\n\r\n* **Docker**: Image size reduced by ~200MB (#29060).\r\n\r\n### Quantization\r\n\r\n* **W4A8**: Marlin kernel support (#24722).\r\n* **NVFP4**: \r\n  - MoE CUTLASS support for SM120 (#29242)\r\n  - TRTLLM MoE NVFP4 kernel (#28892)\r\n  - CuteDSL MoE with NVFP4 DeepEP dispatch (#27141)\r\n  - Non-gated activations support in modelopt path (#29004)\r\n* **AWQ**: Compressed-tensors AWQ support for Turing GPUs (#29732).\r\n* **LoRA**: FusedMoE LoRA Triton kernel for MXFP4 (#29708).\r\n* **Online quantization**: Moved to `model.load_weights` (#26327).\r\n\r\n### API & Frontend\r\n\r\n* **Responses API**: \r\n  - Multi-turn support for non-harmony requests (#29175)\r\n  - Reasoning item input parsing (#28248)\r\n* **Tool Calling**:\r\n  - Parsed tool arguments support (#28820)\r\n  - `parallel_tool_calls` param compliance (#26233)\r\n  - Tool filtering support in ToolServer (#29224)\r\n* **Whisper**: `verbose_json` and `timestamp` features for transcription/translation (#24209).\r\n* **Sampling**: Flat logprob control moved from env var to `SamplingParams` (#28914).\r\n* **GGUF**: Improved HuggingFace loading UX with `repo_id:quant_type` syntax (#29137).\r\n* **Profiling**: Iteration-level profiling for Torch and CUDA profiler (#28987).\r\n* **Logs**: Colorized log output (#29017).\r\n* **Optimization Levels**: `-O0`, `-O1`, `-O2`, `-O3` allow trading startup time for performance, more compilation flags will be added in future releases (#26847)\r\n\r\n### Dependencies\r\n\r\n* **PyTorch 2.9.0** with CUDA 12.9 (#24994) - **Breaking change** requiring environment updates.\r\n* **xgrammar**: Updated to 0.1.27 (#28221).\r\n* **Transformers**: Updated to 4.57.3 (#29418), preparation for v5 with `rope_parameters` (#28542).\r\n* **XPU**: torch & IPEX 2.9 upgrade (#29307).\r\n\r\n### V0 Deprecation & Breaking Changes\r\n\r\n**Removed Parameters**:\r\n* `num_lookahead_slots` (#29000)\r\n* `best_of` (#29090)\r\n* LoRA extra vocab (#28545)\r\n\r\n**Deprecated**:\r\n* `xformers` backend (#29262)\r\n* `seed=None` (#29185)\r\n\r\n**Scheduled Removals** (will be removed in future release):\r\n* `ParallelConfig`'s direct child EPLB fields (#29324)\r\n* `guided_*` config fields (#29326)\r\n* `override_pooler_config` and `disable_log_requests` (#29402)\r\n* `CompilationConfig.use_inductor` (#29323)\r\n* Deprecated metrics (#29330)\r\n\r\n**Other Breaking Changes**:\r\n* PyTorch 2.9.0 upgrade requires CUDA 12.9 environment\r\n* Mistral format auto-detection for model loading (#28659)\r\n\r\n\r\n## New Contributors\r\n* @jesse996 made their first contribution in https://github.com/vllm-project/vllm/pull/28846\r\n* @Nepherpitou made their first contribution in https://github.com/vllm-project/vllm/pull/28960\r\n* @Samoed made their first contribution in https://github.com/vllm-project/vllm/pull/27329\r\n* @j20120307 made their first contribution in https://github.com/vllm-project/vllm/pull/28999\r\n* @vnadathur made their first contribution in https://github.com/vllm-project/vllm/pull/26468\r\n* @zhyajie made their first contribution in https://github.com/vllm-project/vllm/pull/28942\r\n* @IzzyPutterman made their first contribution in https://github.com/vllm-project/vllm/pull/28896\r\n* @rjrock-amd made their first contribution in https://github.com/vllm-project/vllm/pull/28905\r\n* @zq1997 made their first contribution in https://github.com/vllm-project/vllm/pull/27715\r\n* @shengliangxu made their first contribution in https://github.com/vllm-project/vllm/pull/28076\r\n* @prashanth058 made their first contribution in https://github.com/vllm-project/vllm/pull/28972\r\n* @qgallouedec made their first contribution in https://github.com/vllm-project/vllm/pull/28820\r\n* @zhanggzh made their first contribution in https://github.com/vllm-project/vllm/pull/19347\r\n* @pandalee99 made their first contribution in https://github.com/vllm-project/vllm/pull/26628\r\n* @dsuhinin made their first contribution in https://github.com/vllm-project/vllm/pull/29100\r\n* @xli made their first contribution in https://github.com/vllm-project/vllm/pull/29124\r\n* @jeremyteboul made their first contribution in https://github.com/vllm-project/vllm/pull/29059\r\n* @soodoshll made their first contribution in https://github.com/vllm-project/vllm/pull/28875\r\n* @bhagyashrigai made their first contribution in https://github.com/vllm-project/vllm/pull/28957\r\n* @skaraban3807 made their first contribution in https://github.com/vllm-project/vllm/pull/25559\r\n* @Victor49152 made their first contribution in https://github.com/vllm-project/vllm/pull/28892\r\n* @rjrock made their first contribution in https://github.com/vllm-project/vllm/pull/29205\r\n* @FlintyLemming made their first contribution in https://github.com/vllm-project/vllm/pull/29182\r\n* @madskildegaard made their first contribution in https://github.com/vllm-project/vllm/pull/29175\r\n* @nandan2003 made their first contribution in https://github.com/vllm-project/vllm/pull/29189\r\n* @michaelact made their first contribution in https://github.com/vllm-project/vllm/pull/29173\r\n* @yongming-qin made their first contribution in https://github.com/vllm-project/vllm/pull/28958\r\n* @joshiemoore made their first contribution in https://github.com/vllm-project/vllm/pull/29249\r\n* @lim4349 made their first contribution in https://github.com/vllm-project/vllm/pull/29068\r\n* @apinge made their first contribution in https://github.com/vllm-project/vllm/pull/28376\r\n* @gbyu-amd made their first contribution in https://github.com/vllm-project/vllm/pull/28032\r\n* @kflu made their first contribution in https://github.com/vllm-project/vllm/pull/29364\r\n* @Inokinoki made their first contribution in https://github.com/vllm-project/vllm/pull/29200\r\n* @GOavi101 made their first contribution in https://github.com/vllm-project/vllm/pull/29313\r\n* @sts07142 made their first contribution in https://github.com/vllm-project/vllm/pull/29137\r\n* @ivanium made their first contribution in https://github.com/vllm-project/vllm/pull/29143\r\n* @geodavic made their first contribution in https://github.com/vllm-project/vllm/pull/28795\r\n* @Yejing-Lai made their first contribution in https://github.com/vllm-project/vllm/pull/29473\r\n* @Adityayxt made their first contribution in https://github.com/vllm-project/vllm/pull/29491\r\n* @guodongxiaren made their first contribution in https://github.com/vllm-project/vllm/pull/29620\r\n* @askliar made their first contribution in https://github.com/vllm-project/vllm/pull/29426\r\n* @scydas made their first contribution in https://github.com/vllm-project/vllm/pull/29589\r\n* @EanWang211123 made their first contribution in https://github.com/vllm-project/vllm/pull/29594\r\n* @qGentry made their first contribution in https://github.com/vllm-project/vllm/pull/29506\r\n* @HappyAmazonian made their first contribution in https://github.com/vllm-project/vllm/pull/29335\r\n* @rgommers made their first contribution in https://github.com/vllm-project/vllm/pull/29241\r\n* @staugust made their first contribution in https://github.com/vllm-project/vllm/pull/28840\r\n* @mertunsall made their first contribution in https://github.com/vllm-project/vllm/pull/29667\r\n* @dublc made their first contribution in https://github.com/vllm-project/vllm/pull/29728\r\n* @nwaughachukwuma made their first contribution in https://github.com/vllm-project/vllm/pull/29671\r\n* @BowTen made their first contribution in https://github.com/vllm-project/vllm/pull/29731\r\n* @omera-nv made their first contribution in https://github.com/vllm-project/vllm/pull/29004\r\n* @zhangruoxu made their first contribution in https://github.com/vllm-project/vllm/pull/29568\r\n* @KKKZOZ made their first contribution in https://github.com/vllm-project/vllm/pull/29783\r\n* @FredericOdermatt made their first contribution in https://github.com/vllm-project/vllm/pull/29784\r\n* @Abdennacer-Badaoui made their first contribution in https://github.com/vllm-project/vllm/pull/29782\r\n* @knlnguyen1802 made their first contribution in https://github.com/vllm-project/vllm/pull/28525\r\n* @finbarrtimbers made their first contribution in https://github.com/vllm-project/vllm/pull/29796\r\n* @hholtmann made their first contribution in https://github.com/vllm-project/vllm/pull/29711\r\n\r\n**Full Changelog**: https://github.com/vllm-project/vllm/compare/v0.11.1...v0.12.0","reactions":{"url":"https://api.github.com/repos/vllm-project/vllm/releases/267016722/reactions","total_count":172,"+1":62,"-1":0,"laugh":1,"hooray":65,"confused":0,"heart":2,"rocket":41,"eyes":1},"mentions_count":58},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/263888662","assets_url":"https://api.github.com/repos/vllm-project/vllm/releases/263888662/assets","upload_url":"https://uploads.github.com/repos/vllm-project/vllm/releases/263888662/assets{?name,label}","html_url":"https://github.com/vllm-project/vllm/releases/tag/v0.11.2","id":263888662,"author":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"node_id":"RE_kwDOI7xefs4Pup8W","tag_name":"v0.11.2","target_commitish":"main","name":"v0.11.2","draft":false,"immutable":false,"prerelease":false,"created_at":"2025-11-19T22:11:21Z","updated_at":"2025-11-20T07:45:04Z","published_at":"2025-11-20T07:29:19Z","assets":[{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/318704059","id":318704059,"node_id":"RA_kwDOI7xefs4S_wm7","name":"vllm-0.11.2+cu129-cp38-abi3-manylinux1_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":398964798,"digest":"sha256:6d4a063c411a07b24af5b4785c2a8f71a37f9283c66e1f1d7a8efabfc00ff268","download_count":7811,"created_at":"2025-11-20T07:29:53Z","updated_at":"2025-11-20T07:30:06Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.11.2/vllm-0.11.2%2Bcu129-cp38-abi3-manylinux1_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/318704157","id":318704157,"node_id":"RA_kwDOI7xefs4S_wod","name":"vllm-0.11.2+cu130-cp38-abi3-manylinux1_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":230664790,"digest":"sha256:d0ca7af20dac780f0bf188daf41a261a6c1cdf78794ee690382de31d508e3fbf","download_count":719,"created_at":"2025-11-20T07:30:14Z","updated_at":"2025-11-20T07:30:21Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.11.2/vllm-0.11.2%2Bcu130-cp38-abi3-manylinux1_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/318703974","id":318703974,"node_id":"RA_kwDOI7xefs4S_wlm","name":"vllm-0.11.2-cp38-abi3-manylinux1_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":370306629,"digest":"sha256:ea473bd4fde06940fe3f681a00476060652f62b3279ef11aaffac5768856cfe8","download_count":763,"created_at":"2025-11-20T07:29:33Z","updated_at":"2025-11-20T07:29:45Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.11.2/vllm-0.11.2-cp38-abi3-manylinux1_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/318704359","id":318704359,"node_id":"RA_kwDOI7xefs4S_wrn","name":"vllm-0.11.2-cp38-abi3-manylinux2014_aarch64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":368543904,"digest":"sha256:a084f5ca768d22bf55810948cbb50825a35015e07593ab6c9c42fcbe18bdd5cc","download_count":237,"created_at":"2025-11-20T07:30:39Z","updated_at":"2025-11-20T07:30:51Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.11.2/vllm-0.11.2-cp38-abi3-manylinux2014_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/318703947","id":318703947,"node_id":"RA_kwDOI7xefs4S_wlL","name":"vllm-0.11.2.tar.gz","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/x-gtar","state":"uploaded","size":17287801,"digest":"sha256:496d15bb64ca0fe73adbc57a93b29f4671fa12404c09e0ba02f777bfe60af671","download_count":300,"created_at":"2025-11-20T07:29:22Z","updated_at":"2025-11-20T07:29:23Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.11.2/vllm-0.11.2.tar.gz"}],"tarball_url":"https://api.github.com/repos/vllm-project/vllm/tarball/v0.11.2","zipball_url":"https://api.github.com/repos/vllm-project/vllm/zipball/v0.11.2","body":"This release includes 4 bug fixes on top of `v0.11.1`:\r\n- [BugFix] Ray with multiple nodes (https://github.com/vllm-project/vllm/pull/28873)\r\n- [BugFix] Fix false assertion with spec-decode=[2,4,..] and TP>2 (https://github.com/vllm-project/vllm/pull/29036)\r\n- [BugFix] Fix async-scheduling + FlashAttn MLA (https://github.com/vllm-project/vllm/pull/28990)\r\n- [NVIDIA] Guard SM100 CUTLASS MoE macro to SM100 builds v2 (https://github.com/vllm-project/vllm/pull/28938)","reactions":{"url":"https://api.github.com/repos/vllm-project/vllm/releases/263888662/reactions","total_count":60,"+1":30,"-1":0,"laugh":0,"hooray":17,"confused":0,"heart":8,"rocket":1,"eyes":4}},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/263447752","assets_url":"https://api.github.com/repos/vllm-project/vllm/releases/263447752/assets","upload_url":"https://uploads.github.com/repos/vllm-project/vllm/releases/263447752/assets{?name,label}","html_url":"https://github.com/vllm-project/vllm/releases/tag/v0.11.1","id":263447752,"author":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"node_id":"RE_kwDOI7xefs4Ps-TI","tag_name":"v0.11.1","target_commitish":"main","name":"v0.11.1","draft":false,"immutable":false,"prerelease":false,"created_at":"2025-11-18T08:20:45Z","updated_at":"2025-11-20T03:10:18Z","published_at":"2025-11-18T23:03:42Z","assets":[{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/318090670","id":318090670,"node_id":"RA_kwDOI7xefs4S9a2u","name":"vllm-0.11.1+cu129-cp38-abi3-manylinux1_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":399372227,"digest":"sha256:3d9d03e81caf0eb964c50ef1c2fd7f9f3be278402c32f7d4f88191497e33b728","download_count":619,"created_at":"2025-11-18T23:08:02Z","updated_at":"2025-11-18T23:08:14Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.11.1/vllm-0.11.1%2Bcu129-cp38-abi3-manylinux1_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/318090825","id":318090825,"node_id":"RA_kwDOI7xefs4S9a5J","name":"vllm-0.11.1+cu130-cp38-abi3-manylinux1_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":230828290,"digest":"sha256:c0ffe05a6029985f41f84617d884f348e3dcd711c7b998c77cc39259da3738e3","download_count":268,"created_at":"2025-11-18T23:08:21Z","updated_at":"2025-11-18T23:08:27Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.11.1/vllm-0.11.1%2Bcu130-cp38-abi3-manylinux1_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/318090985","id":318090985,"node_id":"RA_kwDOI7xefs4S9a7p","name":"vllm-0.11.1-cp38-abi3-manylinux1_x86_64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":370708817,"digest":"sha256:077da60655466bab811f006ef93b90d45322e49fa0f204ffac6d48c32866b6d1","download_count":3637,"created_at":"2025-11-18T23:08:36Z","updated_at":"2025-11-18T23:08:47Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.11.1/vllm-0.11.1-cp38-abi3-manylinux1_x86_64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/318091191","id":318091191,"node_id":"RA_kwDOI7xefs4S9a-3","name":"vllm-0.11.1-cp38-abi3-manylinux2014_aarch64.whl","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/octet-stream","state":"uploaded","size":368946085,"digest":"sha256:f6fe38afe16f9008bbae6516dc78cf16a153fa470fc4a1c1e0c96c8caa0209f8","download_count":94,"created_at":"2025-11-18T23:08:57Z","updated_at":"2025-11-18T23:09:08Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.11.1/vllm-0.11.1-cp38-abi3-manylinux2014_aarch64.whl"},{"url":"https://api.github.com/repos/vllm-project/vllm/releases/assets/318091463","id":318091463,"node_id":"RA_kwDOI7xefs4S9bDH","name":"vllm-0.11.1.tar.gz","label":"","uploader":{"login":"khluu","id":51931015,"node_id":"MDQ6VXNlcjUxOTMxMDE1","avatar_url":"https://avatars.githubusercontent.com/u/51931015?v=4","gravatar_id":"","url":"https://api.github.com/users/khluu","html_url":"https://github.com/khluu","followers_url":"https://api.github.com/users/khluu/followers","following_url":"https://api.github.com/users/khluu/following{/other_user}","gists_url":"https://api.github.com/users/khluu/gists{/gist_id}","starred_url":"https://api.github.com/users/khluu/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/khluu/subscriptions","organizations_url":"https://api.github.com/users/khluu/orgs","repos_url":"https://api.github.com/users/khluu/repos","events_url":"https://api.github.com/users/khluu/events{/privacy}","received_events_url":"https://api.github.com/users/khluu/received_events","type":"User","user_view_type":"public","site_admin":false},"content_type":"application/x-gtar","state":"uploaded","size":17286131,"digest":"sha256:672af6f0a56117b9e84b8caacbfe437afba9f366cef8e524dcc5f3f03edd954e","download_count":129,"created_at":"2025-11-18T23:09:17Z","updated_at":"2025-11-18T23:09:18Z","browser_download_url":"https://github.com/vllm-project/vllm/releases/download/v0.11.1/vllm-0.11.1.tar.gz"}],"tarball_url":"https://api.github.com/repos/vllm-project/vllm/tarball/v0.11.1","zipball_url":"https://api.github.com/repos/vllm-project/vllm/zipball/v0.11.1","body":"## Highlights\r\nThis release includes 1456 commits from 449 contributors (184 new contributors)!\r\n\r\nKey changes include:\r\n* **PyTorch 2.9.0 + CUDA 12.9.1**: Updated the default CUDA build to `torch==2.9.0+cu129`, enabling Inductor partitioning and landing multiple fixes in graph-partition rules and compile-cache integration.\r\n* **Batch-invariant `torch.compile`**: Generalized batch-invariant support across attention and MoE backends, with explicit support for DeepGEMM and FlashInfer on Hopper and Blackwell GPUs.\r\n* **Robust async scheduling**: Fixed several correctness and stability issues in async scheduling, especially when combined with chunked prefill, structured outputs, priority scheduling, MTP, and DeepEP / DCP. We expect `--async-scheduling` to be enabled by default in the next release.\r\n* **Stronger scheduler + KV ecosystem**: Improved test coverage in CI and made scheduler behavior more robust with KV connectors, prefix caching, and multi-node deployments.\r\n* **Anthropic API Support**: Added support for the `/v1/messages` endpoint, allowing users to interact with `vllm serve` using Anthropic-compatible clients.\r\n\r\nDetailed release notes will be updated in the next few days.\r\n\r\n## What's Changed\r\n* [Bugfix] Improve GLM4 MoE Reasoning Parser's is_reasoning_end Condition (@frankwang28 #25355)\r\n* [Docs] Add Toronto Meetup (@mgoin #25773)\r\n* [CI] Add E2E Blackwell Quantized MoE Test (@mgoin #25723)\r\n* [V1] address post issues related to #20059 (part 1); cascade attention reenable by default (@fhl2000 #23046)\r\n* [CI] Fix FlashInfer AOT in release docker image (@mgoin #25730)\r\n* [spec decode] Consolidate speculative decode method name for MTP (@zixi-qi #25232)\r\n* Reduce the Cuda Graph memory footprint when running with DBO (@SageMoore #25779)\r\n* Kernel-override Determinism [1/n] (@bwasti #25603)\r\n* [Bugfix] Optimize CpuGpuBuffer initialization (@namanlalitnyu #25447)\r\n* [Spec decode] automatically disable mm for text-only draft models (@jmkuebler #25667)\r\n* [Core] Don't count preempted tokens in prefix cache hit rate (@zhuohan123 #25787)\r\n* Add option to restrict media domains (@russellb #25783)\r\n* Add flashinfer-build.sh and register precompiled cu128 wheel in Dockerfile (@mgoin #25782)\r\n* [Multimodal][Speculative Decoding]Eagle Eagle3 mm support, enablement on qwen2.5vl (@david6666666 #22872)\r\n* [Bugfix] Allow Only SDPA Backend for ViT on B200 for Qwen3-VL (@yewentao256 #25788)\r\n* [CI/Build] Consolidate model loader tests and requirements (@DarkLight1337 #25765)\r\n* [CI/Build] Add timing to Model Executor Test (@22quinn #25799)\r\n* [CI/Build] Reorganize root-level V1 tests (@DarkLight1337 #25767)\r\n* [Misc] Fix codeowners override for v1 sample and attention (@22quinn #25037)\r\n* [Misc] Update openai client example file for multimodal (@ywang96 #25795)\r\n* [Bugfix] Add missing `image_size` for phi4_multimodal (@Renovamen #25796)\r\n* [Bugfix] Merge MM embeddings by index instead of token IDs (@DarkLight1337 #16229)\r\n* Validate API tokens in constant time (@russellb #25781)\r\n* Add filtering for chat template kwargs (@russellb #25794)\r\n* Fix GPTQ model loading in Transformers backend (@hmellor #25770)\r\n* [Bugfix] Fix triton import precommit failure (@tlrmchlsmth #25803)\r\n* [Bugfix][WideEP] Apply TP Attn + EP MoE fix to other models (@tlrmchlsmth #24982)\r\n* [docs] Resolve transcriptions API TODO (@yyzxw #25446)\r\n* [env] default nixl side port conflicts  with kv-event zmq port (@panpan0000 #25056)\r\n* [Core] Refactor self.model() to call a helper for subclassing. (@patrick-toulme #25084)\r\n* [torch.compile]: Add VLLM_DEBUG_DUMP_PATH environment variable (@ZJY0516 #25651)\r\n* [Bug]: Set LD_LIBRARY_PATH to include the 'standard' CUDA location (@smarterclayton #25766)\r\n* [Core] GC Debug callback (@Jialin #24829)\r\n* [Bugfix][NIXL] Fix Async Scheduler timeout issue (@NickLucche #25808)\r\n* [MM] Optimize memory profiling for scattered multimodal embeddings (@ywang96 #25810)\r\n* [Bugfix] Fix Qwen3-VL regression from #24982 (@ywang96 #25814)\r\n* [VLM] Update Qwen3-VL max_num_video_tokens calculation for configurable video profiling (@Isotr0py #25557)\r\n* Fix random dataset mismatched token length with config. (@weireweire #24937)\r\n* Update GLM-4.5 Doc transformers version (@zRzRzRzRzRzRzR #25830)\r\n* [Bugfix] fix Qwen3VLMoe load when pp > 1 (@JJJYmmm #25838)\r\n* Remove redundant cudagraph dispatcher warning (@mgoin #25841)\r\n* [Misc] fix tests failure by using current_platform (@kingsmad #25825)\r\n* [P/D] NIXL Updates (@robertgshaw2-redhat #25844)\r\n* Add Phi4FlashForCausalLM to _PREVIOUSLY_SUPPORTED_MODELS (@tdoublep #25832)\r\n* [XPU]Fix xpu spec decoding UTs, avoid using cuda graph (@jikunshang #25847)\r\n* [Bugfix] Fallback ViT attn backend to SDPA for blackwell (@ywang96 #25851)\r\n* [V0 Deprecation][Models] Remove all V0 condition for mm embeddings merge (@Isotr0py #25331)\r\n* [Misc] Remove more `get_input_embeddings_v0` (@DarkLight1337 #25857)\r\n* update to latest deepgemm for dsv3.2 (@youkaichao #25871)\r\n* [Bugfix] Fix requirements paths in install instructions (@yingjun-mou #25827)\r\n* [Model][Bugfix] Fix issues in MiDashengLM implementation for quantized models (@zhoukezi #25854)\r\n* [torch.compile] serialize cudagraph_mode as its enum name instead of value (@ZJY0516 #25868)\r\n* [Cuda2CPU][P/D] Add cuda2cpu support in NixlConnector (@chenxi-yang #24690)\r\n* [Bugfix][Speculative Decoding] Fix Eagle3 quantization config issue (@rahul-tuli #25883)\r\n* [CI/Build] Include Transformers backend test in nightly transformers test (@Isotr0py #25885)\r\n* [Model] Remove MotifForCausalLM (@jeejeelee #25866)\r\n* [Bugfix] Use correct key \"ignore\" for config.json non-quantized layers (@leejnau #25706)\r\n* [BugFix][torch.compile] KV scale calculation issues with FP8 quantization (#21640) (@adabeyta #25513)\r\n* [Doc] Add documentation for vLLM continuous benchmarking and profiling (@namanlalitnyu #25819)\r\n* [Bugfix][ROCm] Fixing trying to import non-existent symbols from libnccl.so (@gshtras #25605)\r\n* [Kernel] Chunk-aligned mamba2 (@tdoublep #24683)\r\n* [Doc] Polish example for torchrun dp (@zhuohan123 #25899)\r\n* [NIXL] Increase default KV block eviction timeout on P (@NickLucche #25897)\r\n* [V0 Deprecation] Remove `vllm.worker` and update according imports (@aarnphm #25901)\r\n* Test Prompt Embeds/LoRA compatibility and Enable LoRA Support for OPT Models  (@qthequartermasterman #25717)\r\n* [Bug] Fix Weight Loading for Block FP8 Cutlass SM90 (@yewentao256 #25909)\r\n* [Benchmark] Support benchmark throughput for external launcher DP (@zhuohan123 #25913)\r\n* Move`VllmConfig` from `config/__init__.py` to `config/vllm.py` (@hmellor #25271)\r\n* [BugFix] Fix DP/EP hang  (@LucasWilkinson #25906)\r\n* [BugFix] Pass config_format via try_get_generation_config (@acisseJZhong #25912)\r\n* [Model][Bugfix] Fix MiDashengLM audio encoder mask by removing incorrect `logical_not` (@zhoukezi #25925)\r\n* [Bugfix]: Clean up chunked prefill logging when using whisper (@simondanielsson #25075)\r\n* [New Model] DeepSeek-V3.2 (Rebased to Main) (@zyongye #25896)\r\n* [Doc] Add Cambricon MLU support (@a120092009 #25942)\r\n* Updated TRL integration docs (@sergiopaniego #25684)\r\n* [Bugfix][Model]fix ernie45 moe gate&bias dtype to float32 (@CSWYF3634076 #25936)\r\n* [Model] Move `vision_feature_select_strategy` into `resolve_visual_encoder_outputs` (@DarkLight1337 #25938)\r\n* [perf] Use CPU tensor to reduce GPU->CPU sync (@lhtin #25884)\r\n* [NIXL] Add support for MLA caches with different latent dim (@NickLucche #25902)\r\n* [CI] Move applicable tests to CPU (@rzabarazesh #24080)\r\n* [Fix] Improve CPU backend compatibility for RISC-V (@ihb2032 #25816)\r\n* [Kernel][Moe Configs] Add more tuned triton configs for ExpertsInt8 and FP8 (@Josephasafg #25858)\r\n* Add Hugging Face Inference Endpoints guide to Deployment docs (@sergiopaniego #25886)\r\n* [Bugfix][Model] Fix inference for Hunyuan dense models (@Anionex #25354)\r\n* [Bugfix] Fix accuracy issue of TRTLLM FP8 MOE and improve logging (@pavanimajety #25895)\r\n* [Bugfix] Token type and position embeddings fail to be applied to `inputs_embeds` (@DarkLight1337 #25922)\r\n* [bugfix][deepseek] fix flashmla kernel selection (@youkaichao #25956)\r\n* [Bug] Fix AttributeError: 'QKVParallelLinear' object has no attribute 'orig_dtype' (@yewentao256 #25958)\r\n* [Doc] Improve MM Pooling model documentation (@DarkLight1337 #25966)\r\n* [Docs] Add moe kernel features doc  (@bnellnm #25297)\r\n* OffloadingConnector: Fix GPU block tracking bug (@orozery #25856)\r\n* [Llama4] [multimodal] Fix misplaced dtype cast of `cos_sin_cache` in `Llama4VisionRotaryEmbedding` (@cjackal #25889)\r\n* [Bench] Add DeepSeekV32 to MoE benchmark (@jeejeelee #25962)\r\n* [V1] [P/D] Add Support for KV Load Failure Recovery (@sdavidbd #19330)\r\n* Add explicit pooling classes for the Transformers backend (@hmellor #25322)\r\n* [Docs] Remove API Reference from search index (@hmellor #25949)\r\n* [gpt-oss] use vLLM instead of openai types for streaming (@qandrew #25186)\r\n* [Misc] Make EP kernels install script support uv (@LucasWilkinson #25785)\r\n* [Model] MTP fallback to eager for DeepSeek v32 (@luccafong #25982)\r\n* Update launch_bounds_utils.h for correct compile on Multiple Cuda Arch - PTXAS out of range Warning (@DrStone1971 #25843)\r\n* [Log] Optimize Log for FP8MOE (@yewentao256 #25709)\r\n* Fix INT8 quantization error on Blackwell GPUs (SM100+) (@certainly-param #25935)\r\n* [MM] Add text-only mode for Qwen3-VL (@ywang96 #26000)\r\n* [Bugfix] Fix `__syncwarp` on ROCM (@zhewenl #25996)\r\n* [BugFix] Fix default kv-cache-dtype default for DeepseekV3.2 (@LucasWilkinson #25988)\r\n* Update to Transformers `v4.56.2` (@hmellor #24638)\r\n* [Misc]allow disable pynccl (@luccafong #25421)\r\n* [Doc] updating torch.compile doc link #25989)\r\n* [BugFix][MM] Fix Nonetype error when video is cache in qwen2.5-omni-thinker (@wwl2755 #26004)\r\n* [Misc] Factor out common `_apply_feature_select_strategy` (@DarkLight1337 #26003)\r\n* [CI] Only capture a single CUDA graph size in CI by default (@hmellor #25951)\r\n* [MISC] Fix misleading batch_size_capture_list when cuda_graph_sizes < 4 (@billishyahao #25829)\r\n* [Benchmark] Finish documented v0.11.0 deprecation of --endpoint-type (@natoscott #26007)\r\n* [Bugfix] Apply same sampling parameters for both `n=1` and `n>1` (@kmaehashi #26005)\r\n* [NVIDIA] Blackwell Family (@johnnynunez #24673)\r\n* Fix test_mamba_ssm_ssd.py due to missing _query_start_loc_to_chunk_indices_offsets (@hl475 #25995)\r\n* [CI] Tweaks to GPT-OSS Eval (Blackwell) for stability (@mgoin #26030)\r\n* [BugFix][DP/EP] Fix CUTLASS MLA hang under load (@LucasWilkinson #26026)\r\n* [ROCm][Build] Add support for AMD Ryzen AI MAX / AI 300 Series (@hyoon1 #25908)\r\n* [Bug] Fix Negative Cuda Memory Usage (@yewentao256 #25683)\r\n* [BugFix] ChunkedLocalAttention is currently not CG compatible (@LucasWilkinson #26034)\r\n* Support RL online quantization with torchao (@jerryzh168 #23014)\r\n* [ROCm][Bugfix] Add missing parameter to ROCm backend (@gshtras #26029)\r\n* [Misc] Make handling of SamplingParams clearer in n>1 case (@njhill #26032)\r\n* Run:ai model streamer add GCS package support (@pwschuurman #24909)\r\n* Update base image to 22.04 (jammy) (@huydhn #26065)\r\n* Change size of single CUDA graph for CI to 4 (@tdoublep #26089)\r\n* [FA/Chore] Bump vllm-flash-attention (@LucasWilkinson #25537)\r\n* [Model] Use `merge_by_field_config` for MM models (A-C) (@DarkLight1337 #26073)\r\n* [Model] Use `merge_by_field_config` for MM models (D-F) (@DarkLight1337 #26076)\r\n* [Platform][CI] Added OOT platform interface e2e test that running on Ascend NPU (@leo-pony #25470)\r\n* [Qwen][ROCm] Flash Attention Rotary Embeddings (@vllmellm #24642)\r\n* [CI] Add Blackwell DeepSeek FP8 FlashInfer MoE tests (@mgoin #26040)\r\n* [CI/Build] Replace `vllm.entrypoints.openai.api_server` entrypoint with `vllm serve` command (@DarkLight1337 #25967)\r\n* [BugFix] Fix FI accuracy issue when used for MLA prefill (@LucasWilkinson #26063)\r\n* [Small] Prevent bypassing media domain restriction via HTTP redirects (@huachenheli #26035)\r\n* [Deepseek v3.2] Support indexer prefill chunking (@heheda12345 #25999)\r\n* EAGLE 3: Fix preamble so that measured speedup over Eagle 1 becomes 32% instead of 5% on MTBench (@ekagra-ranjan #25916)\r\n* [Mamba][KVCacheManager] Simplify kv cache manage logic for mamba + MTP (@heheda12345 #25119)\r\n* [Perf] Fix and reapply move apply w8a8 block fp8 linear to class (@ElizaWszola #25696)\r\n* Fix MTP with deepep_low_latency (@MatthewBonanni #25904)\r\n* [Bugfix] Disable cascade attention with FlashInfer (@mgoin #26130)\r\n* [Log] Optimize DeepGEMM Missing Log (@yewentao256 #26106)\r\n* [Bug][Benchmark] Fix duplicate req in oversampling (@ekagra-ranjan #26140)\r\n* [Attention] Move Backend enum into registry (@MatthewBonanni #25893)\r\n* [CI/Build] Conditionally register cutlass_fp4_group_mm to fix building on Hopper (@mgoin #26138)\r\n* [DeepSeek] Improve performance of DS MLA cache kernel (@MatthewBonanni #26132)\r\n* [Bug]: Limit num_reqs in dummy_run when max_num_seqs is small (@benchislett #26144)\r\n* [gpt-oss] disable tool server initialization if no tool in request (@qandrew #25790)\r\n* [Build/CI] Revert back to Ubuntu 20.04, install python 3.12 with uv (@tlrmchlsmth #26103)\r\n* [ROCm] [VL] [Bugfix] Fix vit flash attn dispatcher logic for ROCm (@tjtanaa #26104)\r\n* [Bugfix] Fix import `gemm_afp4wfp4` failure on AMD (@zhewenl #26068)\r\n* [Model] Use `merge_by_field_config` for MM models (G) (@DarkLight1337 #26117)\r\n* `FusedMoE` support for the Transformers backend (@hmellor #22650)\r\n* [BUG] Reorder model config creation (@ahao-anyscale #26124)\r\n* [Misc] Remove typing.List (@varun-sundar-rabindranath #26150)\r\n* [Input] Remove unused `prompt` field (@DarkLight1337 #26097)\r\n* [Perf] Optimize `reshape_and_cache` CUDA Kernel (@ZJY0516 #25955)\r\n* add(v1): RequestStatesStats to RequestOutput (@huijjj #24947)\r\n* [Model] Use `merge_by_field_config` for MM models (InternVL family) (@DarkLight1337 #26153)\r\n* [test utils] correct wrong typing (@yannicks1 #26159)\r\n* [CI] Fix distributed hybrid tests in CI (@tdoublep #26155)\r\n* [NIXL][Misc] Expose metrics from NIXL for logging to CLI (@NickLucche #25388)\r\n* [openai] Fix missing tool usage check (system message) (@levunet #24768)\r\n* [Multi Modal] Configurable MM Profiling (@wwl2755 #25631)\r\n* [Doc] Fixed shape description for fused_batched_moe.py (@Egor-Krivov #25668)\r\n* Quick fix for IMA with the Prefix Prefill kernel during graph capture (@SageMoore #25983)\r\n* [Renderer] Move Processor out of AsyncLLM  (@KKSK-DON #24138)\r\n* Re-enable prefill of max model length (@yannicks1 #24446)\r\n* [backends][short_conv] CUDA graph piecewise edits (@paulpak58 #24215)\r\n* [Model] Supplement to PR 24862: Pass param prefix to LLMHead (@whx-sjtu #25805)\r\n* [CI/Build] do not enforce precompilation on tpu ci tests (@sixiang-google #25992)\r\n* [Model] Fixed stream generator for gpt-oss + spec-decoding (@astralord #26027)\r\n* [Renderer] Move Processor out of LLMEngine (@DarkLight1337 #26165)\r\n* Fix undefined symbol: cutlass_moe_mm_sm100 (@jasl #26098)\r\n* [BugFix][QWEN-VL]fix wrong apply_rotary_emb_torch selection introduced by #24642 (@xuechendi #26123)\r\n* Stop mergify from keeping stale PRs alive (@hmellor #26169)\r\n* Avoid division by zero in cache DS MLA kernel (@MatthewBonanni #26174)\r\n* Fix V1 engine serialization error with Ray distributed executor (@nrghosh #26148)\r\n* [Quantization/NVFP4] Speed up TRTLLM NVFP4 MOE weight loading and fix K/V scale loading for MLA Attn (@pavanimajety #25968)\r\n* [Perf] Remove hardcoded num_warps=1 (@chelsea0x3b #26183)\r\n* [Refactor] Optimize FP8 MOE Backend Choice and Log (@yewentao256 #26044)\r\n* [responsesAPI] add better error messaging for long prompts (@qandrew #25724)\r\n* [Bugfix] Relax tokenizer regex for mixtral to include 'tokenizer.model' (@BowenBao #25964)\r\n* [CI] Push multiarch manifests as nightly builds (@csahithi #25764)\r\n* [Misc] Add penalties sampling parameters to serve tool (@southfreebird #25974)\r\n* [BugFix] Fix de-functionalization pass for rotary_embedding (@angelayi #23953)\r\n* [CI] Fix Pre-commit Mypy Error (@yewentao256 #26181)\r\n* [GPTOSS][DP/EP][Marlin] Enable GPTOSS DP/EP using Marlin kernels (@varun-sundar-rabindranath #25488)\r\n* Fix issue of using only the part of video frame [Nemotron Nano] (@BloodAxe #26186)\r\n* [Bugfix] Fix qwen3 vl dummy data generation with overrides (@ywang96 #26193)\r\n* [BugFix] Use async Mistral Tokenizer in Chat Completions (@bbrowning #26134)\r\n* Add batch invariant kernel override for FlashInfer backend [2/n] (@bwasti #25769)\r\n* [cpu][perf] Accelerate unquantized-linear for AArch64 through oneDNN/ACL and weight prepack (@fadara01 #25948)\r\n* [V1] [Hybrid] Mamba2 Automatic Prefix Caching (@s3woz #25752)\r\n* Support expert parallel in Transformers backend (@hmellor #26162)\r\n* [Model] Support nested structures for TensorSchema (@DarkLight1337 #26212)\r\n* [Misc] Require `merge_by_field_config` argument (@DarkLight1337 #26214)\r\n* [Misc] Remove unused `executor.apply_model` (@DarkLight1337 #26215)\r\n* [CI Failure] fix_test_auto_prefix_cache_support (@hl475 #26053)\r\n* Revert \"Add batch invariant kernel override for FlashInfer backend [2/n]\" (@DarkLight1337 #26220)\r\n* Add Olmo 3 reasoning parser (@soldni #26054)\r\n* [Core] Enable decode of context length equal to max model length (@yannicks1 #26168)\r\n* [Bugfix] Fix `_reqs_to_process` leak on abort (@NickLucche #26012)\r\n* [Model] CLIP Embedding Support (@DarkLight1337 #26010)\r\n* Fix tensor device and dtype placement in Qwen2VL model (@yuafng #26219)\r\n* [V1] [Hybrid] Remove code to override default CUDA graph configuration (@tdoublep #26226)\r\n* [CPU] Refine batch reorder of CPU attention backend (@bigPYJ1151 #26096)\r\n* [Frontend] Cache chat template kwargs resolution (@Isotr0py #26227)\r\n* [Renderer] Clean up renderer code (@DarkLight1337 #26216)\r\n* [Model] Use `merge_by_field_config` for MM models (H-L) (@DarkLight1337 #26230)\r\n* [Easy] Add str repr for IterationStats (@22quinn #26232)\r\n* [Bugfix] Allow `--skip-tokenizer-init` with `echo and return_token_ids` (@DarkLight1337 #26238)\r\n* Add documentation for granite 4 tool calling (@maxdebayser #26175)\r\n* [Perf][Easy] Early stop in request_block_hasher (@Jialin #26112)\r\n* [Bugfix]: Assertion error when using FlashInfer backend (@simondanielsson #25933)\r\n* [Bugfix] Always apply MM processor even when no MM items are passed (@DarkLight1337 #26240)\r\n* [Bugfix][Hardware][RISC-V] Limit supported dtypes to float32 to avoid scheduler segfault (@ihb2032 #26228)\r\n* [Refactor][Kernel] support loading kernel from other place (@ILikeIneine #25823)\r\n* Convert formatting to use `ruff` instead of `yapf` + `isort` (@hmellor #26247)\r\n* Remove all references to `yapf` as it's no longer used (@hmellor #26251)\r\n* Remove all cases of `fmt: on/off` (@hmellor #26253)\r\n* fix(tests): Resolve late binding of loop variable in assert message lambda (@ihb2032 #26249)\r\n* Fix per file ruff ignores related to typing (@hmellor #26254)\r\n* Update `ruff` pre-commit hooks version (@hmellor #26255)\r\n* [CI] fix mamba kernel test (@ZJY0516 #26250)\r\n* [NVIDIA] flashinfer TRTLLM attention prefill token limit (@jasonlizhengjian #25998)\r\n* Fix per file ruff ignores related to simplification (@hmellor #26259)\r\n* [CI] Add Blackwell LM Eval Small Models test to nightly (@mgoin #26052)\r\n* [DOC] Update production-stack.md (@elieserr #26177)\r\n* [CI] Add comment about the single cudagraph capture size that is used (@tdoublep #26252)\r\n* [V1] [Hybrid] Some additional clean-up in Mamba2 prefix caching (@tdoublep #26222)\r\n* [Doc] Edited minor typo (@orangeng #26266)\r\n* [MISC] Add heheda12345 to CODEOWNERS of vllm/config/cache.py (@heheda12345 #26270)\r\n* [CI][gpt-oss] Enable python tool tests in CI (@wuhang2014 #24315)\r\n* Fix per file ruff ignores related to line length (@hmellor #26262)\r\n* Bump actions/stale from 10.0.0 to 10.1.0 (@dependabot[bot] #26272)\r\n* [Benchmarking] Add disable_shuffle option for dataset loading (@ymoslem #26258)\r\n* [Misc] Clean up unnecessary E501 ignore (@ywang96 #26274)\r\n* [Docs] Edit HF Inference Endpoints documentation (@ariG23498 #26275)\r\n* [Doc] add KAITO to integrations (@abhisheksheth28 #25521)\r\n* [Frontend] Consolidate tokenizer init code (@DarkLight1337 #26276)\r\n* [Model] Use `merge_by_field_config` for MM models (Llava family) (@DarkLight1337 #26280)\r\n* Support expert parallel load balancing in Transformers backend (@hmellor #26287)\r\n* [Bugfix] Fix mrope in Transformers Backend (@zucchini-nlp #26087)\r\n* Fix `DotsOCR` tensor type (@what-in-the-nim #26281)\r\n* [Model] EVS support for nano_nemotron_vl (@tomeras91 #26269)\r\n* [Attention] Remove unused reorder_batch method (@MatthewBonanni #24463)\r\n* [Tests] conftest: Extending VllmRunner and HfRunner to accept token_ids as input (@yannicks1 #26295)\r\n* [CI Bugfix] Make sure TRTLLM attention is available in test_blackwell_moe (@mgoin #26188)\r\n* Support llama3 eagle3 head with llama4 verifier (@rahul-tuli #25961)\r\n* [Misc] auto_tune: kill specific vllm process (@karan #26304)\r\n* [Bugfix][Spec Decode] Fix wrong valid_mask for padded speculation when chunked prefill occurs (@seven-mile #26231)\r\n* Add bias handling to CPUFusedMOE kernel (@cfRod #26289)\r\n* [Bugfix] Fix gemma3 with transformers backend (@zucchini-nlp #23178)\r\n* [Benchmark] Enable MM Embedding benchmarks (@DarkLight1337 #26310)\r\n* [Docs] Fix broken table in moe_kernel_features doc (@varun-sundar-rabindranath #26314)\r\n* [BugFix] Pad input buffers in _dummy_run (@varun-sundar-rabindranath #26209)\r\n* [Bugfix] Allow skipping MoE in NVFP4 (fix for MTP) (@benchislett #25987)\r\n* [ROCm] Split AITER unified attention into its own backend (@gshtras #25507)\r\n* [Perf] Add decode full-graph support to FlashInfer-MLA backend (@benchislett #26313)\r\n* [Misc] Define EP kernel arch list in Dockerfile (@simon-mo #25635)\r\n* [Docs][DBO] Add initial doc that describes the DBO implementation (@SageMoore #26024)\r\n* [Core] Simplify the Dp padding/should ubatch coordination logic (@SageMoore #25768)\r\n* [UX] Support nested dicts in hf_overrides (@mgoin #25727)\r\n* [BUG] Fix file parsing for load_format runai_streamer_sharded (@ahao-anyscale #26324)\r\n* [Model] Define merge_by_field_config MM interface (U-Z) (@ayushsatyam146 #26261)\r\n* [Deprecation] Deprecate `LLM.set_tokenizer` (@DarkLight1337 #26333)\r\n* [responsesAPI][bugfix] serialize harmony messages (@qandrew #26185)\r\n* [Model] Define merge_by_field_config MM interface (R-T) (@ayushsatyam146 #26260)\r\n* [BugFix] Update KV block hash type from BlockHash to ExternalBlockHash in kv_events_subscriber - #26264 (@atalhens #26265)\r\n* [V0 Deprecation] Remove `VLLM_USE_V1` from docs and scripts (@DarkLight1337 #26336)\r\n* Optimize KV cache distribution for asymmetric pipeline parallelism (@gholmes829 #25164)\r\n* Add topk logits torch op for DS3.2. (@dcampora #25945)\r\n* Add TRL example notebook to RLHF docs (@sergiopaniego #26346)\r\n* [Docs] add docs for cuda graph v1 (@fhl2000 #24374)\r\n* [Model] Use `merge_by_field_config` for MM models (Ovis family) (@Isotr0py #26308)\r\n* [Feature][OCP MX] Support mxfp6 and mixed mxfp6-mxfp4 (@fxmarty-amd #21166)\r\n* [Model] Add support for ModernBertForTokenClassification (@antrec #26340)\r\n* [Misc] Move `LRUCache` into its own file (@DarkLight1337 #26342)\r\n* [V0 Deprecation] Remove `VLLM_USE_V1` from tests (@DarkLight1337 #26341)\r\n* [Model] Lfm2Moe (@paulpak58 #26344)\r\n* [ci] Rename `test_mxfp4_moe.py` to `test_ocp_mx_moe.py` (@fxmarty-amd #26364)\r\n* [CI] Add Qwen3 MoE NVFP4 to Blackwell lm-eval (@mgoin #26316)\r\n* [deepseek] add EP8 FusedMOE config for H200 and B200 (@heheda12345 #26331)\r\n* [Bug] Fix Shape Validation for Fallback while Enabling E8M0 for DeepGEMM (@yewentao256 #26322)\r\n* [Bugfix] Add missing sink tensor into flash attn cascade attn implementation (@plliao #26325)\r\n* [Frontend] CompilationConfig overhaul (#20283): deprecate use_inductor in favor of backend, simplify custom_ops (@morrison-turnansky #26113)\r\n* [V1] Logit processors for rejection sampler (@southfreebird #19482)\r\n* [Spec Decode] Enable efficient speculative decoding with FlashInfer-MLA (@benchislett #25984)\r\n* [TPU] update TPU benchmark threshold (@jcyang43 #25713)\r\n* Add more libraries to rlhf.md (@mgoin #26374)\r\n* [Bugfix] Fix MTP+FlashInfer crash when trtllm kernels are available but disabled (@benchislett #26361)\r\n* Revert #24446 and #26168 (@tdoublep #26332)\r\n* [Misc] Clean up cruft from previous FlashMLA sparse implementation (@LucasWilkinson #26125)\r\n* [torchao] safetensors integration (@liangel-02 #25969)\r\n* Add SwigluOAI implementation for CPUFusedMOE (@isharif168 #26347)\r\n* [Core] Simplify setting new_token_ids in CachedRequestData (@njhill #26388)\r\n* fix(v1/kv_cache): resolve async KV transfer bug in cascade attention (@ayushsatyam146 #23485)\r\n* Add gather_indexer_k_quant_cache kernel (@Barry-Delaney #25931)\r\n* [Bugfix] Incorrect MM data format in `vllm bench throughput` (@DarkLight1337 #26395)\r\n* fix[DP][v1]: Prevent hangs from mismatched worker configurations (@ayushsatyam146 #26218)\r\n* [TPU] Rename tpu_commons to tpu_inference (@utkarshsharma1 #26279)\r\n* [Feature] Enable E8M0 by Default on Hopper for DeepGEMM, 5% E2E throughput improvement (@yewentao256 #26197)\r\n* [Misc] add usedforsecurity=False in md5 hash call (@dtrifiro #26357)\r\n* [Model] Allow passing custom number of max tiles to Nano 2 VL (@BloodAxe #26403)\r\n* [Docs] Have mergify leave a comment with the docs preview link (@hmellor #26412)\r\n* [CI] Pooling models mteb test disable enforce_eager  (@noooop #26408)\r\n* [Benchmarks] Add support for Qwen 3 VL MoE tuning (@lgeiger #26419)\r\n* Tidy `vllm/config/__init__.py` to only add classes and functions (@hmellor #26405)\r\n* [NIXL][non-cuda] Add install script for nixl with non-cuda ucx (@xuechendi #25959)\r\n* [Refactor] Refactor FP8 & INT8 Quant Folder inside `w8a8` (@yewentao256 #25293)\r\n* [CI Failure] Fix pre-commit issue for install_nixl_from_source_ubuntu.py (@mgoin #26424)\r\n* [Bugfix] Fix `vllm bench ...` on CPU-only head nodes (@Aydin-ab #25283)\r\n* [Bug] Fix DeepGEMM Attention Test (@yewentao256 #26423)\r\n* [Benchmarks] Fix imports in FP8 tuning script (@lgeiger #26407)\r\n* [Bug] Fix Test in Batch Invariant (@yewentao256 #26128)\r\n* Remove Python 3.9 support ahead of PyTorch 2.9 in v0.11.1 (@hmellor #26416)\r\n* [Feature] Change cache.py with pydantic validation (@vrdn-23 #26390)\r\n* [Attention] Implement universal BACKEND_MAP (@MatthewBonanni #25900)\r\n* [Bugfix][Flashinfer] fix VLLM_USE_TRTLLM_ATTENTION issue for models with diff hyperparameters (@elvischenv #25924)\r\n* [BugFix] Fix failing test quantization/test_compressed_tensors.py::test_compressed_tensors_fp8_block_enabled (@morrison-turnansky #26436)\r\n* [Kernel] Centralize platform kernel import in `current_platform.import_kernels` (@NickLucche #26286)\r\n* [Models] Improve iteration over layers (@lgeiger #26425)\r\n* [Bugfix] Respect min_tokens in scheduler stop check (@elaineyz #26317)\r\n* [Kernels] Modular kernel refactor (@bnellnm #24812)\r\n* [Attention] Register FLASHMLA_SPARSE (@MatthewBonanni #26441)\r\n* Separate MLAAttention class from Attention (@therealnaveenkamal #25103)\r\n* [Misc] Redact ray runtime env before logging (@ruisearch42 #26302)\r\n* [Bugfix] Set the minimum python version for gpt-oss (@jeejeelee #26392)\r\n* [Minor] Change warning->warning_once in preprocess (@zhuohan123 #26455)\r\n* [Bugfix] Catch and log invalid token ids in detokenizer #2 (@njhill #26445)\r\n* [Bugfix] Incorrect another MM data format in vllm bench throughput (@huydhn #26462)\r\n* [Hardware][AMD] Enable FlexAttention backend on ROCm (@mawong-amd #26439)\r\n* [MM][Doc] Add documentation for configurable mm profiling (@wwl2755 #26200)\r\n* [Core][KVConnector] Propagate all tokens on resumed preemptions (@QierLi #24926)\r\n* [Hybrid]: Decouple Kernel Block Size from KV Page Size (@zhiyuan1i #24486)\r\n* [CI/Build] Fix model nightly tests (@DarkLight1337 #26466)\r\n* [Core] Relax the LoRA  max rank (@jeejeelee #26461)\r\n* Update Dockerfile and install runai-model-streamer[gcs] package (@pwschuurman #26464)\r\n* Bump Flashinfer to v0.4.0 (@elvischenv #26326)\r\n* [Model] Gemma3: Fix GGUF loading and quantization (@lucianommartins #26189)\r\n* Enable `RMSNorm` substitution for Transformers backend (@hmellor #26353)\r\n* Add: Support for multiple hidden layers in Eagle3 (@rahul-tuli #26164)\r\n* [torchao] Add support for ModuleFqnToConfig using regex (@jerryzh168 #26001)\r\n* [Misc] Misc code simplifications (@njhill #26450)\r\n* [doc] add Volcengine as a compute sponsor (@youkaichao #26477)\r\n* [Feature] Use pydantic validation in lora.py and load.py configs (@simondanielsson #26413)\r\n* [Misc] Upgrade more code to Python 3.10 (@DarkLight1337 #26463)\r\n* [Bugfix] Fix SHM cache initialization (@DarkLight1337 #26427)\r\n* [Models][Qwen3VL] Optimise `_validate_and_reshape_mm_tensor` (@lgeiger #26426)\r\n* [Bugfix] Move current_platform import to avoid python import cache. (@iwzbi #16601)\r\n* [V0 deprecation] Remove `QKVCrossParallelLinear` implementation  (@Isotr0py #26475)\r\n* [Feature] Use pydantic validation in parallel.py config (@simondanielsson #26417)\r\n* Revert  #26113 \"[Frontend] CompilationConfig overhaul (#20283): deprecate use_inductor in favor of backend, simplify custom_ops\" (@ZJY0516 #26472)\r\n* Upgrade Pydantic to v2.12.0 and remove hack for Python 3.13 (@hmellor #26481)\r\n* [Models][Qwen] Replace `pad` with `cat` for better performance (@lgeiger #26486)\r\n* [Attention][DCP] Support DCP with query length > 1 (MTP) with FA3 (@minosfuture #25049)\r\n* [Model] Apply shared experts overlap optimization to all models with shared experts (@bnellnm #26145)\r\n* [BUGFIX] Add cu_tokens_across_sp to DPMetadata (@SageMoore #26457)\r\n* [Bugfix] Enable padded FP4 quantization (@roikoren755 #25947)\r\n* [Bugfix] Disable moe inplace for torch >= 2.9 (@bnellnm #26497)\r\n* [Flashinfer][gpt-oss] Support FP8-qkv Flashinfer TRTLLM Sinks Attention (@elvischenv #25674)\r\n* [Core] Remove unused `prev_sampled_token_ids_invalid_indices` input batch field (@njhill #26514)\r\n* [UX] Add FlashInfer as default CUDA dependency (@mgoin #26443)\r\n* [Bugfix] Fix CUDA graph selection bug in FlashInfer at high concurrency (@benchislett #26499)\r\n* [Bug] Fix modular_kernel: ZeroDivisionError: integer division or modulo by zero (@yewentao256 #26528)\r\n* [CI] Fix Pre-commit Issue Cannot determine type of \"rank\" and \"world_size\" (@yewentao256 #26448)\r\n* Refactor MistralTokenizer (@juliendenize #26358)\r\n* [DP][ray] Support different VLLM_RAY_DP_PACK_STRATEGY (@ruisearch42 #23849)\r\n* [Core] Small simplification in `GPUModelRunner._update_states()` (@njhill #26508)\r\n* [Chore]: One pythonic tool parser test uses the wrong parser (@bbrowning #26515)\r\n* [Spec-Decode] Support piecewise cudagraphs for Eagle head (@LucasWilkinson #25109)\r\n* fix test_simple_inductor_graph_partition (@BoyuanFeng #26522)\r\n* [deepseek] kernel block size for UniformTypeKVCacheSpecs (@heheda12345 #26559)\r\n* [Metrics] Log multi-modal cache stats and fix reset (@DarkLight1337 #26285)\r\n* [GPT-OSS] Add support for arrays  at tool message content (@luis5tb #25593)\r\n* Remove LoRA bias support (@ashwin-phadke #25807)\r\n* [CI] fix ruff format (@chaunceyjiang #26579)\r\n* [bugfix][DCP] fix block_size of hash in DCP prefix caching (@heheda12345 #26296)\r\n* [NIXL] Ignore abort on already-finished request (@markmc #25067)\r\n* [Bugfix] Convert untraceable GroupShape to list for AMD impl (@Lucaskabela #26535)\r\n* [BugFix] Fix noop elimination edge case (@andylolu2 #26394)\r\n* [CI] fix test_run_batch.py::test_completions - AssertionError (@chaunceyjiang #26578)\r\n* [BugFix][torch.compile] Fix fused_scaled_matmul_reduce_scatter signature for PyTorch 2.8 (@jasonlizhengjian #26038)\r\n* Added test_top_k_per_row to test-pipeline.yaml. (@dcampora #26569)\r\n* [Bugfix] Make DP padding optional in coordinate_batch_across_dp (@SageMoore #26375)\r\n* Silu v2 (@elvircrn #25074)\r\n* [Metrics] Add test for multi-modal cache stats logging (@markmc #26588)\r\n* [torch.compile] Make inductor partition rules respect splitting_ops #25691 (@baonudesifeizhai #25845)\r\n* [Bugfix] fixed top_logprobs: -1 does not appear to work as intended (@chaunceyjiang #26470)\r\n* [Model][Qwen3VL] Compute `cu_seqlens` on CPU to remove  (@lgeiger #26496)\r\n* [Model] Add FlexOlmo model implementation (@2015aroras #24923)\r\n* [Transform] [Quantization] Add QuTLASS support to vLLM (@LopezCastroRoberto #24440)\r\n* Add Qwen3-Omni moe thinker (@wangxiongts #25550)\r\n* Update `pre-commit` hook versions (@hmellor #26591)\r\n* Update CUDA architecture list in build pipeline for 12.9.1 wheels (@wseaton #26592)\r\n* Fix some typing issues found by `mypy==1.18.2` (@hmellor #26596)\r\n* [BUG] Qwen3-next MTP. Fix attn metadata build bug (@vadiklyutiy #26564)\r\n* [BugFix] Fix async scheduling + request preemption (@njhill #26385)\r\n* Cache the environment variable check for batch invariance (@bwasti #26510)\r\n* AOT Compilation for torch.compile (Bundled) (@zhxchen17 #24274)\r\n* [BugFix] Make penalties and bad_words work with async scheduling (@njhill #26467)\r\n* [Frontend] Improve the performance of `is_reasoning_end` (@chaunceyjiang #25735)\r\n* [CI/Build] Fix ppc64le CPU build and tests (@npanpaliya #22443)\r\n* [XPU] Upgrade NIXL to remove CUDA dependency (@zhenwei-intel #26570)\r\n* [MM] Move Qwen3Omni MRoPE impl to model file (@ywang96 #26608)\r\n* [Bugfix][Multi Modal] Fix incorrect Molmo image processing (@sangho-vision #26563)\r\n* [Refactor]: Use M-RoPE interface directly while defining model class instead of maintaining model specific M-RoPE implementation in mrope.py (@divyanshsinghvi #24172)\r\n* fix(nix): Allow local oneDNN path to fix vLLM CPU build failure (@ihb2032 #26401)\r\n* Add EAGLE-3 Speculative Decoding Support for Qwen3 MoE (@rahul-tuli #26485)\r\n* [CPU] fix the issue when the node is '-' cause json decode error. (@muzian666 #26562)\r\n* [Refactor]Reduce duplicate code in serving_chat (@chaunceyjiang #26627)\r\n* [compile] Add patched_fused_scaled_matmul_reduce_scatter (@angelayi #26604)\r\n* [Bugfix][Qwen3VL] fix deepstack in qwen3vl (@JJJYmmm #26626)\r\n* [Bugfix] Fix qwen-moe packed_modules_mapping (@jeejeelee #26634)\r\n* [Benchmark] Support Infinity API (@DarkLight1337 #26641)\r\n* CP: make correct_attn_out robust to 4‑D views and fix Triton arg binding (@hl475 #26509)\r\n* [compile] Fix inductor partition config (@angelayi #26645)\r\n* [EPLB] Support ernie4.5-moe (@HsChen-sys #22100)\r\n* Add @noooop to codeowner for pooling models (@noooop #26652)\r\n* [PERF] [Qwen3-next] Speed up gated RMSNorm (@vadiklyutiy #26207)\r\n* [MISC] Rename the torch profiler filename as instance_id+rank_id for merging the Profiler results of each Rank (@noooop #25867)\r\n* [Bugfix][CI/Build] Fix failing Mteb CI (@Isotr0py #26638)\r\n* [Bugfix][DCP] Set default CUDAGraphMode to PIECEWISE for DCP (@FENP #26574)\r\n* [TEST][BUG FIX] Fix DP GPU_ID issue (@xuechendi #26442)\r\n* Update `Optional[x]` -> `x | None` and `Union[x, y]` to `x | y` (@hmellor #26633)\r\n* [Feature] Add support for naver/splade-v3 (BERT-based sparse embedding model) (@gjgjos #26339)\r\n* [Models][Qwen3VL] Speedup `fast_pos_embed_interpolate` (@lgeiger #26647)\r\n* [easy] fix pre commit error on trunk (@hl475 #26665)\r\n* [CI/Build] Add tool to build vllm-tpu wheel (@mgoin #19165)\r\n* [Misc] cache result of disable_inplace (@bnellnm #26666)\r\n* [Bugfix][Core]Fix block table out-of-range issue in priority scheduling (@quanliu1991 #26661)\r\n* [FIX] Throwing an exception when the model does not support pool tasks (#25840) (@yyzxw #25855)\r\n* docs: wrong command in structured_outputs README (@yihong0618 #26677)\r\n* [Model] Fix  Skywork R1V mlp (@jeejeelee #26673)\r\n* [Model] Add reasoning_parser and tool_parser for Ernie45 thinking (@CSWYF3634076 #25027)\r\n* Ignore large reformatting PRs in `git blame` (@hmellor #26690)\r\n* [Model][0/N] Improve all pooling task | clean up (@noooop #25817)\r\n* [ResponseAPI] Simplify input/output message serialization (@Jialin #26620)\r\n* [Bugfix] Fix out of bound index issue for Jina-embedding-v3 RoPE with cuda graph (@Isotr0py #26687)\r\n* [unrevert] Add batch invariant kernel override for FlashInfer backend [2/n] (@bwasti #26373)\r\n* [Hardware][CPU] Disable torch.compile for RISC-V to prevent APIError (@ihb2032 #26693)\r\n* [FEATURE]: Use pydantic validation in `multimodal.py` config (@andycandy #26629)\r\n* [UX] Speedup DeepGEMM warmup with heuristics (@mgoin #25619)\r\n* [P/D] [NixlConnector] kv load recovery integration (@wseaton #26171)\r\n* [Misc] Separate prompt logging to debug (@aitsvet #26713)\r\n* [CI/Build] upgrade compressed-tensors to 0.12.2 to address LGPLv3 (@csy1204 #26501)\r\n* [Bugfix][Rocm] fix qr error when different inp shape (@haoyangli-amd #25892)\r\n* [Bugfix][Speculative Decoding] Extend Eagle quantization config fix to llama_eagle.py (@rahul-tuli #26590)\r\n* [Model] Use merge_by_field_config for MM models (M-N) (@DarkLight1337 #26710)\r\n* [Log] Optimize Startup Log (@yewentao256 #26601)\r\n* [CI][Release][Arm64]: Build arm64 release for gpu arch 8.9 (@cyb70289 #26698)\r\n* [Quantization] [Performance] Enable Marlin GEMM kernels for the calibration-free RTN-based quantization (@sakogan #26051)\r\n* [Frontend][1/N] Improve all pooling task | Support FP16 Embedding Base64 (Still uses fp32 by default). (@noooop #26414)\r\n* [CI] Fix mypy for `vllm/distributed` (@yewentao256 #26593)\r\n* [CI Perf]Prune Tests in kernel/mamba (@kfhfar #26538)\r\n* [Bug] Fix Assertion error DeepEP/csrc/kernels/intranode.cu:928: 'false and Unsupported type' (@yewentao256 #26532)\r\n* [FrontEnd] UNREVERT CompilationConfig overhaul (#20283): deprecate use_inductor in favor of backend, simplify custom_ops  (@morrison-turnansky #26502)\r\n* Pruning kernel Core Tests (@kfhfar #26727)\r\n* [ResponseAPI] Further polish message serialization and unit tests (@Jialin #26728)\r\n* Add tests for chunked prefill and prefix cache with causal pooling models (@maxdebayser #26526)\r\n* [Misc][DP] support customized aggregated logger for dp (@luccafong #24354)\r\n* [UX] Replace VLLM_ALL2ALL_BACKEND with --all2all-backend (@mgoin #26732)\r\n* [compile] Enable sequence parallelism for full cuda graph without specifying compile sizes (@angelayi #26681)\r\n* [Easy] Fix env type check errors from VLLM_DEBUG_LOG_API_SERVER_RESPONSE (@Jialin #26742)\r\n* [build][torch.compile] upgrade depyf version (@youkaichao #26702)\r\n* [torch.compile] Unwrap fused_marlin_moe custom op (@varun-sundar-rabindranath #26739)\r\n* [Feature][Quantization] auto_round format add support for regex (@n1ck-guo #24024)\r\n* Add support for the /rerank endpoint in vllm bench serve (@maxdebayser #26602)\r\n* [Docs] Add a start tag to build.inc.md (@windsonsea #26747)\r\n* Fix lora tests failure in TPU CI due to the removal of LoRA bias (@vanbasten23 #26723)\r\n* [CI] [ROCm] Automate CC list for ROCm related issue (@vllmellm #26753)\r\n* Adding the test-amd.yaml for test definitions for the AMD backend. (alternative PR) (@Alexei-V-Ivanov-AMD #26718)\r\n* scheduler.py: Update the name of the default scheduler. (@ryanli #26758)\r\n* [Model][Bugfix]fix ernie45 load failed due to ernie45 eplb code (@CSWYF3634076 #26684)\r\n* [CI/Build] Use 127.0.0.1 instead of localhost in utils (@yeqcharlotte #26750)\r\n* fix(frontend): always include usage, when configured to do so (@max-wittig #20983)\r\n* [Plugin] Make plugin group clear (@wangxiyuan #26757)\r\n* [Bugfix] Standardize merging multimodal embeddings (@DarkLight1337 #26771)\r\n* [Model] Use merge_by_field_config for MM models (O-P) (@DarkLight1337 #26776)\r\n* [NIXL][HeteroTP]Enable KV transfer from HND prefill to NHD decode (@xuechendi #26556)\r\n* [Chore] Use `max_transformers_version` for Qwen-VL test (@DarkLight1337 #26792)\r\n* Don't allow `typos` to fix by default (@hmellor #26785)\r\n* [Doc] ruff format some Python examples (@DarkLight1337 #26767)\r\n* [CI] Fix test_tool_id_kimi_k2 (@chaunceyjiang #26787)\r\n* [Chore] Remove `SupportsV0Only` interface and update supported models docs (@DarkLight1337 #26783)\r\n* [Feature] Change vllm.py with pydantic validation (@VladOS95-cyber #26726)\r\n* [CI/Build] Cleanup LoRA test (@jeejeelee #26752)\r\n* [DCP] Support Decode Context Parallel (DCP) for GQA with FlashAttention (@FENP #24864)\r\n* Adjusted the model order of the model registration file (@princepride #26798)\r\n* use combo kernel to fuse qk-norm and qk-rope (@BoyuanFeng #26682)\r\n* [issues template] Encourage the author implement their own ideas (@noooop #26671)\r\n* [KVConnector][Metrics] Aggregate scheduler-side KVConnectorStats (@QierLi #26046)\r\n* [Feature][Responses API] Stream Function Call - harmony (@chaunceyjiang #24317)\r\n* Revert \"[issues template] Encourage the author implement their own ideas\" (@noooop #26814)\r\n* [Config] Remove Unused Environment Variable `VLLM_DISABLE_PAD_FOR_CUDAGRAPH` (@yewentao256 #26743)\r\n* Update coveragerc and add codecov.yml for path fixes (@rzabarazesh #26435)\r\n* [CI] Raise VLLM_MAX_SIZE_MB to 500 due to failing Build wheel - CUDA 12.9 (@mgoin #26722)\r\n* [Kernel][MoE] Add MoE tunings for GLM 4.6-FP8 and GLM 4.5 Air on NVidia B200 (@zklapow #26818)\r\n* [CI Failure] Fix tests with missing TinyLlama-1.1B-Chat-v1.0-FP8-e2e (@mgoin #26816)\r\n* llama4_vision_rope: add HIP override to accept (q, k) and avoid (positions, q, k) mismatch (@hl475 #26790)\r\n* [Attention][Spec Decode] FlashMLA spec decode support (@MatthewBonanni #26541)\r\n* [Core] Reuse empty block lists whenever possible in KVCacheBlocks to mitigate GC costs (@Jialin #24964)\r\n* Notice for deprecation of AutoAWQ (@HDCharles #26820)\r\n* [Perf] Cache vllm.env.__getattr__ result to avoid recomputation (@Jialin #26146)\r\n* Added MoE configs for llama 4, H200 device with tp=4/8 tuning (@Dhruvilbhatt #26837)\r\n* fix: response_format for completion (@Nan2018 #23212)\r\n* [Minor] Group async_scheduling related fields in model runner init (@njhill #26736)\r\n* remove attn output view kernel (@BoyuanFeng #26680)\r\n* [Core] Streamline some structured output related code (@njhill #26737)\r\n* [CI Failure] Fix torchao dep failure for Quantization Test (@mgoin #26824)\r\n* [frontend][gptoss] Add per turn stats into Harmony Context (@lacora #25061)\r\n* [WideEP][P/D] Add usage stats for DP+EP and KV Connector (@tlrmchlsmth #26836)\r\n* [torch.compile] Fix tests for torch==2.9 inductor partition (@ProExpertProg #26116)\r\n* [Core][Easy] Use envs.__getattr__ for all Unify to environment variable access (@Jialin #26810)\r\n* [Bugfix]fix Qwen3 xml tool parser (@Zhikaiiii #26345)\r\n* [BUGFIX][NIXL] quick fix for 'assert self.connector_worker is not None' in get_kv_connector_stats (@xuechendi #26851)\r\n* Disable FlashInfer sampler by default (@mgoin #26859)\r\n* [Frontend][torch.compile] CompilationConfig Overhaul (#20283): name change  compilation level to compilation mode, deprecation compilation level (@morrison-turnansky #26355)\r\n* [Bugfix] Fixes prefix-repetition benchmark script (@kouroshHakha #26828)\r\n* [Model] Add DeepSeek-V3.1 reasoning parser (split from PR #24972) (@taohui #25589)\r\n* [Docs] Move build.inc into arm.inc (@windsonsea #26862)\r\n* [CI/Build][Bugfix] fix qutlass cmake error when set QUTLASS_SRC_DIR (@izhuhaoran #26773)\r\n* [Feature] default --extra-body param to disable thinking in vllm bench serve (@lengrongfu #26784)\r\n* [BugFix] Patch inductor partitioning logic (@angelayi #26735)\r\n* [Bugfix] Fix qwen3-omni audio truncation issue (@Isotr0py #26815)\r\n* [Graph Partition] pass tests for decorator (@BoyuanFeng #26831)\r\n* [Bugfix][Multi Modal] Fix incorrect Molmo token processing (@sangho-vision #26873)\r\n* [DSA][MLA] Tiny refactor on DeepSeek to make it reusable for different backends (@MengqingCao #26656)\r\n* [Misc] Use helper function to generate dummy messages in OpenAI MM tests (@DarkLight1337 #26875)\r\n* [bugfix] Lazy import cv2 (@angelayi #26869)\r\n* [Deepseek-V3.2][Kernel] Integrate cuda indexer k cache gather (@zyongye #26456)\r\n* [CI/Build] Add Qwen2.5-VL-7B-Instruct ChartQA Accuracy Tests in CI (@zhewenl #21810)\r\n* [CI] Fix mypy for `vllm/executor` (@yewentao256 #26845)\r\n* [Doc] ruff format remaining Python examples (@DarkLight1337 #26795)\r\n* [doc] add Context Parallel Deployment doc (@youkaichao #26877)\r\n* [Misc] Update TritonLanguagePlaceholder to have attributes that are used by Flash Linear Attention ops. (@madongfly #26853)\r\n* [Fix] Remove divisibility requirement between num_kv_heads and tp_size in bailing_moe (@ant-yy #26876)\r\n* [Easy] Get rid of unnecessary paraenthesis in kv_cache_manager (@Jialin #26842)\r\n* [Platform] allow platform to init dp group (@wangxiyuan #22243)\r\n* [Lora]Load tuned multi-lora kernel configs from json files (@li2haipeng #26319)\r\n* [Model][2/N] Improve all pooling task | Support multi-vector retrieval (@noooop #25370)\r\n* [Misc] Remove `isort` and `yapf` ignores (@DarkLight1337 #26888)\r\n* [Misc] rename torch_dtype to dtype (@wangxiyuan #26695)\r\n* chore: remove unused marker (@max-wittig #26890)\r\n* [BugFix] Patch inductor memory plan logic (@BoyuanFeng #26878)\r\n* [Chore] Separate out `vllm.utils.func` (@DarkLight1337 #26904)\r\n* [Chore] Separate out `vllm.utils.async_utils` (@DarkLight1337 #26913)\r\n* Lower severity of log when model info cache misses due to exception (@hmellor #26917)\r\n* Olmo 3 tool parser and tests (@pdasigi #26143)\r\n* [Feature]: Use pydantic validation in observability.py config (@cern1710 #26637)\r\n* [ModelOpt] Remove NVFP4 MoE K%16==0 constraint (@XiaobingSuper #26891)\r\n* [Chore] Clean up CODEOWNERS (@WoosukKwon #26923)\r\n* [NVIDIA] Add support for cudnn fp4 gemm via flashinfer (@kaixih #26107)\r\n* Vectorize RMS norm variance using vectorize_read_with_alignment (@bbeckca #26234)\r\n* support flashinfer_fp4 moe for 5090 gpu (@XiaobingSuper #26669)\r\n* [Bug] Temporally Disable `VLLM_ALLREDUCE_USE_SYMM_MEM` by Default (@yewentao256 #26925)\r\n* Move query quantization to attention layer for Flashinfer & Triton. (@adabeyta #26534)\r\n* Adjusting AMD test composition 2025-10-14 (@Alexei-V-Ivanov-AMD #26852)\r\n* [Qwen3-Next] Add tuned MoE config for Qwen3-Next FP8 on H100 tp2 (@felixzhu555 #26887)\r\n* [Bugfix] reasoning_parser parameter handling in run_batch.py (@inc-jeong #26225)\r\n* [ROCm][FEAT] Fuse DeepSeek shared experts into AITER fused_moe ops (@kliuae #24097)\r\n* [CI] Enable Blackwell Llama4 MoE tests (@mgoin #26731)\r\n* [BUG] Allow runai_streamer_sharded in config check (@ahao-anyscale #26958)\r\n* [bugfix] Fix SP + PP without specifying compile size (@angelayi #26955)\r\n* [BugFix] Work around graph partition x torch.compile cache issue (@zou3519 #26956)\r\n* [DOC][XPU]update feature parity with Intel GPU (@xuechendi #26954)\r\n* [Chore] Rename `utils` submodules (@DarkLight1337 #26920)\r\n* [PERF] Qwen3-next MTP speedup (change bool mask indexing to index_select / index_copy to reduce d2h) (@vadiklyutiy #26437)\r\n* Deepseek-v3 Batch Invariant on 8xH100 (@bwasti #26609)\r\n* [CI/Build] Update expected beam search output for Phi3V (@DarkLight1337 #26978)\r\n* [Hardware][CPU][PowerPC]Disable torch.compile() in toptopk sampling (@Akashcodes732 #26987)\r\n* [CI/Build] Fix AMD import failures in CI (@zhewenl #26841)\r\n* [Benchmark] Use truncation by default for pooling benchmarks (@DarkLight1337 #26992)\r\n* [Chore] Separate out `vllm.utils.collections` (@DarkLight1337 #26990)\r\n* [Model][Bugfix] fix ernie45 vl run failed from shared experts optimization (@CSWYF3634076 #26885)\r\n* Cleanup code after Python 3.10 upgrade (@lgeiger #26520)\r\n* [MISC] fix import violations for re and triton modules (@llsj14 #26654)\r\n* [Bugfix] Correct LayerNorm epsilon parameter in modernbert.py (@bogdanminko #27008)\r\n* [Benchmark] Show E2EL by default for pooling models (@DarkLight1337 #27014)\r\n* [Attention] Tune CUTLASS MLA num_splits (@MatthewBonanni #26846)\r\n* [NIXL] Improve request_finished() debug logs (@markmc #25665)\r\n* [docs] standardize Hugging Face env var to `HF_TOKEN` (deprecates `HUGGING_FACE_HUB_TOKEN`) (@yankay #27020)\r\n* [CI] Replace large models with tiny alternatives in tests (@tahsintunan #24057)\r\n* [Feature] Add process_weights_after_loading to AttentionImpl (@lengrongfu #26870)\r\n* [Model] Fix Qwen3VL mm mapping (@jeejeelee #27027)\r\n* Fix Qwen2.5 VL image grid docstring (@skyloevil #27033)\r\n* Support `set` in the CLI generation (@hmellor #27031)\r\n* [gpt-oss][1/N] EZ: refactor serving_responses for modularity (@qandrew #26948)\r\n* Support block size of 256 used by Intel HPU (@mandy-li #26883)\r\n* [Compressed Tensors] Always clone output for compile robustness (@kylesayrs #26849)\r\n* Adding Warmup to Benchmark Serving (@kimbochen #26943)\r\n* [Bug] Fix batch invariant test `has` to `is` (@yewentao256 #27032)\r\n* [GPTOSS][DP/EP][Marlin] Enable GPTOSS Batched DP/EP using Marlin kernels (@varun-sundar-rabindranath #25997)\r\n* [Feature] Migrate DeepGEMM API from `get_m_alignment_for_contiguous_layout` to `get_mk_alignment_for_contiguous_layout` (@yewentao256 #26935)\r\n* [CI] Prune Quantization Tests and skip compilation (@mgoin #27038)\r\n* [Bug] Add Assertion for `random-input-len` / `random-output-len` (@yewentao256 #26834)\r\n* [small][batch invariance] Rename the env and internal flags to simplify usage (@bwasti #26855)\r\n* Refactor Transformers backend to use mixins (@hmellor #26906)\r\n* [NVIDIA] [Perf] Update to leverage flashinfer trtllm FP4 MOE throughput kernel (@jiahanc #26714)\r\n* [torch.compile] Passing only necessary compilation config to inductor pass config (@luccafong #27041)\r\n* [Chore] Separate out `vllm.utils.import_utils` (@DarkLight1337 #27022)\r\n* [torch.compile] fix simple inductor graph partition test (@BoyuanFeng #27050)\r\n* Remove unused imports (@lgeiger #26972)\r\n* vllm bench serve shows num of failed requests (@tomasruizt #26478)\r\n* [Docs] Reduce custom syntax used in docs (@hmellor #27009)\r\n* [Perf] Exploit out-of-band buffers in shm_broadcast (@njhill #26961)\r\n* disable graph partition in custom op (@BoyuanFeng #26952)\r\n* [Bugfix][Qwen] fixes the weights dtype in qwen3_next: it is actually a bfloat16 (@sighingnow #27030)\r\n* [Core] Change `execute_model_with_error_logging()` to be a ctx manager (@njhill #27060)\r\n* [Bugfix] Fix ReplicatedLinearWithLoRA  (@jeejeelee #27065)\r\n* [Kernel] Lazy import FlashInfer (@jeejeelee #26977)\r\n* [CI/Build] Update Llama4 eval yaml (@zhewenl #27070)\r\n* [Model] Always use Transformers backend for PaliGemma and Gemma3-MM (@DarkLight1337 #26715)\r\n* [Model] Add support for LightOnOCR (@staghado #26916)\r\n* [CI/Build] Update compressed tensor test path to fix CPU CI (@bigPYJ1151 #27068)\r\n* [Kernel][Performance] Fuse float cast and renormalize to topk softmax kernel  (@izhuhaoran #26717)\r\n* [CI] fix docs build failed (@chaunceyjiang #27082)\r\n* Update troubleshooting.md and remind VLLM_TRACE_FUNCTION usage (@Prowindy #27069)\r\n* [VLM][Refactor] Remove useless func `get_input_positions` in `MRotaryEmbedding` (@MengqingCao #27088)\r\n* [Docs] Replace all explicit anchors with real links (@hmellor #27087)\r\n* [Docs] Replace `rst` style double-backtick with `md` single-backtick (@hmellor #27091)\r\n* [Model]Improve Qwen3VLMoeForConditionalGeneration packed_modules_mapping (@jeejeelee #27096)\r\n* [Harware][AMD][Model] Triton MoE tuning configs for GLM-4.5 for MI350 and MI355 (@rkarhila-amd #25586)\r\n* Fix incorrect docstring for stop_profile() method (@hyongtao-code #27101)\r\n* [torch.compile] Enable attention and allreduce fusion without custom ops enabled (@ProExpertProg #24604)\r\n* [CI] Nixl integration tests (@NickLucche #27010)\r\n* [Data-parallel] Allow DP>1 for world_size > num_gpus on node (8) (@patrickvonplaten #26367)\r\n* [bugfix] Qwen3-VL fix video incorrect timestamp calculations while do_sample_frames=True (@wulipc #27104)\r\n* [CI] Remove forbidden slash (@NickLucche #27112)\r\n* [ROCM] MoE fp4 CK kernel (@maleksan85 #26545)\r\n* [ROCm][Bugfix][Model] Fix illegal memory access when running qwen3_moe models with  rms_norm (Qwen3-235B-A22B,  Qwen3-30B-A3B, etc.) (@rasmith #26192)\r\n* [Bugfix] [AITER] [ROCm] Fix Quark MoE Quant Config and AITER Fused MoE quant type logic (@vllmellm #27029)\r\n* [Chore] Remove unused `PolyNorm` layer (@Isotr0py #27110)\r\n* [Bugfix] Use PIECEWISE cudagraphs on Blackwell if max_model_len > 131072 (@mgoin #27114)\r\n* [Minor] Remove unnecessary error message (@zhuohan123 #27115)\r\n* [V1][Spec Decode] Fix greedy temperature detection after sampler refactor (@Pradyun92 #27077)\r\n* [Test] Make `test_failure` more stable for batch invariance (@yewentao256 #27054)\r\n* [BugFix][Core] Fix error when enable async-scheduling in multi-node env (@lhtin #25887)\r\n* [Perf]  Add H100 fused MoE config (@skyloevil #25398)\r\n* [CI/Build] tests(v1): feed Triton attention the (num_blocks, 2, …) KV cache layout in backend-correctness tests (@hl475 #26663)\r\n* [GPT-OSS] Structure_Tag support for gpt-oss tool-call in cot (@Hanchenli #25515)\r\n* [Misc] Rev DeepEP (@varun-sundar-rabindranath #27122)\r\n* [DOC][FEATURES][CPU]update cpu feature for v1 (@xuechendi #27135)\r\n* [Test] Add test for /health endpoint on engine failure (@dongbo910220 #26074)\r\n* [Chore] Separate out `vllm.utils.mem_utils` (@iAmir97 #27143)\r\n* [Feature] Batch Invariant: Support DeepGEMM and Blackwell (@yewentao256 #27127)\r\n* [fix][cpu] fix prefill attention in CPU attention backend (@fadara01 #27035)\r\n* [Misc] Refactor `get_kv_cache_spec` into `AttentionLayerBase` (@NickLucche #26587)\r\n* [Models][QwenVL] Remove unnecessary `.contiguous()` calls (@lgeiger #27106)\r\n* [Chore] Clean up pytorch helper functions in `vllm.utils` (@Isotr0py #26908)\r\n* Fix incorrect string formatting in barrier timeout exceptions (@hyongtao-code #27149)\r\n* [Minor] Add some clarifying comments to recent changes (@njhill #27130)\r\n* [BugFix] Fix failing gemma-3-1b-it test: `test_lm_eval_accuracy_v1_engine[google/gemma-3-1b-it]` (@LucasWilkinson #27111)\r\n* [Chore] Separate out profiling utilities from vllm.utils (@dongbo910220 #27150)\r\n* [BugFix] fix graph partition signature (@BoyuanFeng #27139)\r\n* [BugFix] Disable fp8 kv-cache by default for DeepSeek V3.2 (@LucasWilkinson #27121)\r\n* [V1][Metrics][Plugin] Add plugin support for custom `StatLoggerBase` implementations (@ptovam #22456)\r\n* [Minor] Remove unused env variable (@WoosukKwon #27161)\r\n* [BugFix] Fix lazy imports involving outlines_core (@22quinn #27158)\r\n* [Chore] Separate out hashing utilities from vllm.utils (@dongbo910220 #27151)\r\n* [Benchmark] Convenience script for multiple parameter combinations (@DarkLight1337 #27085)\r\n* output type conversion fix (@jianyuh #27159)\r\n* [Chore] Separate out `vllm.utils.network_utils` (@iAmir97 #27164)\r\n* [Misc] Move utils to avoid conflicts with stdlib, and move tests (@DarkLight1337 #27169)\r\n* [Bugfix] Fix error with penalties when speculative decoding and structural output are enabled (@southfreebird #26586)\r\n* Fix typo in ValueError message: use `kv_role` instead of `kv_disagg_role` (@hyongtao-code #27166)\r\n* [Model][VLM] Support Bee-8B Model (@uyzhang #27012)\r\n* [LoRA] LoRA cuda graph specialization (@andylolu2 #25914)\r\n* [Kernel] Accelerate solve_tril with TMA (@ZJY0516 #26746)\r\n* AArch64 CPU Docker pipeline #26931)\r\n* Nemotron Nano V2 VL + EVS Video Support (@BloodAxe #27107)\r\n* [Kernel][Model] Tune fused_moe Triton configs for Qwen3-30B A3/A3B on H100 (FP8/BF16) (@shivampr #26268)\r\n* [Bugfix][CI] Fix `Distributed Tests (4 GPUs)` async_sched+ray test (@NickLucche #27195)\r\n* [Feature][Quantization] auto_round support for mixed bits quantization (@n1ck-guo #23812)\r\n* [ROCm] enable some tests in entrypoints test groups on AMD (@Concurrensee #26725)\r\n* [ez] add uv lock to gitignore (@qandrew #27212)\r\n* [Quantization] Automatically infer AWQ `modules_to_not_convert` field (@Isotr0py #26909)\r\n* [V0 Deprecation] Remove V0 metrics code (@njhill #27215)\r\n* [cpu] Dispatch un-quantized linear to oneDNN/ACL by default for AArch64 (@fadara01 #27183)\r\n* create is_in_the_same_node on cpu (@helunwencser #26832)\r\n* [Frontend] Enforce tokenize=False when applying chat template (@russellb #27205)\r\n* [Feature][Kernel]FusedMoE LoRA (@wcwuwc #21229)\r\n* [BugFix] GPT-OSS Attention DP + MoE TP weight loading issue (@nvpohanh #24032)\r\n* [ModelOpt] Load w13/w2_input_scale for all experts, nvfp4 (@wenscarl #26135)\r\n* [Bugfix] Fix gpt-oss w4a8 DP/EP on B200 (@varun-sundar-rabindranath #26729)\r\n* [Bugfix] Fix broken MTP weight loading for FP8 KV Scales (@benchislett #27227)\r\n* [Fix][Spec Decode] Fix llama4 draft loading with different quantization (@linzebing #27136)\r\n* [Nixl] Minor refactor to handshake related metadata (@NickLucche #26410)\r\n* [MM][Core] Decouple ViT backend from LM backend (@ywang96 #27061)\r\n* [Deepseek v3.2] Optimize top_k_per_row (@dcampora #26763)\r\n* [Chore] Separate out NCCL utilities from vllm.utils (@dongbo910220 #27197)\r\n* [CI] Install pre-release version of `apache-tvm-ffi` for `flashinfer` (@hmellor #27262)\r\n* [ROCM] Enable CompressedTensorsWNA16 (@JartX #27187)\r\n* Add @pavanimajety to .github/codeowners (@pavanimajety #27213)\r\n* [ROCm] Update Triton, Torch, and AITER branches for ROCm base Dockerfile (@micah-wil #27206)\r\n* [Feature] Batch Invariant for R1 TP 8 on Blackwell (@yewentao256 #27229)\r\n* [Bugfix][P/D] Reduce num_threads used by nixl ucx backend (@dagrayvid #27196)\r\n* [V0 Deprecation] Remove V0 executors (@njhill #27142)\r\n* [Bugfix] fixes the decoding metadata of dense mla's fp8 kvcache. (@sighingnow #27144)\r\n* Update PyTorch to 2.9.0+cu129 (@huydhn #24994)\r\n* [Performance] Dual stream execution of \"shared_experts\" and \"selected_experts\" inside FusedMoE (@alexm-redhat #26440)\r\n* Updated xgrammar backend to not deny supported string formats (@ExtReMLapin #27253)\r\n* [Bugfix] skip cuda graph for drafter when running with eager (@benchislett #26821)\r\n* [P/D] KVConnector for decode benchmarking (@tlrmchlsmth #25986)\r\n* [Deepseek v3.2] Remove extra logics in indexer (@IwakuraRein #26465)\r\n* [DOC] [ROCm] Add ROCm quickstart guide (@vllmellm #26505)\r\n* [CI] Nixl integration tests DP-EP (@NickLucche #27199)\r\n* [Benchmark] Add plot utility for parameter sweep (@DarkLight1337 #27168)\r\n* [torch.compile] Enable silu_mul_fp8_quant fusion without custom ops enabled (@ZJY0516 #27146)\r\n* [1/N][Platform] Cleanup useless function (@wangxiyuan #26982)\r\n* Update release pipeline for PyTorch 2.9.0 (@huydhn #27303)\r\n* Remove last `level` references not removed #26355 (@hmellor #27260)\r\n* fixed reasoning streaming with tool_choice=\"required\" (@ExtReMLapin #24108)\r\n* [Frontend][3/N] Improve all pooling task | Support binary embedding response (@noooop #27066)\r\n* [Bugfix][CPU] Disable dual stream execution for experts on CPU (@bigPYJ1151 #27320)\r\n* [Bug] Raise error for `LLM(data_parallel_size=k)` single-process DP Usage (@yewentao256 #27282)\r\n* Bugfix - pass 'max_num_tokens_padded' into 'moe_lora_align_block_size' (@gnovack #27311)\r\n* [Core] Handle MoE LoRA edge cases (@jeejeelee #27335)\r\n* [docs] Update v1 metrics design doc (@markmc #27332)\r\n* Mirroring changes in test-pipeline.yaml into test-amd.yaml (@Alexei-V-Ivanov-AMD #27242)\r\n* [Chore] Separate out optional dependency checks from vllm.utils (@dongbo910220 #27207)\r\n* [Model] Upstream Deepseek-OCR model (@Isotr0py #27247)\r\n* [NIXL] Terminate handshake listener thread in shutdown (@markmc #26404)\r\n* [Bug] Fix DeepSeek-V2.5-1210-FP8 issue (@yewentao256 #27267)\r\n* [bugfix] remove unused parameters to reduce unnecessary vram usage (@ReinForce-II #26789)\r\n* [Bugfix] Add missing 'is_internal_router' attribute to FusedMoEWithLoRA (@jeejeelee #27351)\r\n* [NIXL] use Host buffer to support TP_ratio > 1 for XPU (@xuechendi #27140)\r\n* [Bugfix] Make `get_mrope_input_positions` instance methods (@DarkLight1337 #27342)\r\n* [Bugfix] Fix HF format InternVL large variants video processing (@Isotr0py #27330)\r\n* [Frontend] Require flag for loading text and image embeds (@russellb #27204)\r\n* [P/D] Dynamic `kv_output_aggregator` collect size (@NickLucche #26734)\r\n* Support Anthropic API /v1/messages Endpoint (@LiuLi1998 #22627)\r\n* [Bugfix] Disable FlexAttention direct block mask building for encoder-only models (@Isotr0py #27344)\r\n* [Model] Revert PR #26715: Restore custom PaliGemma and Gemma3-MM impl… (@lucianommartins #27309)\r\n* [Doc] Fix numbering sequence in prefix caching (@gigit0000 #27357)\r\n* [Prefix Cache] Use LoRA name for consistent KV-cache block hashing (@sagiahrac #27211)\r\n* [Feature] publisher default set zmq in kv_event config (@lengrongfu #26915)\r\n* [BugFix] bugfix for Flash Attention MLA with full cuda graph IMA following pr-25490 (@Daisy-Ma-coder #27128)\r\n* [Chore] Separate out system utilities from vllm.utils (@dongbo910220 #27201)\r\n* [MLA] Bump FlashMLA (@MatthewBonanni #27354)\r\n* [Bugfix] Fix deepseek-ocr multi-image inference and add `merge_by_field_config=True` with tensor schema support (@Isotr0py #27361)\r\n* [Bugfix] Fix SLA tuner initialization (@DarkLight1337 #27355)\r\n* [Bugfix] Fix incorrect kv cache metrics in grafana.json (@fangpings #27133)\r\n* [Bugfix][Core] running queue index leakage exception (@CLFutureX #26754)\r\n* [CORE] Support Prefix Caching with Prompt Embeds (@qthequartermasterman #27219)\r\n* [V1][spec decode] return logprobs for spec decoding (@TheEpicDolphin #26060)\r\n* [Model] Add num_cached_tokens for PoolingRequestOutput (@noooop #27378)\r\n* [Chore] Remove duplicate `has_` functions in vllm.utils (@jonathanc-n #27372)\r\n* [CI/Build] Fix Prithvi plugin test (@DarkLight1337 #27393)\r\n* [Bugfix] Fix args settings for guided decoding args (@luccafong #27375)\r\n* [CI/Build] Fix AMD CI: test_cpu_gpu.py (@zhewenl #27388)\r\n* add SLA information into comparison graph for vLLM Benchmark Suite (@louie-tsai #25525)\r\n* [CI] Reorganize entrypoints tests (@chaunceyjiang #27403)\r\n* [Metrics] [KVConnector] Add connector prefix cache hit rate stats (@ptovam #26245)\r\n* [Model] Add MoE support for NemotronH (@tomeras91 #25863)\r\n* Run mypy on the lowest supported Python version instead of system Python (@hmellor #27048)\r\n* [Bugfix] Honor --mm_encoder_attn_backend when used (@bradleyhd #27124)\r\n* [Feature] Pydantic validation for speculative.py (@Navya1707 #27156)\r\n* [Misc] Remove use of CUDA_VISIBLE_DEVICES for device selection (fix DP slow startup time &c) (@ilmarkov #26709)\r\n* [CI/Build] Remove unnecessary flags from test registry (@DarkLight1337 #27353)\r\n* [Frontend][4/N] Improve all pooling task | Add plugin pooling task (@noooop #26973)\r\n* Mirroring the test definitions (2025-10-22) (@Alexei-V-Ivanov-AMD #27362)\r\n* [Bugfix] Fix dp_chunking enablement logic in FusedMoE layer (@alexm-redhat #27220)\r\n* [Bugfix][ROCm][DeepSeek] Fix for forward_hip in rope for DeepSeek (@gshtras #27373)\r\n* [Bugfix] Fix AWQ marlin layer skipping (@Isotr0py #27416)\r\n* [Misc] Add triton_kernels dependency (@varun-sundar-rabindranath #27370)\r\n* [Chore] Separate out `vllm.utils.platform_utils.py` (@jonathanc-n #27374)\r\n* [Attention] Fix FlashMLA metadata builder arguments for q_len > 1 (@MatthewBonanni #27368)\r\n* [Bugfix][DP] Fix creating too many DP Placement Groups (@kebe7jun #26880)\r\n* [Model] Siglip Embedding Support (@piood #27324)\r\n* [Hardware][POWERPC] Disable oneDNN path in vllm/model_executor/layers/utils.py for Powerpc (@Akashcodes732 #27422)\r\n* Granite 4.0 quark quantization support (@xiao-llm #26944)\r\n* Fix pooling adapters for Transformers backend  (@hmellor #27338)\r\n* [Kernel] Add GPTQv2 format support for low-bit or asymmetric quantization, by adapting gptq_gemm (@xxxxyu #26092)\r\n* [Misc] Add TPU usage report when using tpu_inference. (@hfan #27423)\r\n* [Bugfix][CI] Move resolving cudagraph_mode before initializing attn_metadata_builder (@fhl2000 #27427)\r\n* Fix EventPublisherFactory logic for disabled KV cache events (@usberkeley #27419)\r\n* [Chore] remove structural tags logging lines (@aarnphm #27451)\r\n* [Bugfix] Fix Pydantic union resolution for ResponseFunctionToolCall in Responses API (@strinczer #26706)\r\n* [Misc] Avoid \"PyTorch non-writable tensors\" warning in RayPPCommunicator (@ruisearch42 #27443)\r\n* [Docs] remove v1 column for embedding models (@piood #27446)\r\n* [MM][Bugfix] Replace `PatchEmbed`'s conv3d to linear layer (@Isotr0py #27418)\r\n* [BugFix] Fix torchrun DP with LLM class (@22quinn #27395)\r\n* [Refactor] move tool parsing logic from protocol.py to the tool parser (@chaunceyjiang #27383)\r\n* [Benchmark] Enable benchmark to run with `encoding_format=\"bytes\"` (@DarkLight1337 #27467)\r\n* Fix AArch64 CPU Docker pipeline #27331)\r\n* [MISC] `cudagraph_capture_sizes`  related improvements (@fhl2000 #26016)\r\n* Fix test named tool use (@chaunceyjiang #27458)\r\n* [Doc] Fix minor issues in docs/design/metrics.md (@draftbk #27436)\r\n* [cpu][fix] Fix onednn_mm crash on consecutive matmuls with same M,K,N and different dtype (@fadara01 #27472)\r\n* [compile] Turn standalone_compile back on (@zou3519 #27460)\r\n* [NIXL][BUGFIX] delay done_recving queue cleanup to bottom of get_finished (@xuechendi #27297)\r\n* [Bugfix] Fix MultiConnector stats reconstruction across process boundaries (@kouroshHakha #27366)\r\n* [Attention] Add MLA prefill backend: trtllm_ragged_attention_deepseek (@minosfuture #26397)\r\n* [Bugfix] Fix interns1-vit qk norm code path (@Isotr0py #27480)\r\n* [CI/Build] Fix test_torch_utils in AMD CI (@zhewenl #27317)\r\n* [Document] Add ms-swift library to rlhf.md (@hjh0119 #27469)\r\n* [Perf][Async Scheduling] Remove CPU->GPU sync in dummy_run (@lhtin #27455)\r\n* [Distributed] Basic set of configuration for large EP deployment on GB200 (@wpc #27328)\r\n* [Log] Optimize Startup Log (@yewentao256 #26740)\r\n* [Misc][DP] Guard mxfp4 implementation selection (@varun-sundar-rabindranath #27484)\r\n* [KVConnector] Migrate the LMCache integration code to be vLLM native (@ApostaC #25542)\r\n* [CI] Add tests for cudagraph (@ZJY0516 #27391)\r\n* Revert \"[Misc] Remove use of CUDA_VISIBLE_DEVICES for device selectio… (@zhuohan123 #27502)\r\n* [Core][Hybrid allocator + kv connector 1/n] Enable hybrid allocator + KV cache connector (@KuntaiDu #25712)\r\n* [Misc] Simplify max tokens in multimodal registry (@DarkLight1337 #27500)\r\n* [Attention] Add missing kv cache scale setup (@MatthewBonanni #27490)\r\n* [CI/Build] Refactor processing tests (@DarkLight1337 #27470)\r\n* [CI/Build] Use CPU for mm processing test on CI (@Isotr0py #27522)\r\n* [BUGFIX][ROCM] ViT FlashAttention on ROCm (no GFX9) and contiguous on qwen3vl ROCm TORCH_SDPA (@JartX #27190)\r\n* [Bugfix] Fix processor initialization for model from modelscope instead of HF (@lengrongfu #27461)\r\n* [Bugfix] fix empty prompts for async-engine mode in benchmark throughput (@luccafong #27494)\r\n* [Doc] Remove Molmo warning (@DarkLight1337 #27527)\r\n* [Doc] Fix links to GH projects (@DarkLight1337 #27530)\r\n* [Chore]:Extract math and argparse utilities to separate modules (@yeshsurya #27188)\r\n* Revert \"[CI/Build] Use CPU for mm processing test on CI (#27522)\" (@DarkLight1337 #27531)\r\n* [CI/Build] Update causal-conv1d installation (@DarkLight1337 #27529)\r\n* [Model][MiniMax-M2] Support MiniMax-M2 Model (@rogeryoungh #27535)\r\n* fix m2 test (@youkaichao #27536)\r\n* Fix MiniMax-M2 copyright (@rogeryoungh #27537)\r\n* [Model][Bugfix] fix ernie45 moe 300B SharedFusedMoE output tuple (@CSWYF3634076 #27316)\r\n* [Model] Use merge_by_field_config for MM models (Qwen series) (@DarkLight1337 #27546)\r\n* [Docs] reemove the incorrect `enable_reasoning` parameter  (@yyzxw #27550)\r\n* [Performance][LoRA] add context varying params to 'do_not_specialize' in fused moe lora (@gnovack #27445)\r\n* [Model] Deprecate `merge_by_field_config=False` (@DarkLight1337 #27551)\r\n* [Doc] Slight improvement to M2 and beyond (@jeejeelee #27554)\r\n* [Kernel] Adding split_K implementation for fused_moe_lora (@dcmaddix #27291)\r\n* [Misc] Clean up utils (@DarkLight1337 #27552)\r\n* [Bugfix] Limit the default value of `max_model_len` when it is not specified by users (@shen-shanshan #27556)\r\n* [Bugfix] Fixed when return_token_ids=False, the first event still contains prompt_token_ids. (@chaunceyjiang #27561)\r\n* [cpu][perf] Fix low CPU utilization with VLLM_CPU_OMP_THREADS_BIND on AArch64 (@fadara01 #27415)\r\n* [Kernel] Enable moe LoRA kernel support FP16 (@jeejeelee #27468)\r\n* [Hybrid] Added supports_mamba_prefix_caching Protocol (@Josephasafg #27339)\r\n* [Model] Siglip2 Model Support (@piood #27566)\r\n* [Bugfix][LoRA][FusedMoE] Select MxFP4 Backend based on LoRA Enablement (@varun-sundar-rabindranath #27487)\r\n* fixing mm placeholder replacement issue with gemma3 (@tingtingtangmeta #27538)\r\n* [Chore]: Stream tokens vs characters in tool call parser tests (@bbrowning #26513)\r\n* [Misc] Clean up more utils (@DarkLight1337 #27567)\r\n* [ROCm] Update AITER branch for ROCm base docker (@micah-wil #27586)\r\n* Code quality improvements: version update, type annotation enhancement, and enum usage simplification (@usberkeley #27581)\r\n* [gpt-oss][2/N] Support input_messages in responsesRequest (@qandrew #26962)\r\n* [Bugfix][CI] Fix config resolving logic with remote models (@ywang96 #27610)\r\n* [Stability fix] turn off HMA allocator when connector is set (@KuntaiDu #27592)\r\n* [Bugfix] fixed inconsistent finish_reason handling between V0 and V1 engines (@chaunceyjiang #27555)\r\n* [ROCm] [Doc] Update ROCm installation docs  (@vllmellm #27327)\r\n* [Hardware][AMD][Model] Triton MoE tuning configs for GLM-4.6 for MI300X (@minatoaquaMK2 #27323)\r\n* [Bugfix][CPU] Fallback oneDNN linear to torch linear to fix half gemm support on legecy platforms (@bigPYJ1151 #27526)\r\n* [Core][Bookkeeping Optimization] Update against numpy view of is_token_ids tensor (@Jialin #27618)\r\n* [CI/Build] Fix amd model executor test (@zhewenl #27612)\r\n* Fix a robust parsing issue in KimiK2ToolParser that causes IndexError (@wangln19 #27565)\r\n* [V0 Deprecation] Remove vestigial V0 logits_processors.py file (@njhill #27601)\r\n* [Bugfix] In LongRoPE, decide short vs long based on max_model_len (@MatthewBonanni #27431)\r\n* [Misc] Separate out `utils.counter` and move `utils.Device` to engine (@DarkLight1337 #27588)\r\n* [Bug] Fix shape issue for eplb expert weights (@yewentao256 #27589)\r\n* [compile] Add enable_prompt_embeds to compile hash. (@zhxchen17 #27285)\r\n* [Hybrid] Add mamba_block_size to Engine Args (@Josephasafg #27289)\r\n* [compile] Disable dynamo guards check for AOT compilation. (@zhxchen17 #27288)\r\n* fix: allow HuggingFace standard chat template params via **kwargs (@wangln19 #27622)\r\n* [Core] Enable async scheduling for external_launcher mode (@22quinn #27394)\r\n* [Bugfix][Frontend] validate arg priority in frontend LLM class before add request (@junpuf #27596)\r\n* [BugFix] Also consider RAY_EXPERIMENTAL_NOSET_* when storing compilation cache (@HollowMan6 #27294)\r\n* [nit]: lmcache integration import (@sammshen #27600)\r\n* [FLA] Introduce Kimi Delta Attention(KDA) to VLLM (@zhiyuan1i #27654)\r\n* [Bugfix] Fix allocation & free logic of SingleWriterShmRingBuffer (@imkero #27117)\r\n* [Bugfix][CI] Fix v1 attention backend tests and add CI coverage (@mmangkad #26597)\r\n* [Misc] Make `LayerBlockType` a `Literal` instead of `Enum` (@DarkLight1337 #27658)\r\n* [compile] Add fallback path to AOT compile when serialization fails. (@zhxchen17 #27350)\r\n* Add load pattern configuration guide to benchmarks (@mpashkovskii #26886)\r\n* [Misc] Make reorder batch also separate extends (@LucasWilkinson #27367)\r\n* [Test] Batch Invariant: Unit test using parameterized backend (@yewentao256 #27478)\r\n* [Core] Scheduler: Publish connector events after output (@orozery #25875)\r\n* [AsyncScheduling] Make async overlap work with logprobs (@njhill #27615)\r\n* [Misc][qwen2_5_vl][torch.compile] Enable `supports_torch_compile` on generic nn.Module and demonstrate speedup on Qwen Vision model (@Lucaskabela #23207)\r\n* [Bug] Fix deepep low latency use nvlink by default (@yewentao256 #27677)\r\n* [Core] Early return in SlidingWindowManager.remove_skipped_blocks (@Jialin #27673)\r\n* Install pre-built xformers-0.0.32.post2 built with pt-2.9.0 (@huydhn #27598)\r\n* Revert \"Install pre-built xformers-0.0.32.post2 built with pt-2.9.0\" (@simon-mo #27714)\r\n* [Build] Revert triton_kernels requirements (@varun-sundar-rabindranath #27659)\r\n* [NIXL][XPU] update name of nixl wheel (@zhenwei-intel #27631)\r\n* [Model] Fix Qwen3VL and Qwen3Omni after torch.compile changes (@lgeiger #27705)\r\n* [KV cache] Fix lmcache connector (@Shaoting-Feng #27681)\r\n* [CI/Build][Bugfix]Fix Quantized Models Test on AMD (@zhewenl #27712)\r\n* [Bugfix] Fix non-contiguous tensor error in `rocm_unquantized_gemm_impl` (@zhewenl #27605)\r\n* [Speculators] Move tests + fix integration (@dsikka #27308)\r\n* [CI/Build] Move pre-commit only scripts to `tools/pre_commit` (@DarkLight1337 #27657)\r\n* [perf] Enable concurrent execution of \"shared_experts\" and \"selected_experts\" in qwen3-next (@ZJY0516 #27578)\r\n* [Bugfix] Fix modular kernel tests (@bnellnm #27707)\r\n* [Frontend] [gpt-oss] Tool json call parsing error retry (@alecsolder #27675)\r\n* [Frontend] [gpt-oss] Mcp type bug (@alecsolder #27689)\r\n* [Fix] import get_kv_cache_torch_dtype error in vllm_v1_adapter.py (@KevinCheung2259 #27670)\r\n* [Misc] Raise error for missing video metadata in `MultiModalDataParser` (@Isotr0py #27664)\r\n* Feature/video support in random mm dataset (@BloodAxe #25963)\r\n* [chore] Remove models weight on S3 logic (@khluu #27725)\r\n* [VLM] Add Qwen3-VL generation test (@Isotr0py #25185)\r\n* [CI/Build] Skip cpu offloading test on AMD (@zhewenl #27690)\r\n* [Frontend] Add `vllm bench sweep` to CLI (@DarkLight1337 #27639)\r\n* Fix MiniMax-M2 rmsnorm precision and remove useless code (@rogeryoungh #27627)\r\n* [ROCm][Platform] Add MI308X device id in _ROCM_DEVICE_ID_NAME_MAP (@sammysun0711 #27623)\r\n* [CI] Fix flaky `test_two_responses_with_same_prev_id` test  (@NickLucche #27745)\r\n* [Chore] Optimize P2PNCCLEngine `http_address` (@yewentao256 #27488)\r\n* [Core] Exposing engine sleep & wake_up state as prometheus metrics (@dumb0002 #24176)\r\n* [FIXBUG] Qwen3VL hallucinations without Contiguous on Torch.SDPA (@JartX #27744)\r\n* `use_aot_compile` should respect `VLLM_DISABLE_COMPILE_CACHE` (@BoyuanFeng #27698)\r\n* [CI/Build] Test torchrun with 8 cards (@22quinn #27548)\r\n* [Bug] Raise error explicitly if using incompatible backend (@yewentao256 #27424)\r\n* [KVConnector] Add metrics to Prometheus-Grafana dashboard (@NickLucche #26811)\r\n* [Bug] Fix DeepEP low latency `assert self.batched_router_logits.size(-1) == full_router_logits.size(-1)` Bug (@yewentao256 #27682)\r\n* [BugFix] Fix handling of resumed reqs in `SharedStorageConnector` (@njhill #27719)\r\n* [Bug] Fix DBO IMA issue for DeepEPHT (@yewentao256 #27666)\r\n* [Temp fix] Disable torch.compile for Qwen2.5 VL's VisionBlock temporarily.  (@huachenheli #27760)\r\n* [XPU][bugfix] fix rope for llama4 and deepseek (@yma11 #25145)\r\n* [Bugfix] mamba-block-size is set for vision language model (@heheda12345 #27773)\r\n* [XPU] Update latest IPEX 2.8 release (@jikunshang #27735)\r\n* [BugFix] Handle unscheduled requests properly when async scheduling (@njhill #27756)\r\n* [Feat] Adds runai distributed streamer (@bbartels #27230)\r\n* kernels/moe test pruning (@kfhfar #27053)\r\n* [BugFix] Reordering extend logic fix (@LucasWilkinson #27739)\r\n* [Benchmark] Cleanup deprecated nightly benchmark and adjust the docstring for performance benchmark (@KuntaiDu #25786)\r\n* Add more dims for batch invariant shims (@bwasti #27489)\r\n* use stringData in secret yaml to store huggingface token (@yitingdc #25685)\r\n* [CI/Build]Add eval config for Qwen3-235B-A22B-Instruct-2507-FP8 (@hl475 #27113)\r\n* [BugFix][VL] Fix FA selection on Qwen2.5-VL (@zhewenl #27790)\r\n* [V0 deprecation] Remove VLLM_USE_V1 usage in config module (@wangxiyuan #27784)\r\n* [CI Failure] fix test_default_mm_loras (@hl475 #27795)\r\n* [CI] Fix mypy for `vllm/v1/core` and `vllm/v1/engine` (@yewentao256 #27108)\r\n* [Bugfix] Improve GPU validation logging in Ray fallback scenarios (@sairampillai #25775)\r\n* [Frontend][Doc][5/N] Improve all pooling task | Polish encode (pooling) api & Document. (@noooop #25524)\r\n* [CI Failure] Fix test_kv_cache_model_load_and_run (@hl475 #27717)\r\n* [Model] Introduce Kimi Linear to vLLM (@zhiyuan1i #27809)\r\n* [KV offload] Enable CPU KV offload on CUDA alike Platforms (@zhewenl #27770)\r\n* [Model][Ouro] Support Ouro Model (@FlamingoPg #27794)\r\n* [Bugfix][CPU] Fix MRoPE dispatch on the CPU backend (@bigPYJ1151 #27800)\r\n* [BugFix] Stopgap - Flashinfer Autotuner + GPT-OSS + DP/TP (@varun-sundar-rabindranath #27762)\r\n* [Misc] Replace CUDA_VISIBLE_DEVICES in DP with torch.cuda.set_device for device selection on cuda-like devices (@ilmarkov #27564)\r\n* [Docs] add Shanghai Meetup - 2025/10 (@kebe7jun #27545)\r\n* Reapply \"Install pre-built xformers-0.0.32.post2 built with pt-2.9.0\" (@huydhn #27768)\r\n* [MTP] Refactor mtp predictor to avoid d2h operation (@MengqingCao #27643)\r\n* [Model] Use the same fused_moe configs for all H200 devices (@bufferoverflow #23642)\r\n* [Bugfix] Fix 2 precommit issues - (mamba_block_size, kv_cache_config) (@tlrmchlsmth #27811)\r\n* [Core][Bookkeeping] Update cu_num_accepted_tokens for all req_index (@Jialin #27629)\r\n* [EP/DP][API Server] Enable DP-aware routing in OpenAI API requests (@Prowindy #24945)\r\n* [Fix] Skip `record_sleep_state` logic in `PrometheusStatsLogger` if not in dev mode (@SumanthRH #27789)\r\n* [Refactor] Remove `VLLM_DEEPEP_LOW_LATENCY_ALLOW_NVLINK` (@yewentao256 #27750)\r\n* [Core][Perf] Only invoke save_new_computed_blocks when computed blocks are not empty (@Jialin #27799)\r\n* [Feature] Batch invariant torch.compile (@PaulZhang12 #27660)\r\n* [BugFix] Fix broken import in initialize_ray_cluster() (@njhill #27838)\r\n* [Misc] Make all tool scripts executable (@MatthewBonanni #27831)\r\n* [CI/Build][Intel] Enable performance benchmarks for Intel Gaudi 3 (@jakub-sochacki #26919)\r\n* [CI Test] Add Scheduled Integration Test (@yewentao256 #27765)\r\n* [benchmark] Make request IDs unique across clients by default (@eicherseiji #27723)\r\n* [Hardware][Powerpc] Fix VLLM_CPU_OMP_THREADS_BIND=\"auto\"  low CPU utilization for Power (@Akashcodes732 #27734)\r\n* [Kimi-Linear] Correct prefixes and add compatibility to AWQ quants (@toncao #27834)\r\n* [Bugfix] Avoid too small block m/n for FlexAttention kernel option (@Isotr0py #27853)\r\n* [BugFix] Don’t compute reorder threshold when there are no attention groups (@hl475 #27861)\r\n* [Perf] Decouple torch op from GDA to leverage torch.compile (@ZJY0516 #27871)\r\n* [CI/Build] Add gpt-oss LoRA test (@jeejeelee #27870)\r\n* [Bugfix] Allow 64-bit integer values for LoRA IDs to avoid overflow/truncation (@shadeMe #27876)\r\n* [Bugfix] Fix broken MRoPE for GLM-4.1V/GLM-4.5V (@Isotr0py #27860)\r\n* [Bugfix] Missing NIXL metadata for handshake initialization if instance spans multi-node (@GuanLuo #26338)\r\n* Docs update tpu install instructions (@RobMulla #27824)\r\n* [bugfix] Missing cached item in beam search (@fake0fan #27874)\r\n* fix incorrect type annotation in KimiMLP (@skyloevil #27885)\r\n* Flashinfer_CUTLASS_MOE fuses quantization for TP (@wenscarl #27223)\r\n* [Cleanup] Remove no-longer-used `SpeculativeConfig.enable_chunked_prefill` (@njhill #27826)\r\n* [Feature] Pydantic validation for scheduler.py and structured_outputs.py (@vrdn-23 #26519)\r\n* Add FLASHINFER_MLA to test_mla_backends and add B200 CI run (@MatthewBonanni #27663)\r\n* Batch invariance doc (@bwasti #27839)\r\n* [Hybrid] A simpler algorithm to find kernel_block_size (@heheda12345 #26476)\r\n* [Core] Async scheduling + structured outputs compatibility (@njhill #26866)\r\n* [Kernel] Enable FusedMoEModularKernel  support  bias (@jeejeelee #27754)\r\n* [Bugfix] Fix KDA output (@jeejeelee #27905)\r\n* [Multimodal][XPU]Enable vision attn backend for xpu platform (@yma11 #27525)\r\n* Adding SplitK in fused_moe_lora kernel (@yugong333 #27818)\r\n* [CI/Build] Bump transformers version (@DarkLight1337 #27528)\r\n* [Bugfix] [Model] Missing MRoPE function definition from `KeyeForConditionalGeneration` (@tjtanaa #27895)\r\n* [Add] cmdline argument parsing for KV cache offloading modules (@ApostaC #27621)\r\n* feat(benchmarks): support HF model names in multi-turn benchmark (@ai-jz #27850)\r\n* [Docs] Mock all imports for docs (@hmellor #27873)\r\n* [V0 deprecation] Remove VLLM_USE_V1 usage in platform and v1 module (@wangxiyuan #27798)\r\n* [Bugfix] DeepSeek V3.2 MTP metadata & CUDA graph issues (@xiaohajiayou #26779)\r\n* [Bugfix] Python 3.10 compatibility for `Self` (@DarkLight1337 #27918)\r\n* [Core][TPU] Support TPU Data Parallalism (@wenxindongwork #27365)\r\n* [BugFix] Fix mixed penalties batch with async scheduling (@njhill #27910)\r\n* Adds anthropic /v1/messages endpoint to openai api_server (@bbartels #27882)\r\n* [KV offload] Offloading connector async scheduling support (@KevinCheung2259 #27648)\r\n* [CI/Build] Fix flaky test_transcription_validation.py::test_basic_audio_gemma (@bbrowning #27924)\r\n* [Bugfix] Fix Qwen Omni audio inference (@DarkLight1337 #27920)\r\n* Performance fix MistralTokenizer: cache special ids and tokens (@juliendenize #27925)\r\n* [V1] [Hybrid] Mamba1 Automatic Prefix Caching (@Josephasafg #26377)\r\n* [Misc] Provide Siglip2 chat template (@DarkLight1337 #27939)\r\n* [Bugfix][llm]: Abort orphaned requests when llm.chat() batch fails (@Flink-ddd #27420)\r\n* [BugFix][LoRA] use adapter_id instead of id field of lora_request (@biswapanda #27728)\r\n* [Frontend] Align finish_reason when tool is called with OpenAI (@n0gu-furiosa #25054)\r\n* [Hybrid] Pass kernel block size to builders (@tdoublep #27753)\r\n* [Bugfix] Padded Eagle Specdec with Chunked Prefill (@Flechman #26263)\r\n* [XPU]Refine Dockerfile.xpu, avoid oneccl dependency issue (@jikunshang #27964)\r\n* Add ORCA endpoint load metrics support (@efimki #24905)\r\n* [CI/Build] Remove the flaky gpt-oss lora test (@jeejeelee #27966)\r\n* [Model] Add PaddleOCR-VL Model Support  (@zhang-prog #27758)\r\n* Early exit for MoE LoRA kernels (@gnovack #27131)\r\n* [Bugfix] Skip gs:// model paths for speculator detection (@pwschuurman #27846)\r\n* [BUG] Make 'binary' default option for saving torch compile artifacts when using standalone_compile (@ahao-anyscale #27616)\r\n* [CI/Testing] Add basic single node dual batch overlap test (@LucasWilkinson #27235)\r\n* [Spec Decode] Integrate Suffix Decoding from Arctic Inference (@aurickq #25784)\r\n* [Feature][Benchmarks] Support `inf` burstiness (@sducouedic #26941)\r\n* [Bugfix][Qwen][Multimodal] Move Qwen2_5_vl sdpa to custom op and reenable compile (@Lucaskabela #27764)\r\n* [Bugfix] change FlashMLA reorder_batch_threshold (@MatthewBonanni #27777)\r\n* [Docs] add runai_streamer_sharded to LoadConfig (@andyxning #27937)\r\n* Add TP parameter to attention tests (@MatthewBonanni #27683)\r\n* [Bugfix][plugin] fla crash on plugin (@ILikeIneine #27322)\r\n* [Bugfix] Fix MoE Routing Simulation (@tlrmchlsmth #28002)\r\n* Remove the tpu docker image nightly build. (@QiliangCui #27997)\r\n* [Bugfix][ROCm] Fix ViT rotary embeddings for torch.compile compatibility on ROCm (@vllmellm #27748)\r\n* [LoRA] Lora shrink swizzle (@li2haipeng #27694)\r\n* [Refactor] Lazy import tool_parser (@chaunceyjiang #27974)\r\n* [NIXL][XPU] Pin NIXL version to 0.7.0 (@zhenwei-intel #27849)\r\n* [Metrics] Enable sleep state metric outside of dev mode (@markmc #27867)\r\n* [Bug] Batch invariant: Fix flash attn MLA `RuntimeError: scheduler_metadata must have shape (metadata_size)` (@yewentao256 #27884)\r\n* [CPU]Improve dynamic 4bit moe performance (@xiangze-arm #27240)\r\n* [CI/Build] Update LM Eval Version in AMD CI (@zhewenl #27944)\r\n* [KV Connector] Make KVCacheConfig an explicit constructor argument (@markmc #27887)\r\n* [Model] fix ernie45 reasoning_parser (@CSWYF3634076 #27973)\r\n* [CI/Build] Fix OpenAI API correctness on AMD CI (@zhewenl #28022)\r\n* [BugFix][Performance] Restore flashinfer autotuning for all scenarios (@varun-sundar-rabindranath #27904)\r\n* Load tuned fused_moe_lora shrink and expand kernel configs separately (@yugong333 #27435)\r\n* Support using Int4PreshuffledTensor after loading (@jerryzh168 #26066)\r\n* [Core] Enable StatLogger in LLMEngine (@zhuohan123 #28020)\r\n* [Model][Bugfix] fix pipeline parallelism support for NemotronH (@tomeras91 #27968)\r\n* [Model] add optimal triton fused moe configs for NemotronH MoE (@tomeras91 #27967)\r\n* [Kernels] Isolate modular kernel code from FusedMoEMethodBase subclasses. (@bnellnm #27123)\r\n* [BugFix] Fix incorrect preallocated sampled_token_ids tensor size (@njhill #28025)\r\n* [Perf] SM100 - add swap AB optimization to CUTLASS FP8 GEMM (@LyrisZhong #27284)\r\n* [PERF] Decouple projections from GDN custom op (@vadiklyutiy #27512)\r\n* [model] Add support for openPangu_Ultra_MoE (@yt0428 #27521)\r\n* [PerfFix] Avoid separate thread for MP executor shm spin (@njhill #28012)\r\n* [AsyncScheduling] Don't schedule past request max_tokens (@njhill #27922)\r\n* Remove deprecated `--rope-scaling` and `--rope-theta` (@hmellor #28006)\r\n* [ROCm][Perf] New design on ROCm AITER MHA backend Implementation (@ganyi1996ppo #25763)\r\n* Added disable rule to track files under benchmarks/lib (@nadavkluger #28048)\r\n* [Multimodal] Make MediaConnector extensible. (@huachenheli #27759)\r\n* [ROCm] gemm_a16w16 upstreaming (@maleksan85 #26969)\r\n* Revert \"[PERF] Decouple projections from GDN custom op\" (@vadiklyutiy #28080)\r\n* [Qwen3-Next] MOE configs for A100-SXM4-80GB TP4 TP8 (@toulzx #27740)\r\n* [XPU] Add gpt-oss model support for Intel GPU (@jikunshang #27786)\r\n* [CI/Build] Enable some fixed tests in AMD CI (@zhewenl #28078)\r\n* [V0 deprecation] Remove VLLM_USE_V1 usage in most modules (@wangxiyuan #27955)\r\n* [Bugfix] Fix encoder-only model support for transformers backend (@Isotr0py #28021)\r\n* [BugFix] Fix DCP Assert (AssertionError: DCP not support reorder_batch_threshold > 1 now.) (@LucasWilkinson #28100)\r\n* [Model, Core] Support Granite Speech & LoRA for STT (@alex-jw-brooks #24455)\r\n* [Refactor] Lazy-loaded reasoning_parser (@chaunceyjiang #28092)\r\n* [Refactor] to simplify and extract the shared logic between chat completion and responses (@chaunceyjiang #27961)\r\n* [bugfix] fix wrong `dcp_local_seq_lens` calc (@pisceskkk #27518)\r\n* [Hybrid allocator + kv connector] revert connector test changes related to hybrid allocator (@KuntaiDu #28011)\r\n* [Misc] fix import error for DeepSeekR1ReasoningParser (@chaunceyjiang #28114)\r\n* Fix excessive logging noise by reducing the log level of the MinimaxM2ToolParser import success message (@minatoaquaMK2 #27635)\r\n* Bugfix: Cutlass FP8 FusedMoE bad scaling factors (@amirkl94 #27255)\r\n* [Graph Partition][Cache] Use inductor partition ops config (@BoyuanFeng #27702)\r\n* [XPU] Enable custom routing functions in IPEX for Llama4 (@frost-intel #28004)\r\n* add kimi reasoning parser (@MoyanZitto #28128)\r\n* [DCP] check return_lse for all layers in dcp (@heheda12345 #27929)\r\n* [BugFix] Support EP/DP + EPLB with MTP (@ilmarkov #25311)\r\n* Enabling cooperative multi-gpu tests on multi-gpu nodes (@Alexei-V-Ivanov-AMD #27986)\r\n* [ROCm][MLA] Support block-size > 1 for AITER MLA backend  (@ganyi1996ppo #27224)\r\n* [Bugfix] Validate custom logits processor xargs for online serving (@Isotr0py #27560)\r\n* [misc] add vLLM Beijing Meetup (@jjzhang #28127)\r\n* [Kernel] Fuse computation of g and beta for Gated Delta Net (@ZJY0516 #28095)\r\n* [Core] add support for reasoning parser plugins (@walterbm #28075)\r\n* [Bugfix] vLLM should check Inductor config for compile cache enablement status (@gmagogsfm #27637)\r\n* [FlashInfer] Avoid FlashInfer block_size 16 + head_size 256 on blackwell (@heheda12345 #27994)\r\n* [CI]: Add LMCache Unit Tests (@sammshen #27852)\r\n* [Feature] Extend batch invariant torch.compile to B200 (@PaulZhang12 #27856)\r\n* [Bugfix] Fix Qwen3-Reranker-8B load (@noooop #28117)\r\n* [Docs] Clean up README_TUNING.md (@windsonsea #28088)\r\n* [Hardware][IBM Z] Optimize s390x Dockerfile (@R3hankhan123 #28023)\r\n* [Chore] Remove Nemotron-Nano-VL config copy (@Isotr0py #28126)\r\n* [Docs] Add guide to debugging vLLM-torch.compile integration (@zou3519 #28094)\r\n* [Feature]: Add corrupted request metric to V1 metrics system. (@atalhens #27306)\r\n* [CI/Build] Update checking logic in cutlass_group_gemm_supported  (@zhewenl #27948)\r\n* [CI/Build] Fix `test_defaults_with_usage_context` in AMD CI (@zhewenl #27926)\r\n* [Core][Hybrid allocator + connector 2/n] Unify `remove_skipped_blocks` by `get_last_useful_token` (@KuntaiDu #25431)\r\n* [Debugging] Add annotation for easier trace analysis (@dayeol #22496)\r\n* [PERF] Decouple projections from GDN custom op. Attempt 2 (@vadiklyutiy #28083)\r\n* [Bug] Fix cpu disable shared_experts `VLLM_DISABLE_SHARED_EXPERTS_STREAM` (@yewentao256 #28157)\r\n* [Bug] Fix env string `\"0\"` same to `True` (@yewentao256 #28159)\r\n* [Feature] Enable TP + EP `shared_experts` overlap with router, 3.7% E2E performance improvement (@yewentao256 #28164)\r\n* [CI Failure] `nm-testing/Qwen2-0.5B-Instruct-FP8-SkipQKV` was removed from HF. Skip it in tests (@vadiklyutiy #28170)\r\n* [Misc] Remove the duplicate code (@chaunceyjiang #28111)\r\n* [Chore] Clean up deepseek v2/v3 config copy (@Isotr0py #28055)\r\n* [Core][MM] Use non-blocking CPU-GPU copy of multimodal data (@lgeiger #28141)\r\n* Make the cv2 dependency optional (@cmpute #27780)\r\n* [CI] Add compile/test_multimodal_compile.py to CI (@gmagogsfm #28151)\r\n* [flashinfer] fix FI all2all with FI cutlass moe (@mxz297 #28166)\r\n* Patch Mistral Tokenizer (@juliendenize #28146)\r\n* Fix hard-coded parameter name in gemma3n.py (@seungduk-yanolja #27946)\r\n* [CPU] Enable torch profiling (@aditew01 #28130)\r\n* [V0 deprecation]clean up is_v1_supported_oracle (@wangxiyuan #28116)\r\n* [Bugfix][Kernel] fix merge attn states when both prefix and suffix are empty (@courage17340 #28181)\r\n* [Frontend] OpenAI Responses API supports Tool/Function calling - non-harmony  (@chaunceyjiang #26874)\r\n* [CPU]Improve cpu fused moe perf (@xiangze-arm #27244)\r\n* Disable nm-testing models with issues in CI (@mgoin #28206)\r\n* [Docs] Switch to directory style URLs (@hmellor #28058)\r\n* [Kernel][Model] Tune fused_moe Triton configs for MiniMax-M2 on H100 (@minatoaquaMK2 #28200)\r\n* [Doc] Add Arm CPUs are on the list of supported targets in vLLM (@milpuz01 #26018)\r\n* [HARDWARE][CPU] Add Option for Disabling Binding to Specific CPU Cores (@StanHatko #27953)\r\n* [Frontend] Fix logging format when enable response logging (@esmeetu #28049)\r\n* CODEOWNERS: Add myself as reviewer on security docs (@russellb #28216)\r\n* [Structured outputs] Upgrade llguidance to 1.3.0 (@andylolu2 #28039)\r\n* Add llama 4 scaling support (@juliendenize #28145)\r\n* [Chore] eliminate duplicated and unconditional object serialization in anthropic messages api (@vicoooo26 #27792)\r\n* [ROCm] triton fp8 kernel (@maleksan85 #27058)\r\n* [Doc]: Make extraInit containers fully configurable in helm chart (@HanFa #27497)\r\n* [Test] Add non-MoE DP test coverage (@MatthewBonanni #28235)\r\n* [BugFix] Fix FusedMoELoRA + ModularKernel Integration (@varun-sundar-rabindranath #28237)\r\n* Fix failing test for CRadio (@BloodAxe #27738)\r\n* Speed up mm processor kwargs per request by spliting dynamic and static kwargs (@LJH-LBJ #26483)\r\n* [Multimodal][torch.compile] Add compilation config field for turning off ViT/MM compile (@Lucaskabela #28242)\r\n* [CI/Build] Loosen STT LoRA Translate Check (Flaky Test) (@alex-jw-brooks #28247)\r\n* Add runai model streamer e2e test for GCS (@amacaskill #28079)\r\n* Fix issues from #28242 (@hmellor #28257)\r\n* [amd][gptoss] Perf gain because of block alignment (@smitkadvani #28024)\r\n* [Bug] Fix missing token_ids for reasoning parser models in chat completions   #28246 (@baonudesifeizhai #28256)\r\n* [CI] Reduce Blackwell Fusion test runtime by filtering tests and only run all tests in nightly (@Copilot #28074)\r\n* [Kernel] LoRA triton kernels support PDL (@jeejeelee #27402)\r\n* [Perf] Introduce FlattenLogprobs to store logprobs results to reduce GC overhead (@Jialin #28171)\r\n* [FixBug]Aeala/ShareGPT_Vicuna_unfiltered marked as multimodal benchmark (@princepride #28265)\r\n* [CPU]Avoid repeated random sample compile (@xiangze-arm #28260)\r\n* [Misc][Model][Refactor] Pass the prefix into Linear layers (@MengqingCao #28259)\r\n* [fix] Revert \"fixing mm placeholder replacement issue with gemma3\" (@khluu #28285)\r\n* [Core][MM] Add mechanism to configure multimodal fields which should stay on CPU (@lgeiger #28168)\r\n* [Bugfix] Use latency MOE backend as default for Flashinfer and other misc fixes (@pavanimajety #27439)\r\n* [CLI] add --max-tokens to `vllm complete` (@Iceber #28109)\r\n* [Feature] Default `ignore_eos` True for `random` dataset (@yewentao256 #28227)\r\n* [Log] update shm wait time msg (@BoyuanFeng #28255)\r\n* Revert \"[PerfFix] Avoid separate thread for MP executor shm spin (#28012)\" (@NickLucche #28289)\r\n* [README] Add Arm CPUs to the list of supported targets (@fadara01 #28290)\r\n* [doc] add guide about the provided PTX was compiled with an unsupported toolchain (@youkaichao #28305)\r\n* [Build] Fix release pipeline failing annotation (@simon-mo #28272)\r\n* [Bugfix] Fix and add tests for GptOss reasoning parser (@benchislett #28000)\r\n* [Core] Rework handling of async scheduling config (@njhill #28250)\r\n* [PerfFix] Avoid separate thread for MP executor shm spin (take 2) (@njhill #28319)\r\n* Update Flashinfer from `v0.4.1` to `v0.5.2` (@hmellor #27952)\r\n* [XPU] Enable Expert parallel for MoE models (@jikunshang #28263)\r\n* remove resolve_op_overloads and use splitting_ops directly (@BoyuanFeng #28081)\r\n* [Bugfix][LoRA][Spec Decode] Support LoRA with speculative decoding (@xiaohongchen1991 #21068)\r\n* Update gpu.rocm.inc.md to add support for AMD Ryzen AI MAX / AI 300 Series (gfx1151, gfx1150) (@hammmmy #28308)\r\n* [Perf][DeepSeek] Add sigmoid+bias fusion to fused_grouped_topk from TRTLLM (@mgoin #28124)\r\n* Bump arctic-inference requirement (@aurickq #28174)\r\n* [bugfix] support eagle with lora cudagraph specialization (@gnovack #28318)\r\n* [Model] Consolidate Deepseek-MoE implementation with DeepSeek-v2 (@Isotr0py #28101)\r\n* Refactor CPU/GPU extension targets for CMake build (@ashahba #28026)\r\n* [flashinfer][fix] do not check nvcc availability when using pre-downloaded cubins (@mxz297 #27990)\r\n* [Attention] Remove max cudagraph size limit of 992 (@22quinn #27840)\r\n* `reasoning_content` -> `reasoning` (@hmellor #27752)\r\n* [Bugfix] Update device name for H200 detection (@robertgshaw2-redhat #28349)\r\n* [Bugfix] Spec decode + structured output + spec model max len edge case (@andylolu2 #28298)\r\n* [DCP] Support dcp kv_cache interleave size > 1 (@zhangsicheng5 #26696)\r\n* Enhance run_cluster.sh for multi-NIC support (@evberrypi #28328)\r\n* [Feat] Drop-in Torch CUDA Profiler (@benchislett #27841)\r\n* Remove setuptools upper bound constraint (<80) (@ColeMurray #28337)\r\n* [Bugfix] Fix test fused quant layernorm tests (@ElizaWszola #27865)\r\n* [Performance][gpt-oss] Revert gpt-oss max cudagraph size to 1024 (@mmangkad #28345)\r\n* [chore] Move some wikimedia images to S3 (@khluu #28351)\r\n* fix: close issue 28338 by fixed python version (@yihong0618 #28339)\r\n* [Misc] fix typo and add detailed log (@andyxning #28178)\r\n* [ROCm] Add env to enable/disable aiter triton gemm (@sarckk #28321)\r\n* [Misc] Add some comments in qwen3-next (@ZJY0516 #28267)\r\n* [CI] Fix flaky `test_eagle_correctness` test (@NickLucche #28364)\r\n* [Core] Simplify async KV output aggregation (@njhill #28327)\r\n* [Core] Separate out attention metadata building logic from prepare inputs (@LucasWilkinson #26764)\r\n* [BugFix] Fix cu_num_generated_tokens slicing logic in LogprobsLists.slice() method (@usberkeley #28214)\r\n* [CI/Build] Temporary fix to LM Eval Small Models (@zhewenl #28324)\r\n* [Kernel] Fix fused_gdn_gating (@ZJY0516 #28343)\r\n* [ROCm][Platform] Add RX7900XTX device id in _ROCM_DEVICE_ID_NAME_MAP (@JartX #28279)\r\n* [CI] lora/test_mixtral.py : Add additional expected outputs due to flakiness (@varun-sundar-rabindranath #28322)\r\n* [Hardware][AMD][Model] Add Triton MoE tuning support and optimized configs for Qwen3 omni for MI308X (@sammysun0711 #28373)\r\n* [V0 deprecation] Remove no longer used `get_metadata_cls` (@LucasWilkinson #28370)\r\n* Restore PlaMo2 unit test as `pfnet/plamo-2-1b` now supports `transformers >=4.56` (@Alnusjaponica #28019)\r\n* [Metrics] Refactor LoRA state tracking (@markmc #26801)\r\n* [bugfix] fix siglip batch text output error (@piood #28365)\r\n* [Fix] optimize visual token mask with caching and multi-token support (@bo-ke #28374)\r\n* Add @tjtanaa to codeowner for ROCm and multi-modal (@tjtanaa #28360)\r\n* [Rocm][fused_moe][fp4] view weight to torch.float4_e2m1fn_x2 when running aiter fused moe for fp4 model (@zejunchen-zejun #27474)\r\n* [Kernel] Optimization of the mm_k operator. (@caozuoba #28280)\r\n* [RFC][ROCm][AITER] Keep all AITER kernels in `_aiter_ops` class like `_custom_ops` and `_ipex_ops` (@vllmellm #24490)\r\n* [V0 Deprecation] Remove unused `context_len` and `seq_len` from M-RoPE (@DarkLight1337 #28395)\r\n* [Bugfix] Fix persistent_masked_m_silu_mul_quant tests (@varun-sundar-rabindranath #28366)\r\n* [Performance] Support FP8 flashinfer TRTLLM MOE on Qwen3 and Qwen-3next (@jiahanc #27492)\r\n* [Bugfix] Fix llguidance backend, rollback when EOS was encountered (@Flechman #25905)\r\n* [FA/Chore] Bump FA version for FP8 two-level accumulation  (@jmkuebler #27889)\r\n* [Bugfix][EPLB] Disabled shared expert overlap when EPLB is enabled (@SageMoore #28377)\r\n* [Misc] Add more scoping for improved trace (@frank-wei #28329)\r\n* [BugFix] Fix DeepGEMM over-allocating workspace (@LucasWilkinson #28254)\r\n* [Frontend][2/n] remove empty content from _parse_tool_calls_from_content (@qandrew #28331)\r\n* [CI] Fix Plugin Tests Tests (@robertgshaw2-redhat #28413)\r\n* [ROCm] Add missing gemm_a8w8_blockscale import (@sarckk #28378)\r\n* [PERF] Allreduce fusion. Support torch native matching. Tuning of the thresholds (@ilmarkov #24248)\r\n* [Perf] Move gc.freeze logic from EngineCoreProc to EngineCore for better coverage (@Jialin #27896)\r\n* [Bugfix] Ensure calculated KV scales are applied in attention. (@adabeyta #27232)\r\n* [Test] Remove old non-varlen FA2 test (@MatthewBonanni #28420)\r\n* [Feature] Refactor batch invariant fp8 DeepGEMM (@yewentao256 #27606)\r\n* [CI/Test Fix] Fix CP tests on Blackwell (@LucasWilkinson #28404)\r\n* [Feature] Add env var `VLLM_MOE_USE_DEEP_GEMM` (@yewentao256 #28422)\r\n* Only register rocm_aiter_ops if aiter is found (@mgoin #28428)\r\n* Fix rotary embedding benchmark script (@xyang16 #28323)\r\n* [Misc] FlattenLogprobs -> FlatLogprobs (@zhuohan123 #28335)\r\n* [Frontend] Add sagemaker_standards dynamic lora adapter and stateful session management decorators to vLLM OpenAI API server (@zhaozuy #27892)\r\n* [Bugfix] Fix Stream Sync for Shared Expert Overlap (@robertgshaw2-redhat #28430)\r\n* [Doc] Sleep mode documentation  (@iAmir97 #28357)\r\n* [BugFix] Avoid calling KV connector layer APIs when metadata is unset (@sdavidbd #28253)\r\n* [Bugfix] Fix max image size for PaddleOCR-VL (@ywang96 #28442)\r\n* [EPLB] Refactor balance_packing to use numpy and optimize GPU-CPU transfers in EPLB (@SageMoore #28369)\r\n* [Bugfix] fix qwen3-next crash (@ZJY0516 #28202)\r\n* [BugFix] 'DeepseekV2Config' object has no attribute 'use_mla'`  (@faaany #28387)\r\n* [Model][Qwen3VL] Slighly speedup `fast_pos_embed_interpolate` (@lgeiger #28434)\r\n* Multi turn benchmark progress bar for synthetic conversation generation (@segevido #28394)\r\n* [CI] Add mergify rules for `nvidia` label (@mgoin #28417)\r\n* [Attention] Refactor CUDA attention backend selection logic (@MatthewBonanni #24794)\r\n* Fix Fused MoE LoRA Triton kernel bug (@chaojun-zhang #28450)\r\n* [Model] Pass `mm_features` directly into `get_mrope_input_positions` (@DarkLight1337 #28399)\r\n* Add request timeout override for multi-turn benchmarks (@segevido #28386)\r\n* [Docs] Fix grammar in CPU installation guide (@maryamtahhan #28461)\r\n* [Kernels] Split up fused_moe/layer.py, isolate more modular kernel code (@bnellnm #28064)\r\n* [BugFix] Fix Failing Ruff Check (@jvlunteren #28469)\r\n* Add @markmc to CODEOWNERS for Observability (@markmc #28457)\r\n* [BugFix] Fix RuntimeError in PixtralHFAttention on CPU/XPU (@faaany #28444)\r\n* [BugFix] Add test_outputs.py to CI pipeline (@usberkeley #28466)\r\n* [Doc] Fix typo in serving docs (@the-codeboy #28474)\r\n* Remove weight_scale.T special case for SM90 Block FP8 CUTLASS kernel (@mgoin #28431)\r\n* [NIXL] Generalize block-first backend layouts (FlashInfer-like) (@NickLucche #28282)\r\n* [Kernel][Perf] fuse QK Norm and RoPE into one cuda kernel for Qwen Model (@izhuhaoran #27165)\r\n* [ROCm][Quantization] extend AMD Quark to support mixed-precision quantized model (@xuebwang-amd #24239)\r\n* [Quantization] fix attention quantization of gpt_oss model (@xuebwang-amd #27334)\r\n* [CI/Build] Refactor Attention backend for test_prefix_prefill from xformers to SDPA (@zhewenl #28424)\r\n* Prefer FlashAttention MLA as default over FlashMLA (@MatthewBonanni #27363)\r\n* [Kernel] Optimize rms_norm kernel (@xyang16 #27931)\r\n* [BugFix] Fix Siglip2Attention on XPU (@faaany #28448)\r\n* [Misc] Remove unused attention prefix prefill ops functions (@lgeiger #26971)\r\n* [Perf] Use np.ndarray instead of list[list[int]] to reduce GC overhead (@Jialin #28245)\r\n* [V0 deprecation] Clean up num_prefill_tokens logic for V0 (@gcanlin #28203)\r\n* [Misc] fix typo in DCP comment (@Livinfly #28389)\r\n* [LoRA][1/N]Remove LoRA extra vocab (@jeejeelee #28382)\r\n* [TPU] Rename path to tpu platform (@kyuyeunk #28452)\r\n* [Misc] Cleanup Executor interface (@wangxiyuan #28441)\r\n* Add Zurich vLLM Meetup (@mgoin #28488)\r\n* [Bugfix] Disable shared expert overlap if Marlin MoE is used (@mgoin #28410)\r\n* [Feature] Allow configuring FlashInfer workspace size (@maxyanghu #28269)\r\n* Use FLASHINFER MLA backend when testing fp8_kv_scale_compile (@adabeyta #28491)\r\n* [BugFix] Graceful handling of torch symm mem errors. (@ilmarkov #27671)\r\n* [Frontend] Change CompilationMode to a proper Enum (@gmagogsfm #28165)\r\n* [Performance] Cache loaded custom logitsprocs to avoid overheads (@Isotr0py #28462)\r\n* [[V0 deprecation]]Remove VLLM_USE_V1 env (@wangxiyuan #28204)\r\n* [CPU] Refactor CPU attention backend (@bigPYJ1151 #27954)\r\n* `VLLM_USE_TRITON_FLASH_ATTN` V0 variable deprecation (@AndreasKaratzas #27611)\r\n* [Model][Qwen3VL] Simplify `get_mrope_input_positions` using numpy (@lgeiger #28302)\r\n* [Core] Encoder separation for Encode-Prefill-Decode Disaggregation (@fake0fan #25233)\r\n* [BugFix] Add fallback path in `apply_rotary_pos_emb_flashattn` for non-cuda platforms (@faaany #28447)\r\n* [Benchmark] Add retry support to fix workload bias in multi-turn benchmark (@ai-jz #28493)\r\n* [Core] Cache `vllm_is_batch_invariant` (@lgeiger #28304)\r\n* [CI/Build] Fix crash due to removed VLLM_USE_V1 attribute in EPD (@fake0fan #28521)\r\n* [CI] Introduce autorun_on_main feature (@hl475 #27836)\r\n* [BugFix]: --enable-lora with model granite-4.0-micro crash (@yyzxw #27733)\r\n* [Model] fix glm4_moe_mtp load weights with GLM-4.6 checkpoint. (@wuyaoxuehun #27597)\r\n* [XPU]Fix crash due to removed VLLM_USE_V1 attribute (@chaojun-zhang #28520)\r\n* [KVConnector] Enable get_block_ids_with_load_errors() in LMCache connector  (@ziruiliu #27978)\r\n* add cpu option for p/d in nixl_connector (@ZhengHongming888 #28356)\r\n* [ROCm] [Bugfix] Fix `fused_qknorm_rope_kernel` rocm compatibility (@tjtanaa #28500)\r\n* [Bugfix] Fix gpt_oss packed_modules_mapping (@jeejeelee #28536)\r\n* [V0 deprecation] Deprecate use_v1 parameter (@wangxiyuan #28112)\r\n* Fix pre-commit (and XPU) on `main` (@hmellor #28556)\r\n* [Performance][Hopper] Avoid M dim padding to 4x for most cases (due to cuda graphs paddings) (@alexm-redhat #28492)\r\n* [Refactor] Remove redundant TP gather/split in split_qkv in QwenVL (@gcanlin #28271)\r\n* [Misc] Refactor Attention kv transfer methods into decorator (@NickLucche #27816)\r\n* Remove deprecated fields from `CompilationConfig` (@hmellor #27593)\r\n* [Perf] Refactor cudagraph_support to enable full CUDA graphs for spec decoding with FlashInfer (@benchislett #28479)\r\n* Implement ARC KV cache eviction policy (@albertoperdomo2 #27039)\r\n* [EPLB][ROCm]: support EPBL for ROCm backend (@PerryZhang01 #27731)\r\n* [Model] [Config] Correctly identify granite-4.0-micro as non-hybrid model (@tdoublep #28563)\r\n* [CI] Skip \"Multi-Modal Models Test (Extended) 3\" test that's broken in current Transformers (@hmellor #28559)\r\n* [KV connector][WIP] KV cache proxy based on LMCache multi-process mode (@ApostaC #27902)\r\n* [BugFix] Priority scheduling and spec tokens preemption (@andylolu2 #28558)\r\n* [Misc]Fix typo in llm_engine.py (@frank-wei #28584)\r\n* [Performance][B200] Fix deepgemm prologue (@varun-sundar-rabindranath #27897)\r\n* [ROCM] Fix ROCm warnings, environment flag access, and GEMM kernel naming for consistency in `_aiter_ops.py` (@vllmellm #28464)\r\n* [TPU] Support GCS path in VLLM_TORCH_PROFILER_DIR (@QiliangCui #28487)\r\n* [Bugfix] Adjust Marlin CUDA arch selection to 8.0+PTX;9.0+PTX (@mgoin #28294)\r\n* [Core][AMD] Migrate fully transparent sleep mode to ROCm platform (@HollowMan6 #12695)\r\n* [MoE][Kernel][Perf] Improve Shared Expert Stream Overlap (@alexm-redhat #28406)\r\n* Skip models that cannot currently init on Transformers v5 (@hmellor #28471)\r\n* [Docs] Update meetups.md description (@mgoin #28583)\r\n* [ROCm][Bugfix] Revert removing setuptools version restriction (@gshtras #28592)\r\n* [platform] Move get_cu_count to utils (@wangxiyuan #27005)\r\n* [Bugfix] Fix SM100 gpt-oss regression due to faulty attn sink support (@mgoin #28561)\r\n* [BugFix] Fix `mm_encoder_attn_backend` arg type checking (@njhill #28599)\r\n* [Docs] Add some details about what the MoE block needs for the Transformers backend (@hmellor #28588)\r\n* Rename clashing method names for vLLM model protocol (@hmellor #27583)\r\n* [n-gen] DO NOT repeatedly return finished child requests (@Jialin #28591)\r\n* [Frontend] split append tool output (@qandrew #28333)\r\n* [Frontend][responsesAPI][1/n] convert responses API tool input to chat completions tool format (@qandrew #28231)\r\n* [BugFix][ROCm] Fix `get_cu_count` missing variable error (@ganyi1996ppo #28608)\r\n* [XPU] Support Triton path for LoRA operations on XPU   (@faaany #28511)\r\n* Support DeepEP for Kimi-k2-thinking through enabling gemm selection for compressed-tensor marlin wna16 (@luccafong #28574)\r\n* [build][cmake]: Bundle static ACL and torch libgomp for CPU extension builds (@Radu2k #28059)\r\n* [ROCm][BugFix] Remove the usage of `device_info` from aiter (@ganyi1996ppo #28383)\r\n* [Bugfix] Prevent crash on empty grammar string (@tjandy98 #28210)\r\n* Use official xformers-0.0.33 built for PT 2.9 (@huydhn #28600)\r\n* Add NUMA node validation for CPU thread binding (@usberkeley #28555)\r\n* [Bugfix] fix kimi-linear crash (@ZJY0516 #28445)\r\n* [Frontend] supports interleaved thinking (@chaunceyjiang #28531)\r\n* Support all interleaved layer types (@sarckk #28485)\r\n* Fix: Correctly filter special tokens in benchmark_prefix_caching (@dw2761 #28615)\r\n* [BugFix] Fix type error when assign a trition kernel tensor to a torch.nn.Parameter (@liuzijing2014 #28603)\r\n* Fix io processor pooling  #28273 (@baonudesifeizhai #28484)\r\n* [XPU] add sym params to IPEXConfig (@zufangzhu #28611)\r\n* [Bugfix] Fix FPS value type for Qwen2.5-Omni video processing (@faaany #28630)\r\n* [Hardware][PowerPC] Fix fp16 compilation error for Power in cpu attention backend and bump oneDNN version (@Akashcodes732 #28535)\r\n* [ROCm][BugFix]Fix `get_cu_count` in rocm_aiter_fa.py (@ganyi1996ppo #28618)\r\n* [CI/Build] Install uv for AMD MI300: Language Models Tests (Hybrid) %N (@amdfaa #28142)\r\n* [CI Failure] Fix backend selection for encoder-only models (@hl475 #28534)\r\n* [BugFix] DeepSeek-OCR: apply NoRepeatNGramLogitsProcessor to greedy path (@YuanpingSong #28617)\r\n* Fix `get_num_experts` when config sets it explicitly to `None` (@hmellor #28652)\r\n* [Misc] Turn off encoder torch compile by default (@ywang96 #28634)\r\n* Rewrite C++ meta funcs to Python (@janeyx99 #28595)\r\n* [BugFix] Ensure `EngineArgs.create_engine_config` is idempotent (@njhill #28515)\r\n* [TPU] patch TPU wheel build script to resolve metadata issue (@jcyang43 #27279)\r\n* [Performance][B200] silu_mul_quant: pack scales in int32 (@varun-sundar-rabindranath #28358)\r\n* [Bugfix] Fix validate model input for decoder models (@yannicks1 #27099)\r\n* [Attention][Bugfix] Fix FA sink support (@MatthewBonanni #28660)\r\n* [Perf] Support stream interval for reducing host overhead (@elvischenv #27869)\r\n* [bugfix] correct local_chunk_len for DCP in reorg_kvcache with long context (@pisceskkk #28526)\r\n* [Bugfix] Eliminate tuple inputs to submodules in graph partitioning (@gmagogsfm #28533)\r\n* [Bugfix] [CPU] bump torch to 2.9.0 for Darwin to fix segmentation fault (@kebe7jun #27791)\r\n* [Misc] Update CODEOWNERS for simon-mo and comaniac (@simon-mo #28675)\r\n* [CI] Bug: Fix ci entrypoint pooling (@yewentao256 #28684)\r\n* [KV Connector] Test async mode in scheduler tests (@markmc #28550)\r\n* Mirrored test group definitions for AMD (2025-11-11) (@Alexei-V-Ivanov-AMD #28573)\r\n* [quantization][config] enable override existing quant_config (@ILikeIneine #28510)\r\n* [ROCm] Bump up the version of amd-smi to 6.4.3 (@SageMoore #28680)\r\n* [CPU][Bugfix] Fix Apple Silicon M1 compilation failure (@mgoin #28681)\r\n* [ci][amd] fix basic models extra init test (@bradleyhd #28676)\r\n* [Misc] Remove `warn_for_unimplemented_methods` (@DarkLight1337 #28613)\r\n* [XPU][CI]disable lm cache uts (@jikunshang #28696)\r\n* [Misc] Update xformers to 0.33.0.post1 (@ywang96 #28678)\r\n* [Misc] add ignore mapper for quark quantization (@haoyangli-amd #28275)\r\n* [Bugfix][CI/Test][Spec Decode] Fix illegal memory access in offline_inference/spec_decode.py (Issue  27619) (@rasmith #28432)\r\n* [BugFix][CI/Build][ROCM] Fix import error and apply assert in appropriate case in test_struct_output_generate (@rasmith #28311)\r\n* use default CCL_ZE_IPC_EXCHANGE (@yma11 #28700)\r\n* [Bugfix] fix dots.ocr pp support (@ZJY0516 #28705)\r\n* [BugFix] Fix multi-modal async scheduling race condition (@njhill #28706)\r\n* Add output token counting to gsm8k eval (@mgoin #28594)\r\n* [Minor] avoid register new custom and just import silly_attn (@BoyuanFeng #28578)\r\n* [Misc] fix comment in test_envs (@xingliu14 #28529)\r\n* [feat]: log number of preempted requests (@610lyn #28522)\r\n* [Frontend] Added chat-style multimodal support to /classify. (@WorldExplored #27516)\r\n* [Model][MM] Extract conv layer as CustomOp (@shen-shanshan #28455)\r\n* [DCP] Support Decode Context Parallel (DCP) for GQA with Flashinfer (@gjc0824 #25438)\r\n* Fix KV sharing fast prefill with cudagraph enabled (@sarckk #28537)\r\n* [BugFix] Fix FA3 IMA with FULL_AND_PIECEWISE and cascade attention (default) (@LucasWilkinson #28702)\r\n* [Doc] Fix macOS installation dependency resolution issue (@shahfasal #26721)\r\n* [Model] Fix bailing_moe accuracy problem (@zhaozx-cn #28277)\r\n* [Bugfix][Nixl] Fix kernel physical<>logical block_size issue  (@NickLucche #28677)\r\n* [Config] Clean up SchedulerConfig initialization (@DarkLight1337 #28665)\r\n* [Kernels] Enable FlashInfer FP8 Blockscale on SM90 (for TEP DSR1) (@djmmoss #27134)\r\n* [Fix] improve aspect ratio in dummy image generation and add common  VLM tests for PaddleOCR-VL (@dongbo910220 #28711)\r\n* [Docs] Update the name of `Transformers backend` -> `Transformers modeling backend` (@hmellor #28725)\r\n* [CI][CPU] Smoke test for Apple Silicon using GHA MacOS runner (@mgoin #28688)\r\n* [DisaggEverything] Tokens in<>out `/generate` endpoint (@NickLucche #24261)\r\n* [Attention] Bump FA for removed method (@MatthewBonanni #28429)\r\n* Fix typo in comment: existance -> existence (@OthmanMohammad #28737)\r\n* Remove audio optional dependency for mistral-common (@juliendenize #28722)\r\n* [kernel] Improve FP8 PTPC on Hopper for larger shapes (@czhu-cohere #28692)\r\n* docs(lora_resolvers): clarify multi-resolver order and storage path requirement (@wangchen615 #28153)\r\n* LLaMA4 LoRA Adapter Enablement (@kfhfar #28602)\r\n* [Bugfix] [ROCm] [AITER]: Fix aiter block quant not compatible with torch compile dynamo (@tjtanaa #28716)\r\n* [Docs] Enable some more markdown lint rules for the docs (@hmellor #28731)\r\n* [Chore] Rename `SchedulerConfig.chunked_prefill_enabled` (@DarkLight1337 #28735)\r\n* [Bugfix] resolve Qwen3-VL GPTQModel quantized model loading failure (@GuanH #28663)\r\n* [BugFix] Fix misprint introduced by modular_kernel refactoring. (@halyavin #28728)\r\n* [ROCm][Bugfix] Fix compilation errors with fused_qknorm_rope_kernel.cu (@SageMoore #28682)\r\n* [CI] Fix macos smoke test uv cache issue (@mgoin #28736)\r\n* [Bugfix] TypeError: 'NoneType' object is not callable (@mostrowskix #27410)\r\n* [ROCm][CI/Build] Change install location of uv (@gshtras #28741)\r\n* Avoid bytecode hook and simplify TorchCompileWrapperWithCustomDipatch (@laithsakka #25110)\r\n* [Bugfix] Fix incorrect use of hidden_states for shared_experts due to do_naive_dispatch_combine (@alexm-redhat #28740)\r\n* [Bugfix] Fix ChunkedLocalAttention CUDA Graph setting (@benchislett #28739)\r\n* [Hybrid] [Kernel] Fix chunk scan kernel when BLOCK_SIZE_DSTATE > 128 (@tdoublep #28295)\r\n* [Log] Save profiler results to file instead of stdout (@rasmith #28144)\r\n* [ROCm][CI/Build] Upgrade to ROCm 7.1 and AITER main (@gshtras #28753)\r\n* [Test] Rework e2e async scheduling tests (@njhill #28744)\r\n* [Core] Performance: Use list[np.ndarray] instead of list[list[int]] for output tokens for GC optimization (@Jialin #26368)\r\n* [TPU] Fix import error in tpu launch (@QiliangCui #28758)\r\n* [Model][Qwen3VL] Use `mm_position` to compute mrope positions (@lgeiger #28730)\r\n* [Bugfix] Build hadacore kernels on >SM90 (@mgoin #28748)\r\n* Revert \"[Core] Performance: Use list[np.ndarray] instead of list[list… (@njhill #28773)\r\n* Fix IntermediateTensors initialization and add type hints (@OthmanMohammad #28743)\r\n* [NIXL] heterogeneous block_size support (@xuechendi #26759)\r\n* [Performance][DeepGEMM] Estimate expected_m (@varun-sundar-rabindranath #28694)\r\n* [Redo] #26368 (@DarkLight1337 #28771)\r\n* [RL] [V1] Remove unused device argument from reset_kv_cache (@zhuohan123 #28766)\r\n* Use narrow over indexing in `hadacore_transform` to prep for ABI stable (@janeyx99 #28756)\r\n* [Kernel][Moe Configs] llama4 maverick fp8 moe config tp8 on mi325 (@zhewenl #28709)\r\n* [Misc] Make `SchedulerConfig.max_model_len` init-only (@DarkLight1337 #28733)\r\n* [PERF] Remove TRTLLM Gen attn kernel limitation `max_seq_len <=131072` (@vadiklyutiy #28755)\r\n* [compile] Enable sequence parallelism matching w/o custom ops enabled  (@angelayi #27126)\r\n* Allow Gemma3 to take image embeddings (@tingtingtangmeta #28483)\r\n* [Doc] Fix failing doc build (@DarkLight1337 #28772)\r\n* [Model] Fix lmhead init bug of bailing_moe (@hwhaokun #28777)\r\n* Add support for Eagle with separate lm-head and embed_tokens layers (@eldarkurtic #28549)\r\n* [CI] Fix broken pipeline (@njhill #28781)\r\n* [Model][Qwen3VL] Cache positional embedding indices  (@lgeiger #28475)\r\n* [Doc]: fix typos in various files (@didier-durand #28567)\r\n* [BugFix] Fix `AssertionError: DCP not support reorder_batch_threshold > 1 now.`  (@LucasWilkinson #28751)\r\n* Adding a benchmark for batch invariance (@bwasti #28161)\r\n* [Benchmark] Fix client seed synchronization in multi-turn benchmark (@ai-jz #28512)\r\n* [Model] Allow users to control skip reading cache per request. (@noooop #28194)\r\n* [V1] Support MP Executor for multi node distributed inference (@luccafong #23691)\r\n* Fixed gpt-oss _load_weights_other() parameter position bug (@River12 #28715)\r\n* [Bugfix] Fix host and port join for ipv6 in bench serve (@scottzh8 #28679)\r\n* Fix gpt oss weight loading with EP + bf16 (@ashors1 #28765)\r\n* [Doc]: fix typos in various files (@didier-durand #28811)\r\n* fix comment typo (@andyxning #28802)\r\n* [Model][QwenVL] Optimize `Qwen2_5_VisionAttention` q,k preparation (@lgeiger #28769)\r\n* Feature: Support Relu2 in FusedMoE fp8 cutlass path (@amirkl94 #27261)\r\n* [BugFix] Fix async scheduling + chunked prefill + preemption (@njhill #28787)\r\n* [Performance][Fix] update nvfp4 code to support renorm routing (@jiahanc #28569)\r\n* [NIXL][XPU] update install script of NIXL (@zhenwei-intel #28778)\r\n* [ROCm][Qwen3-32B] Fix AITER MHA accuracy issue cause by #25763 (@sammysun0711 #28670)\r\n* [Bugfix][Model] Prevent special token leakage in KimiK2ToolParser streaming mode (@jscaldwell55 #28543)\r\n* [Doc] Add llama4 LoRA tag (@jeejeelee #28825)\r\n* [CPU][Bugfix] Fix _to_list in CPU model runner (@bigPYJ1151 #28824)\r\n* [BugFix] Fix glm4_moe_mtp load weights bug (@wuyaoxuehun #28805)\r\n* [Metrics] Fix KV cache usage percent metric multiproc (@jaywonchung #28792)\r\n* [XPU] work around for sp, avoid custom op import error (@jikunshang #28822)\r\n* [BugFix] Temporary fix for IMA with MTP = 2 and full-cg (@LucasWilkinson #28315)\r\n* [Bugfix][Perf] Revert applying HF processor on text-only inputs for multimodal models  (@ywang96 #28858)\r\n* Cast return value to int64_t for cache size (@tiehexue #28814)\r\n* [Bugfix] Fix GPT-OSS on AMD after #28603 (@zhewenl #28816)\r\n* [Core] Async Scheduling X Spec Decoding Compatibility (@Ronald1995 #24799)\r\n* [BugFix] Fix PP performance and PP kv connector output regression  (@njhill #28768)\r\n* [Quantization] [Eagle] Add complete quantization support to the draft model in Eagle (@shreyas269 #28435)\r\n* [Test] Batch Invariant: Rename and organize tests (@yewentao256 #27421)\r\n* [Model] Add Afmoe architecture implementation (@pranav4501 #28332)\r\n* [BugFix] Corner case that could cause out-of-sync with external launcher mode and dp >1 (@bangshengtang #28774)\r\n* [Misc] Fix wrong comment in scheduler (@zhuohan123 #28880)\r\n* [Bugfix] Fix Kimi-K2 tool parser concatenated tool calls parsing (@bbartels #28831)\r\n* Run macos smoke test workflow on main commit (@mgoin #28752)\r\n* [ROCm][Quantization] add apply_vllm_mapper in quark config for models like gpt-oss (@xuebwang-amd #28638)\r\n* [Refactor] Remove Unused Func in Batch Invariant (@yewentao256 #28881)\r\n* [Bugfix] Fix wrong CLI defaults for dynamic `SchedulerConfig` fields (@DarkLight1337 #28872)\r\n* [Doc]: fix typos in various files (@didier-durand #28863)\r\n* [Misc] Remove unnecessary parentheses from log statements (@andyxning #28897)\r\n* [CI] Fix async scheduling + spec decoding test flake (@njhill #28902)\r\n* [MISC] Remove format.sh (@KuntaiDu #28906)\r\n* [CI/Build] Replace wikipedia url with local server ones (@Isotr0py #28908)\r\n* [BugFix] Fix PP/async scheduling with pooling models (@njhill #28899)\r\n\r\n## New Contributors\r\n* @bwasti first commit is #25603\r\n* @Renovamen first commit is #25796\r\n* @patrick-toulme first commit is #25084\r\n* @kingsmad first commit is #25825\r\n* @yingjun-mou first commit is #25827\r\n* @zhoukezi first commit is #25854\r\n* @leejnau first commit is #25706\r\n* @adabeyta first commit is #25513\r\n* @acisseJZhong first commit is #25912\r\n* @a120092009 first commit is #25942\r\n* @Anionex first commit is #25354\r\n* @DrStone1971 first commit is #25843\r\n* @certainly-param first commit is #25935\r\n* @natoscott first commit is #26007\r\n* @kmaehashi first commit is #26005\r\n* @leo-pony first commit is #25470\r\n* @huijjj first commit is #24947\r\n* @levunet first commit is #24768\r\n* @Egor-Krivov first commit is #25668\r\n* @sixiang-google first commit is #25992\r\n* @astralord first commit is #26027\r\n* @jasl first commit is #26098\r\n* @nrghosh first commit is #26148\r\n* @southfreebird first commit is #25974\r\n* @soldni first commit is #26054\r\n* @yuafng first commit is #26219\r\n* @ILikeIneine first commit is #25823\r\n* @jasonlizhengjian first commit is #25998\r\n* @elieserr first commit is #26177\r\n* @orangeng first commit is #26266\r\n* @ymoslem first commit is #26258\r\n* @abhisheksheth28 first commit is #25521\r\n* @seven-mile first commit is #26231\r\n* @cfRod first commit is #26289\r\n* @atalhens first commit is #26265\r\n* @gholmes829 first commit is #25164\r\n* @dcampora first commit is #25945\r\n* @antrec first commit is #26340\r\n* @plliao first commit is #26325\r\n* @morrison-turnansky first commit is #26113\r\n* @isharif168 first commit is #26347\r\n* @Barry-Delaney first commit is #25931\r\n* @utkarshsharma1 first commit is #26279\r\n* @Aydin-ab first commit is #25283\r\n* @therealnaveenkamal first commit is #25103\r\n* @QierLi first commit is #24926\r\n* @zhiyuan1i first commit is #24486\r\n* @iwzbi first commit is #16601\r\n* @roikoren755 first commit is #25947\r\n* @luis5tb first commit is #25593\r\n* @wangxiongts first commit is #25550\r\n* @sangho-vision first commit is #26563\r\n* @muzian666 first commit is #26562\r\n* @HsChen-sys first commit is #22100\r\n* @FENP first commit is #26574\r\n* @gjgjos first commit is #26339\r\n* @andycandy first commit is #26629\r\n* @aitsvet first commit is #26713\r\n* @cyb70289 first commit is #26698\r\n* @kfhfar first commit is #26538\r\n* @n1ck-guo first commit is #24024\r\n* @ryanli first commit is #26758\r\n* @VladOS95-cyber first commit is #26726\r\n* @zklapow first commit is #26818\r\n* @HDCharles first commit is #26820\r\n* @Dhruvilbhatt first commit is #26837\r\n* @madongfly first commit is #26853\r\n* @li2haipeng first commit is #26319\r\n* @pdasigi first commit is #26143\r\n* @cern1710 first commit is #26637\r\n* @inc-jeong first commit is #26225\r\n* @bogdanminko first commit is #27008\r\n* @mandy-li first commit is #26883\r\n* @kimbochen first commit is #26943\r\n* @staghado first commit is #26916\r\n* @rkarhila-amd first commit is #25586\r\n* @hyongtao-code first commit is #27101\r\n* @jianyuh first commit is #27159\r\n* @uyzhang first commit is #27012\r\n* @shivampr first commit is #26268\r\n* @helunwencser first commit is #26832\r\n* @dagrayvid first commit is #27196\r\n* @ExtReMLapin first commit is #27253\r\n* @ReinForce-II first commit is #26789\r\n* @LiuLi1998 first commit is #22627\r\n* @sagiahrac first commit is #27211\r\n* @fangpings first commit is #27133\r\n* @jonathanc-n first commit is #27372\r\n* @bradleyhd first commit is #27124\r\n* @Navya1707 first commit is #27156\r\n* @piood first commit is #27324\r\n* @xxxxyu first commit is #26092\r\n* @usberkeley first commit is #27419\r\n* @strinczer first commit is #26706\r\n* @hjh0119 first commit is #27469\r\n* @wpc first commit is #27328\r\n* @yeshsurya first commit is #27188\r\n* @rogeryoungh first commit is #27535\r\n* @dcmaddix first commit is #27291\r\n* @tingtingtangmeta first commit is #27538\r\n* @minatoaquaMK2 first commit is #27323\r\n* @wangln19 first commit is #27565\r\n* @junpuf first commit is #27596\r\n* @sammshen first commit is #27600\r\n* @mpashkovskii first commit is #26886\r\n* @KevinCheung2259 first commit is #27670\r\n* @sammysun0711 first commit is #27623\r\n* @dumb0002 first commit is #24176\r\n* @sairampillai first commit is #25775\r\n* @FlamingoPg first commit is #27794\r\n* @SumanthRH first commit is #27789\r\n* @PaulZhang12 first commit is #27660\r\n* @jakub-sochacki first commit is #26919\r\n* @RobMulla first commit is #27824\r\n* @yugong333 first commit is #27818\r\n* @ai-jz first commit is #27850\r\n* @xiaohajiayou first commit is #26779\r\n* @biswapanda first commit is #27728\r\n* @efimki first commit is #24905\r\n* @zhang-prog first commit is #27758\r\n* @xiangze-arm first commit is #27240\r\n* @yt0428 first commit is #27521\r\n* @ganyi1996ppo first commit is #25763\r\n* @nadavkluger first commit is #28048\r\n* @toulzx first commit is #27740\r\n* @frost-intel first commit is #28004\r\n* @jjzhang first commit is #28127\r\n* @walterbm first commit is #28075\r\n* @dayeol first commit is #22496\r\n* @cmpute first commit is #27780\r\n* @seungduk-yanolja first commit is #27946\r\n* @aditew01 first commit is #28130\r\n* @milpuz01 first commit is #26018\r\n* @StanHatko first commit is #27953\r\n* @vicoooo26 first commit is #27792\r\n* @HanFa first commit is #27497\r\n* @amacaskill first commit is #28079\r\n* @smitkadvani first commit is #28024\r\n* @xiaohongchen1991 first commit is #21068\r\n* @hammmmy first commit is #28308\r\n* @ashahba first commit is #28026\r\n* @zhangsicheng5 first commit is #26696\r\n* @evberrypi first commit is #28328\r\n* @ColeMurray first commit is #28337\r\n* @bo-ke first commit is #28374\r\n* @caozuoba first commit is #28280\r\n* @zhaozuy first commit is #27892\r\n* @maryamtahhan first commit is #28461\r\n* @the-codeboy first commit is #28474\r\n* @xuebwang-amd first commit is #24239\r\n* @Livinfly first commit is #28389\r\n* @AndreasKaratzas first commit is #27611\r\n* @wuyaoxuehun first commit is #27597\r\n* @ziruiliu first commit is #27978\r\n* @ZhengHongming888 first commit is #28356\r\n* @albertoperdomo2 first commit is #27039\r\n* @PerryZhang01 first commit is #27731\r\n* @Radu2k first commit is #28059\r\n* @tjandy98 first commit is #28210\r\n* @dw2761 first commit is #28615\r\n* @zufangzhu first commit is #28611\r\n* @amdfaa first commit is #28142\r\n* @YuanpingSong first commit is #28617\r\n* @janeyx99 first commit is #28595\r\n* @xingliu14 first commit is #28529\r\n* @610lyn first commit is #28522\r\n* @WorldExplored first commit is #27516\r\n* @gjc0824 first commit is #25438\r\n* @shahfasal first commit is #26721\r\n* @zhaozx-cn first commit is #28277\r\n* @OthmanMohammad first commit is #28737\r\n* @GuanH first commit is #28663\r\n* @halyavin first commit is #28728\r\n* @mostrowskix first commit is #27410\r\n* @laithsakka first commit is #25110\r\n* @hwhaokun first commit is #28777\r\n* @River12 first commit is #28715\r\n* @scottzh8 first commit is #28679\r\n* @ashors1 first commit is #28765\r\n* @jscaldwell55 first commit is #28543\r\n* @tiehexue first commit is #28814\r\n* @Ronald1995 first commit is #24799\r\n* @shreyas269 first commit is #28435\r\n* @pranav4501 first commit is #28332\r\n\r\n**Full Changelog**: https://github.com/vllm-project/vllm/compare/v0.11.0...v0.11.1","reactions":{"url":"https://api.github.com/repos/vllm-project/vllm/releases/263447752/reactions","total_count":130,"+1":6,"-1":0,"laugh":2,"hooray":32,"confused":0,"heart":50,"rocket":6,"eyes":34},"mentions_count":200}]