41554 blogs · [ { "id": "01a0d4b3-2f68-735a-b0c2-ef4c6cbf04ee", "title": "Watermarking in vLLM", "url": "https://vllm.ai/blog/2026-09-24-watermarking-in-vllm", "published_at": "2026-09-24T00:00:00+00:00" }, { "id": "01a0d4b3-2f68-735a-b0c2-ef4c6d55b5df", "title": "Announcing vllm-metal: Concurrent Serving on Apple Silicon", "url": "https://vllm.ai/blog/2026-09-22-vllm-metal-v0-28-0", "published_at": "2026-09-22T00:00:00+00:00" }, { "id": "01a0d4b3-2f68-735a-b0c2-ef4c6dd72919", "title": "PD Serving of Qwen3.8-2.4T", "url": "https://vllm.ai/blog/2026-09-21-qwen38-pd-serving", "published_at": "2026-09-21T00:00:00+00:00" }, { "id": "01a0d4b3-2f68-735a-b0c2-ef4c6e3d960b", "title": "Scaling Multi-GPU Video Captioning with PyNvVideoCodec and vLLM", "url": "https://vllm.ai/blog/2026-09-18-pynvvideocodec", "published_at": "2026-09-18T00:00:00+00:00" }, { "id": "01a0d4b3-2f68-735a-b0c2-ef4c6ef8d243", "title": "How we trained the fastest DSpark for Kimi-K3 using GB300 NVL72", "url": "https://vllm.ai/blog/2026-09-15-kimi-k3-dspark", "published_at": "2026-09-15T00:00:00+00:00" }, { "id": "01a0d4b3-2f68-735a-b0c2-ef4c6ff7822f", "title": "vLLM x Novita AI: Chord, Faster INT4 MoE for Kimi K2.x. Up to 1.3x on H200, 2.15x on Untuned B300", "url": "https://vllm.ai/blog/2026-09-15-novita-chord-w4a16-moe", "published_at": "2026-09-15T00:00:00+00:00" }, { "id": "01a0d4b3-2f68-735a-b0c2-ef4c7045a6d7", "title": "vime × RL-Kernel × AMD: Bitwise Train–Rollout Consistency on ROCm", "url": "https://vllm.ai/blog/2026-09-14-rl-kernel-v0-1-0-train-rollout-bitwise-consistency", "published_at": "2026-09-14T00:00:00+00:00" }, { "id": "01a0d4b3-2f68-735a-b0c2-ef4c70825be4", "title": "Kimi K3 Performance Optimizations in vLLM: The Road to 2.8× Throughput", "url": "https://vllm.ai/blog/2026-09-13-kimi-k3-performance-optimization", "published_at": "2026-09-13T00:00:00+00:00" }, { "id": "01a08c5f-2bd4-717e-8ed1-14c4e3a61095", "title": "Following the Bottleneck: Optimizing MiniMax M3 on AMD Instinct MI355X", "url": "https://vllm.ai/blog/2026-09-10-minimax-m3-mi355x", "published_at": "2026-09-10T00:00:00+00:00" }, { "id": "01a08cfd-841f-715c-b6ab-a5a40dc3d483", "title": "Tiered KV Cache Offloading in vLLM", "url": "https://vllm.ai/blog/2026-09-10-tiered-kv-offloading", "published_at": "2026-09-10T00:00:00+00:00" }, { "id": "01a0823d-891e-70d0-b8c7-ee6278197388", "title": "GLM 5.3 Optimizations, Part 1: Hybrid HiSparse Offloading in vLLM", "url": "https://vllm.ai/blog/2026-09-08-glm53-part1-hybrid-sparse-offloading", "published_at": "2026-09-08T00:00:00+00:00" }, { "id": "01a084cb-127c-73b8-a9bf-557a06f64b6d", "title": "vLLM x AgentX: Optimizing for Real-World Agentic Serving", "url": "https://vllm.ai/blog/2026-09-08-vllm-agentx", "published_at": "2026-09-08T00:00:00+00:00" }, { "id": "01a0823d-891e-70d0-b8c7-ee627918dfc0", "title": "Serving LLMs on Tenstorrent Hardware: Inside the vLLM TT Plugin", "url": "https://vllm.ai/blog/2026-09-07-vllm-tt-plugin", "published_at": "2026-09-07T00:00:00+00:00" }, { "id": "01a0823d-891e-70d0-b8c7-ee62795d238e", "title": "MiniMax H3 on vLLM-Omni: From System-Wide Optimization to Real-Time Serving with FastVideo’s FastH3", "url": "https://vllm.ai/blog/2026-09-01-minimax-h3-production-serving", "published_at": "2026-09-01T00:00:00+00:00" }, { "id": "01a0823d-891e-70d0-b8c7-ee6279cbea40", "title": "Exploring Speculative Decoding in vLLM on AMD GPUs", "url": "https://vllm.ai/blog/2026-08-23-speculative-decoding-amd-gpus", "published_at": "2026-08-23T00:00:00+00:00" }, { "id": "01a0823d-891e-70d0-b8c7-ee627a521a2b", "title": "Large-Scale Sharded Weight Transfer with Ray Direct Transport (RDT) in vLLM", "url": "https://vllm.ai/blog/2026-08-22-rdt-weight-transfer", "published_at": "2026-08-22T00:00:00+00:00" }, { "id": "01a0823d-891e-70d0-b8c7-ee627adab622", "title": "IsoExec: Unified Execution to Eliminate Trainer-Inference Mismatch in SkyRL", "url": "https://vllm.ai/blog/2026-08-21-isoexec", "published_at": "2026-08-21T00:00:00+00:00" }, { "id": "01a0823d-891e-70d0-b8c7-ee627af2f41d", "title": "VeRL-Omni v0.2.0: Faster Diffusion RL and Stable Omni Training", "url": "https://vllm.ai/blog/2026-08-20-verl-omni-v0-2-0", "published_at": "2026-08-20T00:00:00+00:00" }, { "id": "01a0823d-891e-70d0-b8c7-ee627b23b5a8", "title": "Distributed Layerwise Offload: Scaling Toward 200B+ DiT Models Efficiently in vLLM-Omni", "url": "https://vllm.ai/blog/2026-08-17-distributed-layerwise-offload", "published_at": "2026-08-17T00:00:00+00:00" }, { "id": "01a0823d-891e-70d0-b8c7-ee627bb8d5d6", "title": "Adaptive Verification in vLLM: DSpark confidence-scheduled verification", "url": "https://vllm.ai/blog/2026-08-14-dspark-adaptive-verification", "published_at": "2026-08-14T00:00:00+00:00" }, { "id": "01a0823d-891e-70d0-b8c7-ee627bc14243", "title": "Day 0 Support for Qwen3.8-2.4T-A95B on vLLM", "url": "https://vllm.ai/blog/2026-08-12-qwen3.8", "published_at": "2026-08-12T00:00:00+00:00" }, { "id": "01a0823d-891e-70d0-b8c7-ee627c7be083", "title": "Announcing Day-0 Support for NVIDIA Nemotron 3.5 Lightning on vLLM", "url": "https://vllm.ai/blog/2026-08-10-nemotron-3-5-lightning-vllm", "published_at": "2026-08-10T00:00:00+00:00" }, { "id": "01a0823d-891e-70d0-b8c7-ee627c9c664a", "title": "Efficient Decode Context Parallelism with vLLM for Long Context Workloads", "url": "https://vllm.ai/blog/2026-08-07-decode-context-parallelism", "published_at": "2026-08-07T00:00:00+00:00" }, { "id": "01a0823d-891e-70d0-b8c7-ee627cd252c7", "title": "vLLM Reaches 25K Total TPS/GPU on Qwen3.5", "url": "https://vllm.ai/blog/2026-08-06-qwen35-25k-tps", "published_at": "2026-08-06T00:00:00+00:00" }, { "id": "01a0823d-891e-70d0-b8c7-ee627d94853a", "title": "Optimizing vLLM on Arm CPUs", "url": "https://vllm.ai/blog/2026-07-29-optimizing-vllm-on-arm-cpus", "published_at": "2026-07-29T00:00:00+00:00" }, { "id": "01a0823d-891e-70d0-b8c7-ee627df6d628", "title": "Parallel All the Way Down: Beyond Single-Token Generation with Speculative Decoding", "url": "https://vllm.ai/blog/2026-07-28-speculators-parallel-drafting", "published_at": "2026-07-28T00:00:00+00:00" }, { "id": "01a0823d-891e-70d0-b8c7-ee627ec0a114", "title": "Kimi K3 Is Here: Efficient Day-0 Support on vLLM", "url": "https://vllm.ai/blog/2026-07-27-k3", "published_at": "2026-07-27T00:00:00+00:00" }, { "id": "01a0823d-891e-70d0-b8c7-ee627f639103", "title": "From Day 0 to Production SLAs: Serving GLM-5.2 on 24 NVIDIA B300 GPUs with vLLM", "url": "https://vllm.ai/blog/2026-07-23-glm-5.2-nvfp4-b300-pd", "published_at": "2026-07-23T00:00:00+00:00" }, { "id": "01a0823d-891e-70d0-b8c7-ee627f7b8d58", "title": "Announcing vLLM AFD Plugin: Disaggregating Attention and FFN for Flexible MoE Serving", "url": "https://vllm.ai/blog/2026-07-23-vllm-afd-plugin", "published_at": "2026-07-23T00:00:00+00:00" }, { "id": "01a0823d-891e-70d0-b8c7-ee627fcd1933", "title": "A Preview of Production-Scale Kimi K3 Support on vLLM", "url": "https://vllm.ai/blog/2026-07-22-kimi-k3-preview", "published_at": "2026-07-22T00:00:00+00:00" }, { "id": "01a0823d-891e-70d0-b8c7-ee62804744f3", "title": "Beyond a Single Model: Building Mixture-of-Models Systems with vLLM Semantic Router", "url": "https://vllm.ai/blog/2026-07-21-vllm-sr-new-chapter-mom", "published_at": "2026-07-21T00:00:00+00:00" }, { "id": "01a0823d-891e-70d0-b8c7-ee6280b0155d", "title": "Keeping vLLM Production Quality: A Look Inside CI, Benchmarking, and the Release Process", "url": "https://vllm.ai/blog/2026-07-16-keeping-vllm-production-quality", "published_at": "2026-07-16T00:00:00+00:00" }, { "id": "01a0823d-891e-70d0-b8c7-ee628157f293", "title": "TML Inkling on vLLM: Day-0 Support with Optimized Performance", "url": "https://vllm.ai/blog/2026-07-15-inkling", "published_at": "2026-07-15T00:00:00+00:00" }, { "id": "01a0823d-891e-70d0-b8c7-ee62822aa4cd", "title": "vLLM x TileRT: Specialized Decode for Latency-Critical Serving", "url": "https://vllm.ai/blog/2026-07-14-vllm-tilert-pd", "published_at": "2026-07-14T00:00:00+00:00" }, { "id": "01a0823d-891e-70d0-b8c7-ee628322fadc", "title": "EAGLE3 Speculative Decoding on AMD Instinct GPUs: Training and Serving with vLLM and AMD Quark", "url": "https://vllm.ai/blog/2026-07-13-eagle-3-amd-instinct", "published_at": "2026-07-13T00:00:00+00:00" }, { "id": "01a0823d-891e-70d0-b8c7-ee62841eb8fc", "title": "vime + ROCm: End-to-End RL Post-Training on AMD Instinct™ GPUs", "url": "https://vllm.ai/blog/2026-07-10-vime-rocm", "published_at": "2026-07-10T00:00:00+00:00" }, { "id": "01a0823d-891e-70d0-b8c7-ee62850ca0a5", "title": "vLLM × HPC-Ops: High-Performance Attention and MoE Backends from Tencent Hunyuan", "url": "https://vllm.ai/blog/2026-07-06-vllm-hpc-ops", "published_at": "2026-07-06T00:00:00+00:00" }, { "id": "01a0823d-891e-70d0-b8c7-ee6285fe97bc", "title": "Experience and Lessons Learned from Serving Multi-Stage Qwen3-Omni in vLLM-Omni", "url": "https://vllm.ai/blog/2026-07-01-qwen3-omni-optimization", "published_at": "2026-07-01T00:00:00+00:00" }, { "id": "01a0823d-891e-70d0-b8c7-ee628687857c", "title": "Micro-Agent: Beat Frontier Models with Collaboration inside Model API", "url": "https://vllm.ai/blog/2026-06-29-micro-agent-frontier-models", "published_at": "2026-06-29T00:00:00+00:00" }, { "id": "01a0823d-891e-70d0-b8c7-ee6287238d0f", "title": "Engineering TTS Inference in vLLM-Omni", "url": "https://vllm.ai/blog/2026-06-23-vllm-omni-tts", "published_at": "2026-06-23T00:00:00+00:00" }, { "id": "01a0823d-891e-70d0-b8c7-ee628796b1ca", "title": "Beyond One Model: Fusion in vLLM Semantic Router", "url": "https://vllm.ai/blog/2026-06-16-vllm-sr-fusion-api", "published_at": "2026-06-16T00:00:00+00:00" }, { "id": "01a0823d-891e-70d0-b8c7-ee6287f5b790", "title": "MiniMax M3 in vLLM: Day-0 Serving for 1M-Token Multimodal Reasoning", "url": "https://vllm.ai/blog/2026-06-12-minimax-m3-vllm", "published_at": "2026-06-12T00:00:00+00:00" }, { "id": "01a0823d-891e-70d0-b8c7-ee628882e729", "title": "DiffusionGemma: The First Diffusion LLM (dLLM) Natively Supported in vLLM", "url": "https://vllm.ai/blog/2026-06-10-diffusion-gemma", "published_at": "2026-06-10T00:00:00+00:00" }, { "id": "01a0823d-891e-70d0-b8c7-ee62889cf49c", "title": "Announcing vime: A Simple, Stable, and Efficient RL Framework for LLMs", "url": "https://vllm.ai/blog/2026-06-09-announcing-vime", "published_at": "2026-06-09T00:00:00+00:00" }, { "id": "01a0823d-891e-70d0-b8c7-ee62896b599c", "title": "vLLM Semantic Router v0.3 Themis: From Signals to Stateful Production Routing", "url": "https://vllm.ai/blog/2026-06-05-v0.3-vllm-sr-themis-release", "published_at": "2026-06-05T00:00:00+00:00" }, { "id": "01a0823d-891e-70d0-b8c7-ee628a5dadbb", "title": "Announcing Day-0 Support for NVIDIA Nemotron 3 Ultra on vLLM", "url": "https://vllm.ai/blog/2026-06-04-nemotron-3-ultra-vllm", "published_at": "2026-06-04T00:00:00+00:00" }, { "id": "01a0823d-891e-70d0-b8c7-ee628a991914", "title": "Fast & Efficient LLM Inference with vLLM: A New Course with DeepLearning.AI", "url": "https://vllm.ai/blog/2026-06-03-deeplearning-ai-vllm-course", "published_at": "2026-06-03T00:00:00+00:00" }, { "id": "01a0823d-891e-70d0-b8c7-ee628abf674f", "title": "Session-Aware Agentic Routing: Continuity-Aware Model Selection for Long-Horizon LLM Agents", "url": "https://vllm.ai/blog/2026-06-02-session-aware-agentic-routing", "published_at": "2026-06-02T00:00:00+00:00" }, { "id": "01a0823d-891e-70d0-b8c7-ee628ada74c2", "title": "Accelerating vLLM-Omni Inference with AutoRound Quantization", "url": "https://vllm.ai/blog/2026-06-02-vllm-omni-autoround", "published_at": "2026-06-02T00:00:00+00:00" }, { "id": "01a0823d-891e-70d0-b8c7-ee628b869c2c", "title": "vLLM on the DGX Spark: Architecture, Configuration, and Local Evaluation", "url": "https://vllm.ai/blog/2026-06-01-vllm-dgx-spark", "published_at": "2026-06-01T00:00:00+00:00" }, { "id": "01a0823d-891e-70d0-b8c7-ee628b879b56", "title": "Accelerating Laguna XS.2 Inference with vLLM, Speculators, and LLM Compressor", "url": "https://vllm.ai/blog/2026-05-28-laguna-xs2-dflash-llm-compressor", "published_at": "2026-05-28T00:00:00+00:00" }, { "id": "01a0823d-891e-70d0-b8c7-ee628c3b8664", "title": "Native RL APIs in vLLM", "url": "https://vllm.ai/blog/2026-05-28-native-rl-apis", "published_at": "2026-05-28T00:00:00+00:00" }, { "id": "01a0823d-891e-70d0-b8c7-ee628d04942a", "title": "Speculators v0.5.0: DFlash Support and Online Training", "url": "https://vllm.ai/blog/2026-05-28-speculators-v050", "published_at": "2026-05-28T00:00:00+00:00" }, { "id": "01a0823d-891e-70d0-b8c7-ee628d2e545e", "title": "From Text to Multimodal Routing: Hardening Vision Signals in vLLM Semantic Router", "url": "https://vllm.ai/blog/2026-05-28-vllm-sr-vision-encoder-hardening", "published_at": "2026-05-28T00:00:00+00:00" }, { "id": "01a0823d-891e-70d0-b8c7-ee628d41b264", "title": "EAGLE 3.1: Advancing Speculative Decoding Through Collaboration Between the EAGLE Team, vLLM, and TorchSpec", "url": "https://vllm.ai/blog/2026-05-26-eagle-3-1", "published_at": "2026-05-26T00:00:00+00:00" }, { "id": "01a0823d-891e-70d0-b8c7-ee628d791f92", "title": "vLLM x Novita AI: PegaFlow for Production-Grade External KV Cache", "url": "https://vllm.ai/blog/2026-05-18-pegaflow", "published_at": "2026-05-18T00:00:00+00:00" }, { "id": "01a0823d-891e-70d0-b8c7-ee628e2bfb71", "title": "Elastic Expert Parallelism in vLLM", "url": "https://vllm.ai/blog/2026-05-14-elastic-expert-parallelism", "published_at": "2026-05-14T00:00:00+00:00" }, { "id": "01a0823d-891e-70d0-b8c7-ee628edb76c6", "title": "Announcing VeRL-Omni: Easy, Fast, and Stable RL Training for Diffusion and Omni-Modality Models", "url": "https://vllm.ai/blog/2026-05-14-verl-omni", "published_at": "2026-05-14T00:00:00+00:00" }, { "id": "01a0823d-891e-70d0-b8c7-ee628f6ac399", "title": "A First Comprehensive Study of TurboQuant: Accuracy and Performance", "url": "https://vllm.ai/blog/2026-05-11-turboquant", "published_at": "2026-05-11T00:00:00+00:00" }, { "id": "01a0823d-891e-70d0-b8c7-ee62904de457", "title": "vLLM Tops the Artificial Analysis Leaderboard", "url": "https://vllm.ai/blog/2026-05-11-vllm-tops-artificial-analysis", "published_at": "2026-05-11T00:00:00+00:00" }, { "id": "01a0823d-891e-70d0-b8c7-ee6290f948fa", "title": "Serving Agentic Workloads at Scale with vLLM x Mooncake", "url": "https://vllm.ai/blog/2026-05-06-mooncake-store", "published_at": "2026-05-06T00:00:00+00:00" }, { "id": "01a07df2-0d25-7342-8321-fed1362bd817", "title": "Run Highly Efficient Multimodal Agentic AI with NVIDIA Nemotron 3 Nano Omni Using vLLM", "url": "https://vllm.ai/blog/2026-04-28-nemotron-omni", "published_at": "2026-04-28T00:00:00+00:00" } ] posts Claim your blog
Back to vLLM Blog
Blog · corpus.blog/blogs/vllm.ai/posts

vLLM Blog

vllm.ai

2026