<?xml version="1.0" encoding="utf-8" standalone="yes"?>
<rss version="2.0" xmlns:atom="http://www.w3.org/2005/Atom" xmlns:content="http://purl.org/rss/1.0/modules/content/">
  <channel>
    <title>VLLM on Ge Zhang · 技术笔记</title>
    <link>https://zhangge.dev/tags/vllm/</link>
    <description>Recent content in VLLM on Ge Zhang · 技术笔记</description>
    <image>
      <title>Ge Zhang · 技术笔记</title>
      <url>https://zhangge.dev/images/site-card.png</url>
      <link>https://zhangge.dev/images/site-card.png</link>
    </image>
    <generator>Hugo</generator>
    <language>zh-CN</language>
    <copyright>2026 Ge Zhang</copyright>
    <lastBuildDate>Thu, 08 Oct 2026 00:00:00 +0000</lastBuildDate>
    <atom:link href="https://zhangge.dev/tags/vllm/index.xml" rel="self" type="application/rss+xml" />
    <item>
      <title>从 Padding 到 Continuous Batching：LLM 推理中的长度感知调度</title>
      <link>https://zhangge.dev/model-inference/length-aware-scheduling/</link>
      <pubDate>Thu, 08 Oct 2026 00:00:00 +0000</pubDate>
      <guid>https://zhangge.dev/model-inference/length-aware-scheduling/</guid>
      <description>从 Length Bucketing、Token Budget、Packed/Ragged Batching 到 Continuous Batching 与 Chunked Prefill，梳理变长请求在 LLM 推理中的组批、存放和调度方式。</description>
    </item>
  </channel>
</rss>
