Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions astro.config.mjs
Original file line number Diff line number Diff line change
Expand Up @@ -21,6 +21,7 @@ export default defineConfig({
components: {
Header: './src/components/Header.astro',
PageTitle: './src/components/PageTitle.astro',
PageSidebar: './src/components/PageSidebar.astro',
Footer: './src/components/Footer.astro',
},
sidebar: [
Expand Down
Binary file added public/diffusion-denoising.gif
Loading
Sorry, something went wrong. Reload?
Sorry, we cannot display this file.
Sorry, this file is invalid so it cannot be displayed.
144 changes: 144 additions & 0 deletions src/components/GenerationScalingChart.astro
Original file line number Diff line number Diff line change
@@ -0,0 +1,144 @@
---
type Point = { rate: number; throughput: number };
type Panel = {
title: string;
xMax: number;
yMax: number;
xTicks: number[];
yTicks: number[];
fluxServe: Point[];
sgLang: Point[];
};

// Approximate values read from the supplied BigCodeBench plot.
const panels: Panel[] = [
{
title: 'LLaDA2.0-mini', xMax: 400, yMax: 1400,
xTicks: [0, 100, 200, 300, 400], yTicks: [0, 400, 800, 1200],
fluxServe: [{ rate: 81, throughput: 1290 }, { rate: 114, throughput: 915 }, { rate: 165, throughput: 660 }, { rate: 231, throughput: 460 }, { rate: 350, throughput: 350 }],
sgLang: [{ rate: 55, throughput: 879 }, { rate: 82, throughput: 648 }, { rate: 135, throughput: 540 }, { rate: 214, throughput: 429 }, { rate: 371, throughput: 370 }],
},
{
title: 'LLaDA2.1-mini', xMax: 820, yMax: 2700,
xTicks: [0, 200, 400, 600, 800], yTicks: [0, 500, 1000, 1500, 2000, 2500],
fluxServe: [{ rate: 160, throughput: 2550 }, { rate: 230, throughput: 1840 }, { rate: 340, throughput: 1360 }, { rate: 513, throughput: 1020 }, { rate: 780, throughput: 780 }],
sgLang: [{ rate: 86, throughput: 1383 }, { rate: 120, throughput: 955 }, { rate: 257, throughput: 1023 }, { rate: 444, throughput: 883 }, { rate: 773, throughput: 773 }],
},
{
title: 'LLaDA2.0-flash', xMax: 260, yMax: 750,
xTicks: [0, 50, 100, 150, 200, 250], yTicks: [0, 150, 300, 450, 600, 750],
fluxServe: [{ rate: 45, throughput: 720 }, { rate: 61, throughput: 494 }, { rate: 86, throughput: 344 }, { rate: 124, throughput: 250 }, { rate: 199, throughput: 200 }],
sgLang: [{ rate: 33, throughput: 528 }, { rate: 49, throughput: 386 }, { rate: 77, throughput: 305 }, { rate: 116, throughput: 232 }, { rate: 244, throughput: 243 }],
},
{
title: 'LLaDA2.1-flash', xMax: 520, yMax: 1300,
xTicks: [0, 100, 200, 300, 400, 500], yTicks: [0, 200, 400, 600, 800, 1000, 1200],
fluxServe: [{ rate: 78, throughput: 1247 }, { rate: 111, throughput: 891 }, { rate: 165, throughput: 662 }, { rate: 260, throughput: 518 }, { rate: 394, throughput: 394 }],
sgLang: [{ rate: 46, throughput: 782 }, { rate: 86, throughput: 701 }, { rate: 136, throughput: 554 }, { rate: 250, throughput: 500 }, { rate: 490, throughput: 490 }],
},
];

const plotX = 66;
const plotY = 48;
const plotWidth = 454;
const plotHeight = 244;
const xy = (point: Point, panel: Panel) => ({
x: plotX + point.rate / panel.xMax * plotWidth,
y: plotY + plotHeight - point.throughput / panel.yMax * plotHeight,
});
const path = (points: Point[], panel: Panel) => points.map((point, index) => {
const { x, y } = xy(point, panel);
return `${index ? 'L' : 'M'}${x.toFixed(1)} ${y.toFixed(1)}`;
}).join(' ');
---

<figure class="generation-scaling" aria-describedby="scaling-caption">
<div class="chart-scroller" tabindex="0" aria-label="Scrollable BigCodeBench generation scaling chart">
<svg viewBox="0 0 1120 824" role="img" aria-labelledby="scaling-title scaling-description">
<title id="scaling-title">Performance Scaling</title>
<desc id="scaling-description">Four plots compare FluxServe and SGLang GPU throughput against user throughput for LLaDA 2.0 and 2.1 mini and flash models.</desc>

<line class="title-rule" x1="18" x2="18" y1="15" y2="55" />
<text class="chart-title" x="40" y="44">Performance scaling</text>
<g class="legend">
<line class="flux-line" x1="740" x2="774" y1="37" y2="37" />
<circle class="flux-point" cx="757" cy="37" r="5" />
<text x="784" y="42">FluxServe</text>
<line class="sg-line" x1="960" x2="994" y1="37" y2="37" />
<rect class="sg-point" x="972" y="32" width="10" height="10" />
<text x="1004" y="42">SGLang</text>
</g>

{panels.map((panel, index) => {
const originX = 12 + index % 2 * 554;
const originY = 68 + Math.floor(index / 2) * 378;
return <g class="panel" transform={`translate(${originX} ${originY})`}>
<text class="panel-title" x={plotX + plotWidth / 2} y="26" text-anchor="middle">{panel.title}</text>
{panel.yTicks.map((tick) => {
const y = plotY + plotHeight - tick / panel.yMax * plotHeight;
return <g>
<line class="grid-line" x1={plotX} x2={plotX + plotWidth} y1={y} y2={y} />
<text class="tick-label" x={plotX - 10} y={y + 4} text-anchor="end">{tick}</text>
</g>;
})}
{panel.xTicks.map((tick) => {
const x = plotX + tick / panel.xMax * plotWidth;
return <g>
<line class="grid-line" x1={x} x2={x} y1={plotY} y2={plotY + plotHeight} />
<text class="tick-label" x={x} y={plotY + plotHeight + 19} text-anchor="middle">{tick}</text>
</g>;
})}
<line class="axis" x1={plotX} x2={plotX + plotWidth} y1={plotY + plotHeight} y2={plotY + plotHeight} />
<line class="axis" x1={plotX} x2={plotX} y1={plotY} y2={plotY + plotHeight} />
<text class="axis-label" x={plotX + plotWidth / 2} y="335" text-anchor="middle">User Throughput (token/s)</text>
<text class="axis-label" transform={`translate(18 ${plotY + plotHeight / 2}) rotate(-90)`} text-anchor="middle">GPU Throughput (token/s)</text>

<path class="flux-line series-line" d={path(panel.fluxServe, panel)} pathLength="1" />
<path class="sg-line series-line" d={path(panel.sgLang, panel)} pathLength="1" />
{panel.fluxServe.map((point) => {
const { x, y } = xy(point, panel);
return <g class="point-group">
<circle class="flux-point" cx={x} cy={y} r="5" />
<title>FluxServe: {point.rate} user tokens/s, {point.throughput} GPU tokens/s</title>
</g>;
})}
{panel.sgLang.map((point) => {
const { x, y } = xy(point, panel);
return <g class="point-group">
<rect class="sg-point" x={x - 5} y={y - 5} width="10" height="10" />
<title>SGLang: {point.rate} user tokens/s, {point.throughput} GPU tokens/s</title>
</g>;
})}
</g>;
})}
</svg>
</div>
<figcaption id="scaling-caption">Results are reported by sweeping cocurrency from 1 to 16 on BigCodeBench dataset.</figcaption>
</figure>

<style>
.generation-scaling { width: 90%; max-width: 800px; min-width: 0; margin: 0 auto; overflow: hidden; border: 1px solid var(--line); border-radius: 12px; background: #fffdf7; }
.chart-scroller { overflow-x: auto; overscroll-behavior-inline: contain; }
.chart-scroller:focus-visible { outline: 2px solid var(--brand); outline-offset: -3px; }
svg { display: block; width: 100%; min-width: 700px; height: auto; }
svg text { fill: #3b4144; font-family: var(--sl-font-mono); }
.title-rule { stroke: #ff7770; stroke-width: 4; }
.chart-title { font-size: 27px; letter-spacing: .02em; }
.legend text { font-size: 18px; }
.panel-title { font-size: 20px; font-weight: 400; }
.tick-label { font-size: 14px; fill: #62696c; }
.axis-label { font-size: 16px; fill: #555d60; }
.grid-line { stroke: #e5e6e3; stroke-width: 1; }
.axis { stroke: #646b6e; stroke-width: 1.3; }
.flux-line { fill: none; stroke: #dc6c62; stroke-width: 3; stroke-linecap: round; stroke-linejoin: round; }
.sg-line { fill: none; stroke: #e58c1c; stroke-width: 3; stroke-linecap: round; stroke-linejoin: round; }
.flux-point { fill: #fc9b90; stroke: #dc6c62; stroke-width: 1.5; }
.sg-point { fill: #ffb45d; stroke: #e58c1c; stroke-width: 1.5; }
figcaption { padding: .65rem 1rem; border-top: 1px solid #e7e2d9; color: #555d60; background: #fffdf7; font-size: .72rem; line-height: 1.5; }
:global([data-theme='dark']) .generation-scaling, :global([data-theme='dark']) figcaption { background: #202124; }
:global([data-theme='dark']) svg text { fill: #e7e4df; }
:global([data-theme='dark']) .tick-label, :global([data-theme='dark']) .axis-label, :global([data-theme='dark']) figcaption { color: #c9c5be; fill: #c9c5be; }
:global([data-theme='dark']) .grid-line { stroke: #42484a; }
:global([data-theme='dark']) .axis { stroke: #a4aaac; }
:global([data-theme='dark']) figcaption { border-top-color: #454545; }
</style>
8 changes: 8 additions & 0 deletions src/components/PageSidebar.astro
Original file line number Diff line number Diff line change
@@ -0,0 +1,8 @@
---
import DefaultPageSidebar from '@astrojs/starlight/components/PageSidebar.astro';

const toc = Astro.locals.starlightRoute.toc;
if (toc) toc.items = toc.items.flatMap((item) => item.slug === '_top' ? item.children : [item]);
---

<DefaultPageSidebar />
10 changes: 5 additions & 5 deletions src/components/SpeedBenchmarkChart.astro
Original file line number Diff line number Diff line change
@@ -1,9 +1,9 @@
---
const results = [
{ model: 'LLaDA-2.0-mini', fluxServe: 1300, sgLang: 827 },
{ model: 'LLaDA-2.0-flash', fluxServe: 729, sgLang: 507 },
{ model: 'LLaDA-2.1-mini', fluxServe: 2451, sgLang: 1299 },
{ model: 'LLaDA-2.1-flash', fluxServe: 1279, sgLang: 805 },
{ model: 'LLaDA2.0-mini', fluxServe: 1300, sgLang: 827 },
{ model: 'LLaDA2.0-flash', fluxServe: 729, sgLang: 507 },
{ model: 'LLaDA2.1-mini', fluxServe: 2451, sgLang: 1299 },
{ model: 'LLaDA2.1-flash', fluxServe: 1279, sgLang: 805 },
];

const baseline = 520;
Expand Down Expand Up @@ -43,7 +43,7 @@ const barTop = (value: number) => baseline - value / maxValue * chartHeight;
})}
</svg>
</div>
<figcaption id="speed-caption">Mini models are run with TP=EP=1, and flash models are run with TP=EP=4 on 4 x NVIDIA GH200 GPUs.</figcaption>
<figcaption id="speed-caption">Mini models are run with TP=EP=1, and flash models are run with TP=EP=4 on NVIDIA GH200 GPUs.</figcaption>
</figure>

<script>
Expand Down
53 changes: 29 additions & 24 deletions src/content/blog/introducing-fluxserve.md
Original file line number Diff line number Diff line change
@@ -1,62 +1,68 @@
---
title: "FluxServe: A Flexible and High-Performance Inference Engine for Open Diffusion Language Models"
description: We are excited to announce FluxServe, a new inference engine built specifically for open diffusion language models. FluxServe is designed and implemented to deliver low latency and high throughput for autoregressive (AR) diffusion models through an optimized attention runtime, dynamic block-level scheduling, and efficient multi-GPU serving.
description: FluxServe is a lightweight inference serving engine built specifically for open diffusion language models. FluxServe is designed and implemented to deliver low latency and high throughput for autoregressive (AR) diffusion models through an optimized attention runtime, dynamic block-level scheduling, and efficient multi-GPU serving.
date: 2026-09-27
author: FluxServe Team
draft: false
---

Recently, diffusion large language models (dLLMs) have emerged as a highly promising alternative to traditional autoregressive (AR) LLMs. The rapidly growing popularity of dLLMs stems from their unique architectural advantage: combining long-horizon causal sequencing with high-fidelity iterative refinement, allowing multiple tokens to be denoised simultaneously in a bidirectional manner. Exemplified by recent open-weight releases, dLLMs offer compelling generative capabilities with superior compute utilization, promoting a more efficient token economy for modern AI workloads.
Recently, diffusion large language models (dLLMs) have emerged as a highly promising alternative to traditional autoregressive (AR) LLMs.
The rapidly growing popularity of dLLMs stems from their unique architectural advantage: combining long-horizon causal sequencing with high-fidelity iterative refinement, allowing multiple tokens to be denoised simultaneously in a bidirectional manner.
Exemplified by recent open-weight releases, dLLMs offer compelling generative capabilities with superior compute utilization, promoting a more efficient token economy for modern AI workloads.

### What is a Diffusion Language Model (dLLM)?
Diffusion models are a well-established class of generative models that learn to transform noise into data through an iterative denoising process.
While widely adopted in image and video generation, where models progressively refine random noise into high-quality visuals, applying diffusion to language is a rapidly emerging frontier.

Diffusion models are a well-established class of generative models that learn to transform noise into data through an iterative denoising process. While widely adopted in image and video generation—where models progressively refine random noise into high-quality visuals—applying diffusion to language is a rapidly emerging frontier.

Instead of predicting text token-by-token, dLLMs take a block of masked tokens and gradually refine them into coherent text. This unique parallel decoding structure provides dLLMs with bidirectional context and enables block-level parallelism during generation. However, this also introduces significant challenges for existing AR serving stacks. Because current inference components are heavily optimized for sequential token generation, they are fundamentally sub-optimal for the block-level workloads required by dLLMs.
Instead of predicting text token-by-token, dLLMs take a block of masked tokens and gradually refine them into coherent text. This unique parallel decoding structure provides dLLMs with bidirectional context and enables block-level parallelism during generation.
However, this also introduces significant challenges for efficient deployment.
Exsiting AR serving stacks are heavily optimized for sequential token generation, thereby are fundamentally sub-optimal for the block-level workloads required by dLLMs.

<div data-block-diffusion-slot></div>

### FluxServe Overview

[FluxServe](https://github.com/FLX-OSS/FluxServe) is a lightweight and high-performance serving engine engineered specifically for diffusion language models. It is designed to deliver low-latency and high-throughput inference for autoregressive diffusion models across a variety of hardware setups, ranging from single-GPU batched inference to multi-GPU distributed serving.
[FluxServe](https://github.com/FLX-OSS/FluxServe) is a lightweight and high-performance serving engine engineered specifically for diffusion language models.
It is designed to deliver low-latency and high-throughput inference for autoregressive diffusion models across a variety of hardware setups, ranging from single-GPU batched inference to multi-GPU distributed serving.

At launch, FluxServe’s core features include:

- **Native Block-Causal Attention**: FluxServe implements an efficient block-causal attention runtime tailored for AR diffusion in real-world scenarios. It supports both variable-length (varlen) prefill and varlen block-decode, with backend support for both FlashInfer and FA4.
- **Dynamic Request Scheduler**: FluxServe features a hybrid scheduling architecture, pairing a low-overhead C++ control plane with a Python execution plane. This enables the fine-grained, block-level request management essential for diffusion models.
- **Unified Diffusion Playground**: FluxServe provides native support for a wide range of open diffusion language models, such as [LLaDA 2.X](https://github.com/inclusionAI/LLaDA2.X) and [Diffusion-Gemma](https://huggingface.co/google/diffusiongemma-26B-A4B-it), establishing a standardized benchmarking platform for both academic researchers and industry practitioners.
- **Unified Diffusion Playground**: FluxServe provides native support for a wide range of open diffusion language models, such as [LLaDA2.X](https://github.com/inclusionAI/LLaDA2.X) and [Diffusion-Gemma](https://huggingface.co/google/diffusiongemma-26B-A4B-it), establishing a standardized benchmarking platform for both academic researchers and industry practitioners.

### Performance Results

Here, we present preliminary benchmark results comparing FluxServe against SGLang. To ensure a fair comparison, we launched both engines as API endpoints and utilized the third-party evaluation tool [evalscope](https://github.com/modelscope/evalscope) to measure system performance across multiple datasets.

The figure below highlights the LLaDA 2.0 and 2.1 performance of FluxServe versus SGLang. Across four different model configurations, FluxServe achieves a consistent decode throughput improvement over AR-oriented serving stacks, delivering an average speedup of 1.6x.
Here, we present preliminary benchmark results comparing FluxServe against [SGLang](https://github.com/sgl-project/sglang).
To ensure a fair comparison, we launched both engines as API endpoints and utilized the third-party evaluation tool [evalscope](https://github.com/modelscope/evalscope) to measure system performance across multiple datasets.
The figure below highlights the throughput performance on LLaDA2 models.
Across different model configurations, FluxServe achieves a consistent decode throughput improvement over AR-oriented serving stacks, delivering an average speedup of 1.6x.


<div data-benchmark-slot></div>

We further show the Pareto curves of FluxServe and SGLang on the BigCodeBench dataset.
Each curve plots user throughput (x-axis) against GPU throughput (y-axis).
FluxServe achieves higher GPU throughput than SGLang at comparable user throughput.
For LLaDA2.1-flash, at around 100 user tokens/s, FluxServe delivers over 1000 token/s, showing roughly 60% higher GPU throughput than SGLang.

### Roadmaps
<div data-scaling-slot></div>

### Short-term Implementations
- Extensive Model Support: [Nemotron-Labs-Diffusion](https://github.com/FLX-OSS/FluxServe/pull/14)
- Advanced Quantization: FP8 & NVFP4
- NVIDIA Blackwell GPU Support
- Production Model Gateway: gRPC


### Long-term Goals
- AMD ROCm Support
- Multi-Modal Diffusion Support
### Roadmaps

- Model Support: [Nemotron-Labs-Diffusion](https://github.com/FLX-OSS/FluxServe/pull/14), [Diffusion-Gemma](https://github.com/FLX-OSS/FluxServe/issues/15)
- [NVIDIA Blackwell GPU Support](https://github.com/FLX-OSS/FluxServe/issues/16)
- Advanced Quantization: FP8 & NVFP4


### External Contributions
FluxServe is built as a lightweight and performance-oriented serving infrastructure project. Due the widespread use of AI agents, we will be intentionally selective about the submitted PRs, and conduct thorough discussion and validation before merging.
FluxServe is built as a lightweight and performance-oriented serving infrastructure project. Due the widespread use of AI agents, we will be intentionally selective about the submitted issues and PRs, and conduct thorough discussion and validation before merging.
We appreciate everyone's ideas, support and feedback to our project.
We welcome external contributions, especially:
- Obvious bug fixes
- Performance optimizations that fit the existing codebase style without additional unnecessary complexity
- Documentation, tooling, and benchmarking improvements
- Benchmarks, observability and documentation improvements

### Acknowledgements

Expand All @@ -69,7 +75,7 @@ Our system design was inspired by, and incorporates reused code from, the follow
title = {FluxServe: A Flexible and High-Performance Inference Engine for Open Diffusion Language Models},
year = {2026},
month = {September},
howpublished = {\url{[https://github.com/FLX-OSS/FluxServe](https://github.com/FLX-OSS/FluxServe)}}
howpublished = {\url{(https://github.com/FLX-OSS/FluxServe)}}
}
```

Expand All @@ -78,4 +84,3 @@ Our system design was inspired by, and incorporates reused code from, the follow
- Project Lead & Creator: [Youpeng Zhao](https://kennethzhao24.github.io/)
- Model Runtime: [Meiling Wang](https://meiling0131.github.io/), [Depng Zhu](https://github.com/zhudp3)
- Benchmark & Documentation: [Zhiben Chen](https://www.linkedin.com/in/zhiben-chen/), [Ziyan Wang](https://www.linkedin.com/in/ziyan-wang-00a163228/)

Loading
Loading