{
  "source": "https://github.com/Eliovp-BV/paiton-vllm-plugin/blob/6586aa618610aed596d2720fbb5768c181bf1011/models/Qwen-Image-2.1/BENCHMARKS.md#container-validation",
  "release": "v1.0.2",
  "workload": "One 32 GB Radeon AI PRO R9700; balanced MXFP4 checkpoint; 2048 x 2048; 40 steps; guidance 1.0; batch one; complete HTTP request through PNG receipt; excludes startup and download",
  "timings": [
    { "runtime": "v1.0.1 historical repeat", "warmSeconds": 165.14, "firstSeconds": 179.32, "peakWholeDeviceGiB": 28.34, "processes": 3 },
    { "runtime": "v1.0.2 exact", "warmSeconds": 133.74, "firstSeconds": 147.83, "peakWholeDeviceGiB": 25.33, "processes": 1 },
    { "runtime": "v1.0.2 default", "warmSeconds": 103.29, "firstSeconds": 113.95, "peakWholeDeviceGiB": 25.55, "processes": 3 }
  ],
  "qualifier": "The three-process times are medians. Exact is one validation run. Peak VRAM is the maximum sampled whole-device usage. These are separate release repeats, not one matched experiment."
}
