{
  "id": "draft-window",
  "story": "qwen-next-flash-exl3-350hq-mtp-draft-q8-2026-09-16",
  "title": "Three draft tokens hit the sweet spot",
  "unit": "tokens / second",
  "rows": [
    {
      "label": "Drafting off",
      "value": 64.67,
      "highlight": false
    },
    {
      "label": "2 draft tokens",
      "value": 84.38,
      "highlight": false
    },
    {
      "label": "3 draft tokens",
      "value": 88.16,
      "highlight": true
    },
    {
      "label": "4 draft tokens",
      "value": 80.71,
      "highlight": false
    },
    {
      "label": "6 draft tokens",
      "value": 62.35,
      "highlight": false
    }
  ],
  "sources": [
    {
      "path": "qwen-next-flash/exl3-350hq-mtp-draft-q8-2026-09-16/data/draft-window-sweep.json",
      "url": "https://github.com/groxaxo/experimentos/blob/6550ead3945b7ecacbaac1e0767edc49a9de9747/qwen-next-flash/exl3-350hq-mtp-draft-q8-2026-09-16/data/draft-window-sweep.json",
      "sha256": "887e9f5aabf5cdccea761f75e114a4a16fe0fd4043440d50b84e89d3a1e2e6e0"
    }
  ],
  "revision": "6550ead3945b7ecacbaac1e0767edc49a9de9747",
  "scope": "Short-prose mean · EXL3 · Q8 cache · three RTX 3090s",
  "caveat": "The separate longer prompt and concurrent streams are different workloads.",
  "kind": "bar",
  "maximum": null,
  "threshold": null,
  "decimals": 1,
  "origin": "structured measurements"
}