{
  "id": "expert-split",
  "story": "qwen-next-flash-udq4kxl-cpu-moe-200k-2026-09-20",
  "title": "Balancing the expert split improved decode",
  "unit": "tokens / second",
  "rows": [
    {
      "label": "8 / 8 / 8",
      "value": 18.9,
      "highlight": false
    },
    {
      "label": "10 / 10 / 10",
      "value": 22.1,
      "highlight": false
    },
    {
      "label": "11 / 11 / 11",
      "value": 26.1,
      "highlight": false
    },
    {
      "label": "12 / 12 / 12",
      "value": 31.1,
      "highlight": true
    },
    {
      "label": "12 / 13 / 12",
      "value": 30.9,
      "highlight": false
    }
  ],
  "sources": [
    {
      "path": "qwen-next-flash/udq4kxl-cpu-moe-200k-2026-09-20/README.md",
      "url": "https://github.com/groxaxo/experimentos/blob/6550ead3945b7ecacbaac1e0767edc49a9de9747/qwen-next-flash/udq4kxl-cpu-moe-200k-2026-09-20/README.md",
      "sha256": "3cc5b19973ab0807bf894d38eb676314fa0f45fd8c83fcb79a12b865865e2f35"
    }
  ],
  "revision": "6550ead3945b7ecacbaac1e0767edc49a9de9747",
  "scope": "GPU expert blocks · 204,800 context · Q8 KV · 300 generated tokens · three RTX 3090s",
  "caveat": "Values transcribed from the report table; cross-runtime comparison uses a different quant.",
  "kind": "bar",
  "maximum": null,
  "threshold": null,
  "decimals": 1,
  "origin": "reported table"
}