{
  "id": "embedding-throughput",
  "story": "nemotron-embed-1b-vs-8b-awq-2026-09-18",
  "title": "The smaller encoder moved more documents",
  "unit": "documents / second",
  "rows": [
    {
      "label": "1B · native 2048d",
      "value": 1003.7522851503292,
      "highlight": true
    },
    {
      "label": "8B · native 4096d",
      "value": 180.95503599040228,
      "highlight": false
    }
  ],
  "sources": [
    {
      "path": "nemotron/embed-1b-vs-8b-awq-2026-09-18/data/summary.json",
      "url": "https://github.com/groxaxo/experimentos/blob/6550ead3945b7ecacbaac1e0767edc49a9de9747/nemotron/embed-1b-vs-8b-awq-2026-09-18/data/summary.json",
      "sha256": "257a82dcb3b17e533accbce9db75e9e640d7b51e93a3ecc29beff7a095519588"
    }
  ],
  "revision": "6550ead3945b7ecacbaac1e0767edc49a9de9747",
  "scope": "275 documents · batch size 32 · AWQ W4A16",
  "caveat": "Amortized throughput on this corpus, not single-query latency.",
  "kind": "bar",
  "maximum": null,
  "threshold": null,
  "decimals": 1,
  "origin": "structured measurements"
}