{
  "id": "embedding-memory",
  "story": "nemotron-embed-1b-vs-8b-awq-2026-09-18",
  "title": "Memory above the idle GPU baseline",
  "unit": "MiB",
  "rows": [
    {
      "label": "1B",
      "value": 2380.0,
      "highlight": true
    },
    {
      "label": "8B",
      "value": 7530.0,
      "highlight": false
    }
  ],
  "sources": [
    {
      "path": "nemotron/embed-1b-vs-8b-awq-2026-09-18/data/summary.json",
      "url": "https://github.com/groxaxo/experimentos/blob/6550ead3945b7ecacbaac1e0767edc49a9de9747/nemotron/embed-1b-vs-8b-awq-2026-09-18/data/summary.json",
      "sha256": "257a82dcb3b17e533accbce9db75e9e640d7b51e93a3ecc29beff7a095519588"
    }
  ],
  "revision": "6550ead3945b7ecacbaac1e0767edc49a9de9747",
  "scope": "Peak GPU allocation delta · 4 MiB baseline",
  "caveat": "Native dimensions differ. Index storage is a separate cost.",
  "kind": "bar",
  "maximum": null,
  "threshold": null,
  "decimals": 0,
  "origin": "structured measurements"
}