{
 "schema_version": 1,
 "as_of": "2026-10-07",
 "description": "Weight formats, KV-cache types, system RAM bandwidth presets and the stated assumptions used by the Run AI at home calculators. Every number is either arithmetic from a published layout, a figure from the cited source, or an assumption labeled as one.",
 "weight_formats": [
  {
   "id": "gguf-q2_k",
   "label": "Q2_K",
   "group": "GGUF (llama.cpp, Ollama, LM Studio)",
   "engine": "llama.cpp",
   "bpw": 3.1593,
   "basis": "Effective bits per weight of a whole Q2_K file of Llama 3.1 8B in llama.cpp's table (2.95 GiB GiB). Mixes keep some tensors at higher precision, so other architectures differ by a few percent.",
   "source_url": "https://github.com/ggml-org/llama.cpp/blob/9c2e0e491a822adae1f0b1c831adb4160057d24f/tools/quantize/README.md#L141-L176",
   "source_label": "llama.cpp quantize README"
  },
  {
   "id": "gguf-q3_k_s",
   "label": "Q3_K_S",
   "group": "GGUF (llama.cpp, Ollama, LM Studio)",
   "engine": "llama.cpp",
   "bpw": 3.6429,
   "basis": "Effective bits per weight of a whole Q3_K_S file of Llama 3.1 8B in llama.cpp's table (3.41 GiB GiB). Mixes keep some tensors at higher precision, so other architectures differ by a few percent.",
   "source_url": "https://github.com/ggml-org/llama.cpp/blob/9c2e0e491a822adae1f0b1c831adb4160057d24f/tools/quantize/README.md#L141-L176",
   "source_label": "llama.cpp quantize README"
  },
  {
   "id": "gguf-q3_k_m",
   "label": "Q3_K_M",
   "group": "GGUF (llama.cpp, Ollama, LM Studio)",
   "engine": "llama.cpp",
   "bpw": 3.996,
   "basis": "Effective bits per weight of a whole Q3_K_M file of Llama 3.1 8B in llama.cpp's table (3.74 GiB GiB). Mixes keep some tensors at higher precision, so other architectures differ by a few percent.",
   "source_url": "https://github.com/ggml-org/llama.cpp/blob/9c2e0e491a822adae1f0b1c831adb4160057d24f/tools/quantize/README.md#L141-L176",
   "source_label": "llama.cpp quantize README"
  },
  {
   "id": "gguf-iq4_xs",
   "label": "IQ4_XS",
   "group": "GGUF (llama.cpp, Ollama, LM Studio)",
   "engine": "llama.cpp",
   "bpw": 4.4597,
   "basis": "Effective bits per weight of a whole IQ4_XS file of Llama 3.1 8B in llama.cpp's table (4.17 GiB GiB). Mixes keep some tensors at higher precision, so other architectures differ by a few percent.",
   "source_url": "https://github.com/ggml-org/llama.cpp/blob/9c2e0e491a822adae1f0b1c831adb4160057d24f/tools/quantize/README.md#L141-L176",
   "source_label": "llama.cpp quantize README"
  },
  {
   "id": "gguf-q4_k_s",
   "label": "Q4_K_S",
   "group": "GGUF (llama.cpp, Ollama, LM Studio)",
   "engine": "llama.cpp",
   "bpw": 4.6672,
   "basis": "Effective bits per weight of a whole Q4_K_S file of Llama 3.1 8B in llama.cpp's table (4.36 GiB GiB). Mixes keep some tensors at higher precision, so other architectures differ by a few percent.",
   "source_url": "https://github.com/ggml-org/llama.cpp/blob/9c2e0e491a822adae1f0b1c831adb4160057d24f/tools/quantize/README.md#L141-L176",
   "source_label": "llama.cpp quantize README"
  },
  {
   "id": "gguf-q4_k_m",
   "label": "Q4_K_M",
   "group": "GGUF (llama.cpp, Ollama, LM Studio)",
   "engine": "llama.cpp",
   "bpw": 4.8944,
   "basis": "Effective bits per weight of a whole Q4_K_M file of Llama 3.1 8B in llama.cpp's table (4.58 GiB GiB). Mixes keep some tensors at higher precision, so other architectures differ by a few percent. Q4_K itself is 4.5 bits (144 bytes per 256 weights); Q4_K_M stores half of the attention V and FFN down tensors, and the output tensor, in Q6_K.",
   "source_url": "https://github.com/ggml-org/llama.cpp/blob/9c2e0e491a822adae1f0b1c831adb4160057d24f/tools/quantize/README.md#L141-L176",
   "source_label": "llama.cpp quantize README"
  },
  {
   "id": "gguf-q5_k_s",
   "label": "Q5_K_S",
   "group": "GGUF (llama.cpp, Ollama, LM Studio)",
   "engine": "llama.cpp",
   "bpw": 5.5704,
   "basis": "Effective bits per weight of a whole Q5_K_S file of Llama 3.1 8B in llama.cpp's table (5.21 GiB GiB). Mixes keep some tensors at higher precision, so other architectures differ by a few percent.",
   "source_url": "https://github.com/ggml-org/llama.cpp/blob/9c2e0e491a822adae1f0b1c831adb4160057d24f/tools/quantize/README.md#L141-L176",
   "source_label": "llama.cpp quantize README"
  },
  {
   "id": "gguf-q5_k_m",
   "label": "Q5_K_M",
   "group": "GGUF (llama.cpp, Ollama, LM Studio)",
   "engine": "llama.cpp",
   "bpw": 5.7036,
   "basis": "Effective bits per weight of a whole Q5_K_M file of Llama 3.1 8B in llama.cpp's table (5.33 GiB GiB). Mixes keep some tensors at higher precision, so other architectures differ by a few percent.",
   "source_url": "https://github.com/ggml-org/llama.cpp/blob/9c2e0e491a822adae1f0b1c831adb4160057d24f/tools/quantize/README.md#L141-L176",
   "source_label": "llama.cpp quantize README"
  },
  {
   "id": "gguf-q6_k",
   "label": "Q6_K",
   "group": "GGUF (llama.cpp, Ollama, LM Studio)",
   "engine": "llama.cpp",
   "bpw": 6.5633,
   "basis": "Effective bits per weight of a whole Q6_K file of Llama 3.1 8B in llama.cpp's table (6.14 GiB GiB). Mixes keep some tensors at higher precision, so other architectures differ by a few percent.",
   "source_url": "https://github.com/ggml-org/llama.cpp/blob/9c2e0e491a822adae1f0b1c831adb4160057d24f/tools/quantize/README.md#L141-L176",
   "source_label": "llama.cpp quantize README"
  },
  {
   "id": "gguf-q8_0",
   "label": "Q8_0",
   "group": "GGUF (llama.cpp, Ollama, LM Studio)",
   "engine": "llama.cpp",
   "bpw": 8.5008,
   "basis": "Effective bits per weight of a whole Q8_0 file of Llama 3.1 8B in llama.cpp's table (7.95 GiB GiB). Mixes keep some tensors at higher precision, so other architectures differ by a few percent. Q8_0 itself is 8.5 bits (34 bytes per 32 weights).",
   "source_url": "https://github.com/ggml-org/llama.cpp/blob/9c2e0e491a822adae1f0b1c831adb4160057d24f/tools/quantize/README.md#L141-L176",
   "source_label": "llama.cpp quantize README"
  },
  {
   "id": "awq-int4",
   "label": "AWQ / GPTQ 4-bit (group size 128)",
   "group": "GPU engines (vLLM, SGLang)",
   "engine": "vllm",
   "vram_precision": "int4",
   "bpw": 4.15625,
   "basis": "4-bit weights plus a 16-bit scale and a 4-bit zero point per group of 128 (4.156 bits); embeddings and LM head kept at 16-bit, as the VRAM calculator counts them. Group size 128 is AutoAWQ's and GPTQModel's default.",
   "source_url": "https://github.com/casper-hansen/AutoAWQ/blob/bcaa8a3689e6ae2e84ec6b57e995ee8a7904a19e/awq/models/_config.py#L11-L13",
   "source_label": "AutoAWQ defaults",
   "source_url_2": "https://github.com/ModelCloud/GPTQModel/blob/8f69d5d4123e4c69091742df6f0995a57f277dbf/gptqmodel/quantization/config.py#L3030-L3031"
  },
  {
   "id": "fp8",
   "label": "FP8 (8-bit float)",
   "group": "GPU engines (vLLM, SGLang)",
   "engine": "vllm",
   "vram_precision": "fp8",
   "bpw": 8,
   "basis": "1 byte per weight; embeddings and LM head kept at 16-bit. Fast only on GPUs with FP8 tensor cores (NVIDIA Ada and newer).",
   "source_url": "https://docs.vllm.ai/en/latest/features/quantization/fp8/",
   "source_label": "vLLM FP8 docs"
  },
  {
   "id": "mxfp4",
   "label": "MXFP4 (how gpt-oss is published)",
   "group": "As published",
   "engine": "llama.cpp or vLLM",
   "vram_precision": "mxfp4",
   "bpw": 4.25,
   "basis": "4-bit values plus one 8-bit shared scale per block of 32 (17 bytes per 32 weights = 4.25 bits); the tensors the checkpoint keeps in BF16 stay at 16-bit.",
   "source_url": "https://github.com/ggml-org/llama.cpp/blob/9c2e0e491a822adae1f0b1c831adb4160057d24f/ggml/src/ggml-common.h#L219",
   "source_label": "ggml block layout"
  },
  {
   "id": "f16",
   "label": "F16 / BF16 (unquantized)",
   "group": "Unquantized",
   "engine": "any",
   "vram_precision": "bf16",
   "bpw": 16,
   "basis": "2 bytes per weight. llama.cpp's table lists F16 at 16.0005 bits for the whole file.",
   "source_url": "https://github.com/ggml-org/llama.cpp/blob/9c2e0e491a822adae1f0b1c831adb4160057d24f/tools/quantize/README.md#L141-L176",
   "source_label": "llama.cpp quantize README"
  }
 ],
 "kv_formats": [
  {
   "id": "f16",
   "label": "F16 (llama.cpp default) / BF16",
   "bytes_per_value": 2,
   "basis": "16-bit keys and values; llama.cpp's default for --cache-type-k and --cache-type-v.",
   "source_url": "https://github.com/ggml-org/llama.cpp/blob/9c2e0e491a822adae1f0b1c831adb4160057d24f/tools/server/README.md#L71",
   "source_label": "llama.cpp server README"
  },
  {
   "id": "q8_0",
   "label": "q8_0 (llama.cpp)",
   "bytes_per_value": 1.0625,
   "basis": "32 int8 values and one 16-bit scale per block: 34 bytes per 32 values. A quantized V cache needs flash attention, which llama.cpp turns on by default (-fa auto).",
   "source_url": "https://github.com/ggml-org/llama.cpp/blob/9c2e0e491a822adae1f0b1c831adb4160057d24f/ggml/src/ggml-common.h#L256",
   "source_label": "ggml block layout",
   "note": "Quantized KV cache (q8_0): llama.cpp needs flash attention for a quantized V cache and enables it by default. The effect on output quality depends on the model; check it on your own prompts."
  },
  {
   "id": "q4_0",
   "label": "q4_0 (llama.cpp)",
   "bytes_per_value": 0.5625,
   "basis": "32 4-bit values and one 16-bit scale per block: 18 bytes per 32 values. Needs flash attention for the V cache, as above.",
   "source_url": "https://github.com/ggml-org/llama.cpp/blob/9c2e0e491a822adae1f0b1c831adb4160057d24f/ggml/src/ggml-common.h#L199",
   "source_label": "ggml block layout",
   "note": "Quantized KV cache (q4_0): a quarter of the F16 cache, with a larger quality risk than q8_0, especially at long context. llama.cpp needs flash attention for a quantized V cache. Check it on your own prompts."
  },
  {
   "id": "fp8",
   "label": "FP8 (vLLM --kv-cache-dtype fp8)",
   "vram_precision": "fp8",
   "bytes_per_value": 1,
   "basis": "1 byte per value; the VRAM calculator's layout, including the FP8 layout vLLM uses for latent-attention (MLA) caches.",
   "source_url": "https://docs.vllm.ai/en/latest/features/quantization/quantized_kvcache/",
   "source_label": "vLLM quantized KV cache docs"
  }
 ],
 "ram_presets": [
  {
   "id": "ddr4-3200-2ch",
   "label": "DDR4-3200, two channels",
   "mts": 3200,
   "channels": 2,
   "gbs": 51.2
  },
  {
   "id": "ddr5-4800-2ch",
   "label": "DDR5-4800, two channels",
   "mts": 4800,
   "channels": 2,
   "gbs": 76.8
  },
  {
   "id": "ddr5-5600-2ch",
   "label": "DDR5-5600, two channels",
   "mts": 5600,
   "channels": 2,
   "gbs": 89.6
  },
  {
   "id": "ddr5-6400-2ch",
   "label": "DDR5-6400, two channels",
   "mts": 6400,
   "channels": 2,
   "gbs": 102.4
  },
  {
   "id": "ddr5-7200-2ch",
   "label": "DDR5-7200, two channels",
   "mts": 7200,
   "channels": 2,
   "gbs": 115.2
  }
 ],
 "ram_bandwidth": {
  "text": "Peak bandwidth is the transfer rate in MT/s times 8 bytes per 64-bit channel times the number of channels: DDR5-5600 on two channels is 89.6 GB/s, the figure Intel states for its Core i9-14900K. Current desktop CPUs from AMD and Intel have two channels (Ryzen 9 9950X up to DDR5-5600; Core Ultra 9 285K up to DDR5-6400; Core Ultra 7 270K Plus up to DDR5-7200). This is a peak: Crucial's table of effective memory bandwidth gives 69.21 GB/s for DDR5-5600, about 77% of it, so real offload speed sits further below the ceiling.",
  "sources": [
   {
    "label": "Crucial: DDR5 bandwidth",
    "url": "https://www.crucial.com/articles/about-memory/everything-about-ddr5-ram"
   },
   {
    "label": "Intel Core i9-14900K specifications",
    "url": "https://www.intel.com/content/www/us/en/products/sku/236773/intel-core-i9-processor-14900k-36m-cache-up-to-6-00-ghz/specifications.html"
   },
   {
    "label": "AMD Ryzen 9 9950X",
    "url": "https://www.amd.com/en/products/processors/desktops/ryzen/9000-series/amd-ryzen-9-9950x.html"
   },
   {
    "label": "Intel Core Ultra 9 285K",
    "url": "https://www.intel.com/content/www/us/en/products/sku/241060/intel-core-ultra-9-processor-285k-36m-cache-up-to-5-70-ghz/specifications.html"
   },
   {
    "label": "Intel Core Ultra 7 270K Plus",
    "url": "https://www.intel.com/content/www/us/en/products/sku/245692/intel-core-ultra-7-processor-270k-plus-36m-cache-up-to-5-50-ghz/specifications.html"
   }
  ]
 },
 "llama_cpp_sources": {
  "input_layer_on_cpu": {
   "text": "\"there is very little benefit to offloading the input layer, so always keep it on the CPU\" (the token embeddings)",
   "url": "https://github.com/ggml-org/llama.cpp/blob/9c2e0e491a822adae1f0b1c831adb4160057d24f/src/llama-model.cpp#L1615-L1617"
  },
  "output_layer_first": {
   "text": "The output layer counts as layer n_layer + 1 and is the first one -ngl sends to the GPU.",
   "url": "https://github.com/ggml-org/llama.cpp/blob/9c2e0e491a822adae1f0b1c831adb4160057d24f/src/llama-model.cpp#L1601-L1626"
  },
  "ngl": {
   "text": "-ngl, --n-gpu-layers: max. number of layers to store in VRAM (default: auto)",
   "url": "https://github.com/ggml-org/llama.cpp/blob/9c2e0e491a822adae1f0b1c831adb4160057d24f/tools/server/README.md#L84"
  },
  "swa_full": {
   "text": "--swa-full: use full-size SWA cache (default: false), so sliding-window layers cache only their window",
   "url": "https://github.com/ggml-org/llama.cpp/blob/9c2e0e491a822adae1f0b1c831adb4160057d24f/tools/server/README.md#L55"
  },
  "quantized_v_needs_fa": {
   "text": "quantized V cache requires flash_attn to be enabled (enabled automatically with -fa auto)",
   "url": "https://github.com/ggml-org/llama.cpp/blob/9c2e0e491a822adae1f0b1c831adb4160057d24f/src/llama-context.cpp#L3962-L3971"
  }
 },
 "assumptions": {
  "overhead_gib": {
   "value": 1,
   "text": "The runtime allowance (1 GiB per GPU by default) stands for llama.cpp's compute buffers and the GPU driver's own context. It is a round assumption, not a measurement: llama.cpp prints its buffers when it loads a model (\"compute buffer size = ... MiB\"), and a GPU that also drives your monitors uses some VRAM for the desktop, so raise it to match.",
   "source_url": "https://github.com/ggml-org/llama.cpp/blob/master/src/llama-context.cpp"
  },
  "ram_reserve_gib": {
   "value": 4,
   "text": "The RAM kept for the operating system and other programs (4 GiB by default) is also an assumption; a desktop with a browser open often uses more, so check your task manager or Activity Monitor."
  },
  "system_ram_gb": {
   "value": 32,
   "text": "Default system RAM in the form; change it to yours."
  },
  "ram_bandwidth_preset": {
   "value": "ddr5-5600-2ch"
  },
  "ram_bandwidth_gbs": {
   "value": 89.6,
   "text": "DDR5-5600 on two channels, the top speed AMD lists for the Ryzen 9 9950X."
  },
  "unified_gpu_share": {
   "value": null,
   "text": "Where the vendor documents no GPU limit on unified memory (DGX Spark, RTX Spark PCs, a custom device), the calculator lets the GPU use all of the memory except the RAM kept for the operating system. Apple's and AMD's limits are in the device notes."
  },
  "rest_of_system_w": {
   "value": {
    "gpu-board": 100,
    "chip": 30,
    "whole-system": 0
   },
   "text": "Rest of the system is an assumption: 100 W for a desktop around a graphics card (CPU, board, memory, drives, fans and power-supply losses), 30 W around an APU whose figure covers only the chip, and 0 W when the vendor's figure is already the whole computer (Apple's desktop Macs, DGX Spark). A plug-in power meter at the wall replaces all of this with a measurement."
  }
 }
}
