{
  "schema_version": 1,
  "as_of": "2026-10-07",
  "description": "Architecture presets for the LLM VRAM calculator. Shapes are copied from each model's config.json on Hugging Face; total parameters are the safetensors metadata count from the Hugging Face API (what the checkpoint actually holds, including any vision encoder or multi-token-prediction layer).",
  "fields": {
    "attention": "KV-cache groups. full: every token cached. sliding: at most `window` tokens cached per layer. mla: one latent of latent_dim values per token per layer, shared by all heads (not split across tensor-parallel GPUs). indexer: fixed bytes per token per layer for a sparse-attention indexer. linear: a fixed recurrent state per sequence that does not grow with context.",
    "params_kept_16bit_when_quantized": "Parameters that common quantized checkpoints leave in 16-bit (embeddings and LM head; for MXFP4 checkpoints, every BF16 tensor).",
    "active_params": "Parameters used per token. Equals params_total for dense models."
  },
  "engine_sources": {
    "gpu_memory_utilization": {
      "value": 0.92,
      "text": "vLLM's default gpu_memory_utilization: the fraction of GPU memory one vLLM instance may use for weights, activations and KV cache.",
      "url": "https://github.com/vllm-project/vllm/blob/main/vllm/config/cache.py"
    },
    "kv_head_replication": {
      "text": "When the tensor-parallel size is at least the number of KV heads, vLLM gives each GPU one KV head and replicates heads across GPUs; otherwise heads are split evenly.",
      "url": "https://github.com/vllm-project/vllm/blob/main/vllm/model_executor/layers/linear.py"
    },
    "mla_fp8_layout": {
      "value": 656,
      "text": "vLLM's fp8_ds_mla KV layout for sparse MLA stores 656 bytes per token per layer: 512 FP8 latent values, 4 FP32 scales and 64 BF16 RoPE values.",
      "url": "https://vllm.ai/blog/2025-09-29-deepseek-v3-2"
    },
    "indexer_layout": {
      "value": 132,
      "text": "The sparse-attention indexer caches index_head_dim (128) FP8 values plus one FP32 scale per 128 values, per token per layer.",
      "url": "https://github.com/vllm-project/vllm/blob/main/vllm/model_executor/models/deepseek_v2.py"
    },
    "linear_state_dtype": {
      "text": "Qwen3.5-family configs set mamba_ssm_dtype to float32, and vLLM's mamba cache dtype 'auto' follows the model config, so the recurrent state is counted at 4 bytes per value and the convolution state at 2.",
      "url": "https://github.com/vllm-project/vllm/blob/main/vllm/config/cache.py"
    }
  },
  "excluded": [
    {
      "model": "deepseek-ai/DeepSeek-V4-Flash and DeepSeek-V4-Pro",
      "reason": "Compressed sparse attention whose cache layout we have not verified against an inference engine's implementation. Planned."
    },
    {
      "model": "Llama 4 Scout / Maverick",
      "reason": "Gated repository and not among the most-downloaded open models on Hugging Face on 2026-10-07."
    }
  ],
  "models": [
    {
      "id": "llama-3.1-8b",
      "name": "Llama 3.1 8B Instruct",
      "publisher": "Meta",
      "hf_repo": "meta-llama/Llama-3.1-8B-Instruct",
      "params_total": 8030261248,
      "active_params": 8030261248,
      "active_params_basis": "dense: every parameter is used for every token",
      "moe": null,
      "layers": 32,
      "hidden_size": 4096,
      "num_attention_heads": 32,
      "num_kv_heads": 8,
      "head_dim": 128,
      "vocab_size": 128256,
      "tie_word_embeddings": false,
      "max_context": 131072,
      "params_kept_16bit_when_quantized": 1050673152,
      "params_kept_16bit_basis": "embeddings + LM head = vocab_size x hidden_size x 2",
      "checkpoint_dtype": "bf16",
      "multimodal": false,
      "attention": [
        {
          "type": "full",
          "layers": 32,
          "kv_heads": 8,
          "head_dim": 128
        }
      ],
      "config_url": "https://huggingface.co/unsloth/Llama-3.1-8B-Instruct/blob/main/config.json",
      "params_url": "https://huggingface.co/api/models/meta-llama/Llama-3.1-8B-Instruct?expand[]=safetensors",
      "card_url": "https://huggingface.co/meta-llama/Llama-3.1-8B-Instruct",
      "as_of": "2026-10-07",
      "note": "",
      "config_note": "meta-llama/Llama-3.1-8B-Instruct is gated on Hugging Face, so its config.json was read from unsloth/Llama-3.1-8B-Instruct, an ungated copy. The parameter count comes from meta-llama/Llama-3.1-8B-Instruct's own public safetensors metadata."
    },
    {
      "id": "llama-3.3-70b",
      "name": "Llama 3.3 70B Instruct",
      "publisher": "Meta",
      "hf_repo": "meta-llama/Llama-3.3-70B-Instruct",
      "params_total": 70553706496,
      "active_params": 70553706496,
      "active_params_basis": "dense: every parameter is used for every token",
      "moe": null,
      "layers": 80,
      "hidden_size": 8192,
      "num_attention_heads": 64,
      "num_kv_heads": 8,
      "head_dim": 128,
      "vocab_size": 128256,
      "tie_word_embeddings": false,
      "max_context": 131072,
      "params_kept_16bit_when_quantized": 2101346304,
      "params_kept_16bit_basis": "embeddings + LM head = vocab_size x hidden_size x 2",
      "checkpoint_dtype": "bf16",
      "multimodal": false,
      "attention": [
        {
          "type": "full",
          "layers": 80,
          "kv_heads": 8,
          "head_dim": 128
        }
      ],
      "config_url": "https://huggingface.co/unsloth/Llama-3.3-70B-Instruct/blob/main/config.json",
      "params_url": "https://huggingface.co/api/models/meta-llama/Llama-3.3-70B-Instruct?expand[]=safetensors",
      "card_url": "https://huggingface.co/meta-llama/Llama-3.3-70B-Instruct",
      "as_of": "2026-10-07",
      "note": "",
      "config_note": "meta-llama/Llama-3.3-70B-Instruct is gated on Hugging Face, so its config.json was read from unsloth/Llama-3.3-70B-Instruct, an ungated copy. The parameter count comes from meta-llama/Llama-3.3-70B-Instruct's own public safetensors metadata."
    },
    {
      "id": "llama-3.1-405b",
      "name": "Llama 3.1 405B Instruct",
      "publisher": "Meta",
      "hf_repo": "meta-llama/Llama-3.1-405B-Instruct",
      "params_total": 405853388800,
      "active_params": 405853388800,
      "active_params_basis": "dense: every parameter is used for every token",
      "moe": null,
      "layers": 126,
      "hidden_size": 16384,
      "num_attention_heads": 128,
      "num_kv_heads": 8,
      "head_dim": 128,
      "vocab_size": 128256,
      "tie_word_embeddings": false,
      "max_context": 131072,
      "params_kept_16bit_when_quantized": 4202692608,
      "params_kept_16bit_basis": "embeddings + LM head = vocab_size x hidden_size x 2",
      "checkpoint_dtype": "bf16",
      "multimodal": false,
      "attention": [
        {
          "type": "full",
          "layers": 126,
          "kv_heads": 8,
          "head_dim": 128
        }
      ],
      "config_url": "https://huggingface.co/unsloth/Meta-Llama-3.1-405B-Instruct-bnb-4bit/blob/main/config.json",
      "params_url": "https://huggingface.co/api/models/meta-llama/Llama-3.1-405B-Instruct?expand[]=safetensors",
      "card_url": "https://huggingface.co/meta-llama/Llama-3.1-405B-Instruct",
      "as_of": "2026-10-07",
      "note": "Architecture fields read from unsloth's ungated copy of the config (the file also carries a bitsandbytes quantization block, which does not change the shapes). head_dim is not stated in this config; it is hidden_size / num_attention_heads = 128.",
      "config_note": "meta-llama/Llama-3.1-405B-Instruct is gated on Hugging Face, so its config.json was read from unsloth/Meta-Llama-3.1-405B-Instruct-bnb-4bit, an ungated copy. The parameter count comes from meta-llama/Llama-3.1-405B-Instruct's own public safetensors metadata."
    },
    {
      "id": "qwen3-8b",
      "name": "Qwen3 8B",
      "publisher": "Alibaba Qwen",
      "hf_repo": "Qwen/Qwen3-8B",
      "params_total": 8190735360,
      "active_params": 8190735360,
      "active_params_basis": "dense: every parameter is used for every token",
      "moe": null,
      "layers": 36,
      "hidden_size": 4096,
      "num_attention_heads": 32,
      "num_kv_heads": 8,
      "head_dim": 128,
      "vocab_size": 151936,
      "tie_word_embeddings": false,
      "max_context": 40960,
      "params_kept_16bit_when_quantized": 1244659712,
      "params_kept_16bit_basis": "embeddings + LM head = vocab_size x hidden_size x 2",
      "checkpoint_dtype": "bf16",
      "multimodal": false,
      "attention": [
        {
          "type": "full",
          "layers": 36,
          "kv_heads": 8,
          "head_dim": 128
        }
      ],
      "config_url": "https://huggingface.co/Qwen/Qwen3-8B/blob/main/config.json",
      "params_url": "https://huggingface.co/api/models/Qwen/Qwen3-8B?expand[]=safetensors",
      "card_url": "https://huggingface.co/Qwen/Qwen3-8B",
      "as_of": "2026-10-07",
      "note": ""
    },
    {
      "id": "qwen3-14b",
      "name": "Qwen3 14B",
      "publisher": "Alibaba Qwen",
      "hf_repo": "Qwen/Qwen3-14B",
      "params_total": 14768307200,
      "active_params": 14768307200,
      "active_params_basis": "dense: every parameter is used for every token",
      "moe": null,
      "layers": 40,
      "hidden_size": 5120,
      "num_attention_heads": 40,
      "num_kv_heads": 8,
      "head_dim": 128,
      "vocab_size": 151936,
      "tie_word_embeddings": false,
      "max_context": 40960,
      "params_kept_16bit_when_quantized": 1555824640,
      "params_kept_16bit_basis": "embeddings + LM head = vocab_size x hidden_size x 2",
      "checkpoint_dtype": "bf16",
      "multimodal": false,
      "attention": [
        {
          "type": "full",
          "layers": 40,
          "kv_heads": 8,
          "head_dim": 128
        }
      ],
      "config_url": "https://huggingface.co/Qwen/Qwen3-14B/blob/main/config.json",
      "params_url": "https://huggingface.co/api/models/Qwen/Qwen3-14B?expand[]=safetensors",
      "card_url": "https://huggingface.co/Qwen/Qwen3-14B",
      "as_of": "2026-10-07",
      "note": ""
    },
    {
      "id": "qwen3-32b",
      "name": "Qwen3 32B",
      "publisher": "Alibaba Qwen",
      "hf_repo": "Qwen/Qwen3-32B",
      "params_total": 32762123264,
      "active_params": 32762123264,
      "active_params_basis": "dense: every parameter is used for every token",
      "moe": null,
      "layers": 64,
      "hidden_size": 5120,
      "num_attention_heads": 64,
      "num_kv_heads": 8,
      "head_dim": 128,
      "vocab_size": 151936,
      "tie_word_embeddings": false,
      "max_context": 40960,
      "params_kept_16bit_when_quantized": 1555824640,
      "params_kept_16bit_basis": "embeddings + LM head = vocab_size x hidden_size x 2",
      "checkpoint_dtype": "bf16",
      "multimodal": false,
      "attention": [
        {
          "type": "full",
          "layers": 64,
          "kv_heads": 8,
          "head_dim": 128
        }
      ],
      "config_url": "https://huggingface.co/Qwen/Qwen3-32B/blob/main/config.json",
      "params_url": "https://huggingface.co/api/models/Qwen/Qwen3-32B?expand[]=safetensors",
      "card_url": "https://huggingface.co/Qwen/Qwen3-32B",
      "as_of": "2026-10-07",
      "note": ""
    },
    {
      "id": "qwen2.5-7b",
      "name": "Qwen2.5 7B Instruct",
      "publisher": "Alibaba Qwen",
      "hf_repo": "Qwen/Qwen2.5-7B-Instruct",
      "params_total": 7615616512,
      "active_params": 7615616512,
      "active_params_basis": "dense: every parameter is used for every token",
      "moe": null,
      "layers": 28,
      "hidden_size": 3584,
      "num_attention_heads": 28,
      "num_kv_heads": 4,
      "head_dim": 128,
      "vocab_size": 152064,
      "tie_word_embeddings": false,
      "max_context": 32768,
      "params_kept_16bit_when_quantized": 1089994752,
      "params_kept_16bit_basis": "embeddings + LM head = vocab_size x hidden_size x 2",
      "checkpoint_dtype": "bf16",
      "multimodal": false,
      "attention": [
        {
          "type": "full",
          "layers": 28,
          "kv_heads": 4,
          "head_dim": 128
        }
      ],
      "config_url": "https://huggingface.co/Qwen/Qwen2.5-7B-Instruct/blob/main/config.json",
      "params_url": "https://huggingface.co/api/models/Qwen/Qwen2.5-7B-Instruct?expand[]=safetensors",
      "card_url": "https://huggingface.co/Qwen/Qwen2.5-7B-Instruct",
      "as_of": "2026-10-07",
      "note": "config.json sets sliding_window but use_sliding_window is false, so every layer is full attention."
    },
    {
      "id": "qwen2.5-72b",
      "name": "Qwen2.5 72B Instruct",
      "publisher": "Alibaba Qwen",
      "hf_repo": "Qwen/Qwen2.5-72B-Instruct",
      "params_total": 72706203648,
      "active_params": 72706203648,
      "active_params_basis": "dense: every parameter is used for every token",
      "moe": null,
      "layers": 80,
      "hidden_size": 8192,
      "num_attention_heads": 64,
      "num_kv_heads": 8,
      "head_dim": 128,
      "vocab_size": 152064,
      "tie_word_embeddings": false,
      "max_context": 32768,
      "params_kept_16bit_when_quantized": 2491416576,
      "params_kept_16bit_basis": "embeddings + LM head = vocab_size x hidden_size x 2",
      "checkpoint_dtype": "bf16",
      "multimodal": false,
      "attention": [
        {
          "type": "full",
          "layers": 80,
          "kv_heads": 8,
          "head_dim": 128
        }
      ],
      "config_url": "https://huggingface.co/Qwen/Qwen2.5-72B-Instruct/blob/main/config.json",
      "params_url": "https://huggingface.co/api/models/Qwen/Qwen2.5-72B-Instruct?expand[]=safetensors",
      "card_url": "https://huggingface.co/Qwen/Qwen2.5-72B-Instruct",
      "as_of": "2026-10-07",
      "note": "config.json sets sliding_window but use_sliding_window is false, so every layer is full attention."
    },
    {
      "id": "qwen3.8-27b",
      "name": "Qwen3.8 27B",
      "publisher": "Alibaba Qwen",
      "hf_repo": "Qwen/Qwen3.8-27B",
      "params_total": 27781427952,
      "active_params": 27781427952,
      "active_params_basis": "dense: every parameter is used for every token",
      "moe": null,
      "layers": 64,
      "hidden_size": 5120,
      "num_attention_heads": 24,
      "num_kv_heads": 4,
      "head_dim": 256,
      "vocab_size": 248320,
      "tie_word_embeddings": false,
      "max_context": 262144,
      "params_kept_16bit_when_quantized": 2542796800,
      "params_kept_16bit_basis": "embeddings + LM head = vocab_size x hidden_size x 2",
      "checkpoint_dtype": "bf16",
      "multimodal": true,
      "attention": [
        {
          "type": "full",
          "layers": 16,
          "kv_heads": 4,
          "head_dim": 256
        },
        {
          "type": "linear",
          "layers": 48,
          "value_heads": 48,
          "key_heads": 16,
          "key_head_dim": 128,
          "value_head_dim": 128,
          "conv_kernel": 4,
          "state_bytes": 4,
          "conv_bytes": 2
        }
      ],
      "config_url": "https://huggingface.co/Qwen/Qwen3.8-27B/blob/main/config.json",
      "params_url": "https://huggingface.co/api/models/Qwen/Qwen3.8-27B?expand[]=safetensors",
      "card_url": "https://huggingface.co/Qwen/Qwen3.8-27B",
      "as_of": "2026-10-07",
      "note": "Hybrid attention: 3 of every 4 layers are Gated DeltaNet linear attention with a fixed-size state per sequence; only the full-attention layers keep a KV cache. Parameter count includes the vision encoder."
    },
    {
      "id": "mistral-7b-v0.3",
      "name": "Mistral 7B Instruct v0.3",
      "publisher": "Mistral AI",
      "hf_repo": "mistralai/Mistral-7B-Instruct-v0.3",
      "params_total": 7248023552,
      "active_params": 7248023552,
      "active_params_basis": "dense: every parameter is used for every token",
      "moe": null,
      "layers": 32,
      "hidden_size": 4096,
      "num_attention_heads": 32,
      "num_kv_heads": 8,
      "head_dim": 128,
      "vocab_size": 32768,
      "tie_word_embeddings": false,
      "max_context": 32768,
      "params_kept_16bit_when_quantized": 268435456,
      "params_kept_16bit_basis": "embeddings + LM head = vocab_size x hidden_size x 2",
      "checkpoint_dtype": "bf16",
      "multimodal": false,
      "attention": [
        {
          "type": "full",
          "layers": 32,
          "kv_heads": 8,
          "head_dim": 128
        }
      ],
      "config_url": "https://huggingface.co/mistralai/Mistral-7B-Instruct-v0.3/blob/main/config.json",
      "params_url": "https://huggingface.co/api/models/mistralai/Mistral-7B-Instruct-v0.3?expand[]=safetensors",
      "card_url": "https://huggingface.co/mistralai/Mistral-7B-Instruct-v0.3",
      "as_of": "2026-10-07",
      "note": ""
    },
    {
      "id": "mistral-small-3.2-24b",
      "name": "Mistral Small 3.2 24B Instruct (2506)",
      "publisher": "Mistral AI",
      "hf_repo": "mistralai/Mistral-Small-3.2-24B-Instruct-2506",
      "params_total": 24011361280,
      "active_params": 24011361280,
      "active_params_basis": "dense: every parameter is used for every token",
      "moe": null,
      "layers": 40,
      "hidden_size": 5120,
      "num_attention_heads": 32,
      "num_kv_heads": 8,
      "head_dim": 128,
      "vocab_size": 131072,
      "tie_word_embeddings": false,
      "max_context": 131072,
      "params_kept_16bit_when_quantized": 1342177280,
      "params_kept_16bit_basis": "embeddings + LM head = vocab_size x hidden_size x 2",
      "checkpoint_dtype": "bf16",
      "multimodal": true,
      "attention": [
        {
          "type": "full",
          "layers": 40,
          "kv_heads": 8,
          "head_dim": 128
        }
      ],
      "config_url": "https://huggingface.co/mistralai/Mistral-Small-3.2-24B-Instruct-2506/blob/main/config.json",
      "params_url": "https://huggingface.co/api/models/mistralai/Mistral-Small-3.2-24B-Instruct-2506?expand[]=safetensors",
      "card_url": "https://huggingface.co/mistralai/Mistral-Small-3.2-24B-Instruct-2506",
      "as_of": "2026-10-07",
      "note": "Parameter count includes the vision encoder."
    },
    {
      "id": "gemma-4-31b",
      "name": "Gemma 4 31B IT",
      "publisher": "Google",
      "hf_repo": "google/gemma-4-31B-it",
      "params_total": 31273088876,
      "active_params": 31273088876,
      "active_params_basis": "dense: every parameter is used for every token",
      "moe": null,
      "layers": 60,
      "hidden_size": 5376,
      "num_attention_heads": 32,
      "num_kv_heads": 16,
      "head_dim": 256,
      "vocab_size": 262144,
      "tie_word_embeddings": true,
      "max_context": 262144,
      "params_kept_16bit_when_quantized": 1409286144,
      "params_kept_16bit_basis": "embeddings + LM head = vocab_size x hidden_size x 1 (tied)",
      "checkpoint_dtype": "bf16",
      "multimodal": true,
      "attention": [
        {
          "type": "sliding",
          "layers": 50,
          "kv_heads": 16,
          "head_dim": 256,
          "window": 1024
        },
        {
          "type": "full",
          "layers": 10,
          "kv_heads": 4,
          "head_dim": 512
        }
      ],
      "config_url": "https://huggingface.co/google/gemma-4-31B-it/blob/main/config.json",
      "params_url": "https://huggingface.co/api/models/google/gemma-4-31B-it?expand[]=safetensors",
      "card_url": "https://huggingface.co/google/gemma-4-31B-it",
      "as_of": "2026-10-07",
      "note": "5 of every 6 layers use 1,024-token sliding-window attention; the global layers use 4 KV heads of dimension 512 with keys and values from one projection (attention_k_eq_v). We count both K and V as cached, the conservative reading. Parameter count includes the vision encoder."
    },
    {
      "id": "qwen3-30b-a3b",
      "name": "Qwen3 30B-A3B Instruct (2507)",
      "publisher": "Alibaba Qwen",
      "hf_repo": "Qwen/Qwen3-30B-A3B-Instruct-2507",
      "params_total": 30532122624,
      "active_params": 3300000000,
      "active_params_basis": "model card: '30.5B in total and 3.3B activated'",
      "moe": {
        "experts": 128,
        "experts_per_token": 8
      },
      "layers": 48,
      "hidden_size": 2048,
      "num_attention_heads": 32,
      "num_kv_heads": 4,
      "head_dim": 128,
      "vocab_size": 151936,
      "tie_word_embeddings": false,
      "max_context": 262144,
      "params_kept_16bit_when_quantized": 622329856,
      "params_kept_16bit_basis": "embeddings + LM head = vocab_size x hidden_size x 2",
      "checkpoint_dtype": "bf16",
      "multimodal": false,
      "attention": [
        {
          "type": "full",
          "layers": 48,
          "kv_heads": 4,
          "head_dim": 128
        }
      ],
      "config_url": "https://huggingface.co/Qwen/Qwen3-30B-A3B-Instruct-2507/blob/main/config.json",
      "params_url": "https://huggingface.co/api/models/Qwen/Qwen3-30B-A3B-Instruct-2507?expand[]=safetensors",
      "card_url": "https://huggingface.co/Qwen/Qwen3-30B-A3B-Instruct-2507",
      "as_of": "2026-10-07",
      "note": ""
    },
    {
      "id": "qwen3-235b-a22b",
      "name": "Qwen3 235B-A22B Instruct (2507)",
      "publisher": "Alibaba Qwen",
      "hf_repo": "Qwen/Qwen3-235B-A22B-Instruct-2507",
      "params_total": 235093634560,
      "active_params": 22000000000,
      "active_params_basis": "model card: '235B in total and 22B activated'",
      "moe": {
        "experts": 128,
        "experts_per_token": 8
      },
      "layers": 94,
      "hidden_size": 4096,
      "num_attention_heads": 64,
      "num_kv_heads": 4,
      "head_dim": 128,
      "vocab_size": 151936,
      "tie_word_embeddings": false,
      "max_context": 262144,
      "params_kept_16bit_when_quantized": 1244659712,
      "params_kept_16bit_basis": "embeddings + LM head = vocab_size x hidden_size x 2",
      "checkpoint_dtype": "bf16",
      "multimodal": false,
      "attention": [
        {
          "type": "full",
          "layers": 94,
          "kv_heads": 4,
          "head_dim": 128
        }
      ],
      "config_url": "https://huggingface.co/Qwen/Qwen3-235B-A22B-Instruct-2507/blob/main/config.json",
      "params_url": "https://huggingface.co/api/models/Qwen/Qwen3-235B-A22B-Instruct-2507?expand[]=safetensors",
      "card_url": "https://huggingface.co/Qwen/Qwen3-235B-A22B-Instruct-2507",
      "as_of": "2026-10-07",
      "note": ""
    },
    {
      "id": "qwen3.6-35b-a3b",
      "name": "Qwen3.6 35B-A3B",
      "publisher": "Alibaba Qwen",
      "hf_repo": "Qwen/Qwen3.6-35B-A3B",
      "params_total": 35951822704,
      "active_params": 3000000000,
      "active_params_basis": "model card: '35B in total and 3B activated'",
      "moe": {
        "experts": 256,
        "experts_per_token": 8
      },
      "layers": 40,
      "hidden_size": 2048,
      "num_attention_heads": 16,
      "num_kv_heads": 2,
      "head_dim": 256,
      "vocab_size": 248320,
      "tie_word_embeddings": false,
      "max_context": 262144,
      "params_kept_16bit_when_quantized": 1017118720,
      "params_kept_16bit_basis": "embeddings + LM head = vocab_size x hidden_size x 2",
      "checkpoint_dtype": "bf16",
      "multimodal": true,
      "attention": [
        {
          "type": "full",
          "layers": 10,
          "kv_heads": 2,
          "head_dim": 256
        },
        {
          "type": "linear",
          "layers": 30,
          "value_heads": 32,
          "key_heads": 16,
          "key_head_dim": 128,
          "value_head_dim": 128,
          "conv_kernel": 4,
          "state_bytes": 4,
          "conv_bytes": 2
        }
      ],
      "config_url": "https://huggingface.co/Qwen/Qwen3.6-35B-A3B/blob/main/config.json",
      "params_url": "https://huggingface.co/api/models/Qwen/Qwen3.6-35B-A3B?expand[]=safetensors",
      "card_url": "https://huggingface.co/Qwen/Qwen3.6-35B-A3B",
      "as_of": "2026-10-07",
      "note": "Hybrid attention (Gated DeltaNet + full attention every 4th layer). Parameter count includes the vision encoder."
    },
    {
      "id": "qwen3.5-122b-a10b",
      "name": "Qwen3.5 122B-A10B",
      "publisher": "Alibaba Qwen",
      "hf_repo": "Qwen/Qwen3.5-122B-A10B",
      "params_total": 125086497008,
      "active_params": 10000000000,
      "active_params_basis": "model card: '122B in total and 10B activated'",
      "moe": {
        "experts": 256,
        "experts_per_token": 8
      },
      "layers": 48,
      "hidden_size": 3072,
      "num_attention_heads": 32,
      "num_kv_heads": 2,
      "head_dim": 256,
      "vocab_size": 248320,
      "tie_word_embeddings": false,
      "max_context": 262144,
      "params_kept_16bit_when_quantized": 1525678080,
      "params_kept_16bit_basis": "embeddings + LM head = vocab_size x hidden_size x 2",
      "checkpoint_dtype": "bf16",
      "multimodal": true,
      "attention": [
        {
          "type": "full",
          "layers": 12,
          "kv_heads": 2,
          "head_dim": 256
        },
        {
          "type": "linear",
          "layers": 36,
          "value_heads": 64,
          "key_heads": 16,
          "key_head_dim": 128,
          "value_head_dim": 128,
          "conv_kernel": 4,
          "state_bytes": 4,
          "conv_bytes": 2
        }
      ],
      "config_url": "https://huggingface.co/Qwen/Qwen3.5-122B-A10B/blob/main/config.json",
      "params_url": "https://huggingface.co/api/models/Qwen/Qwen3.5-122B-A10B?expand[]=safetensors",
      "card_url": "https://huggingface.co/Qwen/Qwen3.5-122B-A10B",
      "as_of": "2026-10-07",
      "note": "Hybrid attention (Gated DeltaNet + full attention every 4th layer). Parameter count includes the vision encoder."
    },
    {
      "id": "gemma-4-26b-a4b",
      "name": "Gemma 4 26B-A4B IT",
      "publisher": "Google",
      "hf_repo": "google/gemma-4-26B-A4B-it",
      "params_total": 25805936206,
      "active_params": 3800000000,
      "active_params_basis": "model card: 'Active Parameters 3.8B'",
      "moe": {
        "experts": 128,
        "experts_per_token": 8
      },
      "layers": 30,
      "hidden_size": 2816,
      "num_attention_heads": 16,
      "num_kv_heads": 8,
      "head_dim": 256,
      "vocab_size": 262144,
      "tie_word_embeddings": true,
      "max_context": 262144,
      "params_kept_16bit_when_quantized": 738197504,
      "params_kept_16bit_basis": "embeddings + LM head = vocab_size x hidden_size x 1 (tied)",
      "checkpoint_dtype": "bf16",
      "multimodal": true,
      "attention": [
        {
          "type": "sliding",
          "layers": 25,
          "kv_heads": 8,
          "head_dim": 256,
          "window": 1024
        },
        {
          "type": "full",
          "layers": 5,
          "kv_heads": 2,
          "head_dim": 512
        }
      ],
      "config_url": "https://huggingface.co/google/gemma-4-26B-A4B-it/blob/main/config.json",
      "params_url": "https://huggingface.co/api/models/google/gemma-4-26B-A4B-it?expand[]=safetensors",
      "card_url": "https://huggingface.co/google/gemma-4-26B-A4B-it",
      "as_of": "2026-10-07",
      "note": "Sliding-window (1,024 tokens) on 5 of every 6 layers; global layers use 2 KV heads of dimension 512 (attention_k_eq_v; both K and V counted). Parameter count includes the vision encoder."
    },
    {
      "id": "gpt-oss-20b",
      "name": "gpt-oss-20b",
      "publisher": "OpenAI",
      "hf_repo": "openai/gpt-oss-20b",
      "params_total": 20914757184,
      "active_params": 3600000000,
      "active_params_basis": "model card: '21B parameters with 3.6B active parameters'",
      "moe": {
        "experts": 32,
        "experts_per_token": 4
      },
      "layers": 24,
      "hidden_size": 2880,
      "num_attention_heads": 64,
      "num_kv_heads": 8,
      "head_dim": 64,
      "vocab_size": 201088,
      "tie_word_embeddings": false,
      "max_context": 131072,
      "params_kept_16bit_when_quantized": 1804459584,
      "params_kept_16bit_basis": "BF16 tensors in the published MXFP4 checkpoint (attention, router, embeddings, LM head)",
      "checkpoint_dtype": "mxfp4",
      "multimodal": false,
      "attention": [
        {
          "type": "sliding",
          "layers": 12,
          "kv_heads": 8,
          "head_dim": 64,
          "window": 128
        },
        {
          "type": "full",
          "layers": 12,
          "kv_heads": 8,
          "head_dim": 64
        }
      ],
      "config_url": "https://huggingface.co/openai/gpt-oss-20b/blob/main/config.json",
      "params_url": "https://huggingface.co/api/models/openai/gpt-oss-20b?expand[]=safetensors",
      "card_url": "https://huggingface.co/openai/gpt-oss-20b",
      "as_of": "2026-10-07",
      "note": "Published with MoE weights in MXFP4; attention, router, embeddings and LM head stay BF16. Alternating 128-token sliding-window and full-attention layers."
    },
    {
      "id": "gpt-oss-120b",
      "name": "gpt-oss-120b",
      "publisher": "OpenAI",
      "hf_repo": "openai/gpt-oss-120b",
      "params_total": 116829156672,
      "active_params": 5100000000,
      "active_params_basis": "model card: '117B parameters with 5.1B active parameters'",
      "moe": {
        "experts": 128,
        "experts_per_token": 4
      },
      "layers": 36,
      "hidden_size": 2880,
      "num_attention_heads": 64,
      "num_kv_heads": 8,
      "head_dim": 64,
      "vocab_size": 201088,
      "tie_word_embeddings": false,
      "max_context": 131072,
      "params_kept_16bit_when_quantized": 2167371072,
      "params_kept_16bit_basis": "BF16 tensors in the published MXFP4 checkpoint (attention, router, embeddings, LM head)",
      "checkpoint_dtype": "mxfp4",
      "multimodal": false,
      "attention": [
        {
          "type": "sliding",
          "layers": 18,
          "kv_heads": 8,
          "head_dim": 64,
          "window": 128
        },
        {
          "type": "full",
          "layers": 18,
          "kv_heads": 8,
          "head_dim": 64
        }
      ],
      "config_url": "https://huggingface.co/openai/gpt-oss-120b/blob/main/config.json",
      "params_url": "https://huggingface.co/api/models/openai/gpt-oss-120b?expand[]=safetensors",
      "card_url": "https://huggingface.co/openai/gpt-oss-120b",
      "as_of": "2026-10-07",
      "note": "Published with MoE weights in MXFP4; attention, router, embeddings and LM head stay BF16. Alternating 128-token sliding-window and full-attention layers."
    },
    {
      "id": "glm-4.5-air",
      "name": "GLM-4.5-Air",
      "publisher": "Z.ai",
      "hf_repo": "zai-org/GLM-4.5-Air",
      "params_total": 110468824832,
      "active_params": 12000000000,
      "active_params_basis": "model card: '106 billion total parameters and 12 billion active parameters'",
      "moe": {
        "experts": 128,
        "experts_per_token": 8
      },
      "layers": 46,
      "hidden_size": 4096,
      "num_attention_heads": 96,
      "num_kv_heads": 8,
      "head_dim": 128,
      "vocab_size": 151552,
      "tie_word_embeddings": false,
      "max_context": 131072,
      "params_kept_16bit_when_quantized": 1241513984,
      "params_kept_16bit_basis": "embeddings + LM head = vocab_size x hidden_size x 2",
      "checkpoint_dtype": "bf16",
      "multimodal": false,
      "attention": [
        {
          "type": "full",
          "layers": 46,
          "kv_heads": 8,
          "head_dim": 128
        }
      ],
      "config_url": "https://huggingface.co/zai-org/GLM-4.5-Air/blob/main/config.json",
      "params_url": "https://huggingface.co/api/models/zai-org/GLM-4.5-Air?expand[]=safetensors",
      "card_url": "https://huggingface.co/zai-org/GLM-4.5-Air",
      "as_of": "2026-10-07",
      "note": "Parameter count from the checkpoint (110.5B) includes the multi-token-prediction layer; the model card's 106B does not."
    },
    {
      "id": "minimax-m2.7",
      "name": "MiniMax-M2.7",
      "publisher": "MiniMax",
      "hf_repo": "MiniMaxAI/MiniMax-M2.7",
      "params_total": 228689764864,
      "active_params": 11030553088,
      "active_params_basis": "computed from config.json: total parameters minus the routed experts a token does not use, (experts - experts_per_token) x 3 x hidden_size x moe_intermediate_size per MoE layer; the publisher's card does not state it. Includes embeddings.",
      "moe": {
        "experts": 256,
        "experts_per_token": 8
      },
      "layers": 62,
      "hidden_size": 3072,
      "num_attention_heads": 48,
      "num_kv_heads": 8,
      "head_dim": 128,
      "vocab_size": 200064,
      "tie_word_embeddings": false,
      "max_context": 204800,
      "params_kept_16bit_when_quantized": 1229193216,
      "params_kept_16bit_basis": "embeddings + LM head = vocab_size x hidden_size x 2",
      "checkpoint_dtype": "fp8",
      "multimodal": false,
      "attention": [
        {
          "type": "full",
          "layers": 62,
          "kv_heads": 8,
          "head_dim": 128
        }
      ],
      "config_url": "https://huggingface.co/MiniMaxAI/MiniMax-M2.7/blob/main/config.json",
      "params_url": "https://huggingface.co/api/models/MiniMaxAI/MiniMax-M2.7?expand[]=safetensors",
      "card_url": "https://huggingface.co/MiniMaxAI/MiniMax-M2.7",
      "as_of": "2026-10-07",
      "note": "Published as an FP8 (block-scaled) checkpoint."
    },
    {
      "id": "deepseek-v3.2",
      "name": "DeepSeek-V3.2",
      "publisher": "DeepSeek",
      "hf_repo": "deepseek-ai/DeepSeek-V3.2",
      "params_total": 685355329792,
      "active_params": 40959240448,
      "active_params_basis": "computed from config.json: total parameters minus the routed experts a token does not use, (experts - experts_per_token) x 3 x hidden_size x moe_intermediate_size per MoE layer; the publisher's card does not state it. Includes embeddings and the MTP layer.",
      "moe": {
        "experts": 256,
        "experts_per_token": 8
      },
      "layers": 61,
      "hidden_size": 7168,
      "num_attention_heads": 128,
      "num_kv_heads": 128,
      "head_dim": 56,
      "vocab_size": 129280,
      "tie_word_embeddings": false,
      "max_context": 163840,
      "params_kept_16bit_when_quantized": 1853358080,
      "params_kept_16bit_basis": "embeddings + LM head = vocab_size x hidden_size x 2",
      "checkpoint_dtype": "fp8",
      "multimodal": false,
      "attention": [
        {
          "type": "mla",
          "layers": 61,
          "latent_dim": 576,
          "fp8_bytes_per_token_layer": 656
        },
        {
          "type": "indexer",
          "layers": 61,
          "bytes_per_token_layer": 132
        }
      ],
      "config_url": "https://huggingface.co/deepseek-ai/DeepSeek-V3.2/blob/main/config.json",
      "params_url": "https://huggingface.co/api/models/deepseek-ai/DeepSeek-V3.2?expand[]=safetensors",
      "card_url": "https://huggingface.co/deepseek-ai/DeepSeek-V3.2",
      "as_of": "2026-10-07",
      "note": "Multi-head latent attention (MLA) caches one 576-dim latent per token per layer, shared by all heads, plus a small FP8 key for the sparse-attention indexer. Published as FP8. Parameter count includes the 1-layer multi-token-prediction module, which is loaded only for speculative decoding."
    },
    {
      "id": "glm-5.3",
      "name": "GLM-5.3",
      "publisher": "Z.ai",
      "hf_repo": "zai-org/GLM-5.3",
      "params_total": 753329940480,
      "active_params": 41841764352,
      "active_params_basis": "computed from config.json: total parameters minus the routed experts a token does not use, (experts - experts_per_token) x 3 x hidden_size x moe_intermediate_size per MoE layer; the publisher's card does not state it. Includes embeddings and the MTP layer.",
      "moe": {
        "experts": 256,
        "experts_per_token": 8
      },
      "layers": 78,
      "hidden_size": 6144,
      "num_attention_heads": 64,
      "num_kv_heads": 64,
      "head_dim": 192,
      "vocab_size": 154880,
      "tie_word_embeddings": false,
      "max_context": 1048576,
      "params_kept_16bit_when_quantized": 1903165440,
      "params_kept_16bit_basis": "embeddings + LM head = vocab_size x hidden_size x 2",
      "checkpoint_dtype": "fp8",
      "multimodal": false,
      "attention": [
        {
          "type": "mla",
          "layers": 78,
          "latent_dim": 576,
          "fp8_bytes_per_token_layer": 656
        },
        {
          "type": "indexer",
          "layers": 21,
          "bytes_per_token_layer": 132
        }
      ],
      "config_url": "https://huggingface.co/zai-org/GLM-5.3/blob/main/config.json",
      "params_url": "https://huggingface.co/api/models/zai-org/GLM-5.3?expand[]=safetensors",
      "card_url": "https://huggingface.co/zai-org/GLM-5.3",
      "as_of": "2026-10-07",
      "note": "DeepSeek-style MLA with a sparse-attention indexer; config.json marks 21 of 78 layers as having their own indexer ('full') and 57 as reusing one ('shared'), so the indexer cache is counted on 21 layers. FP8 KV layout assumed to match DeepSeek-V3.2's in vLLM. Parameter count includes the MTP layer."
    },
    {
      "id": "kimi-k2-0905",
      "name": "Kimi K2 Instruct (0905)",
      "publisher": "Moonshot AI",
      "hf_repo": "moonshotai/Kimi-K2-Instruct-0905",
      "params_total": 1026470735448,
      "active_params": 32000000000,
      "active_params_basis": "model card: '32 billion activated parameters and a total of 1 trillion parameters'",
      "moe": {
        "experts": 384,
        "experts_per_token": 8
      },
      "layers": 61,
      "hidden_size": 7168,
      "num_attention_heads": 64,
      "num_kv_heads": 64,
      "head_dim": 112,
      "vocab_size": 163840,
      "tie_word_embeddings": false,
      "max_context": 262144,
      "params_kept_16bit_when_quantized": 2348810240,
      "params_kept_16bit_basis": "embeddings + LM head = vocab_size x hidden_size x 2",
      "checkpoint_dtype": "fp8",
      "multimodal": false,
      "attention": [
        {
          "type": "mla",
          "layers": 61,
          "latent_dim": 576
        }
      ],
      "config_url": "https://huggingface.co/moonshotai/Kimi-K2-Instruct-0905/blob/main/config.json",
      "params_url": "https://huggingface.co/api/models/moonshotai/Kimi-K2-Instruct-0905?expand[]=safetensors",
      "card_url": "https://huggingface.co/moonshotai/Kimi-K2-Instruct-0905",
      "as_of": "2026-10-07",
      "note": "DeepSeek-V3 architecture (MLA). Published as FP8."
    }
  ]
}
