{
  "name": "AIPCs local model memory catalog",
  "license": "CC BY 4.0 (https://creativecommons.org/licenses/by/4.0/)",
  "attribution": "AIPCs (https://www.aipcs.store/tools/local-llm-calculator/)",
  "fields": {
    "fileGb": "download size, GB (1e9 bytes)",
    "activeGb": "bytes read per generated token, GB",
    "kvKibPerToken": "KV cache per token of context at 16-bit, KiB",
    "kvFixedMib": "fixed KV cache for sliding-window layers, MiB"
  },
  "generated": "2026-09-25",
  "models": [
    {
      "id": "llama-3.1-8b",
      "name": "Llama 3.1 8B",
      "quant": "Q4_K_M",
      "paramsB": 8,
      "arch": "llama",
      "moe": false,
      "fileGb": 4.92,
      "activeGb": 4.62,
      "sharedGb": 4.62,
      "expertGb": 0,
      "kvKibPerToken": 128,
      "kvFixedMib": 0,
      "fullAttnLayers": 32,
      "layers": 32,
      "contextMax": 131072,
      "source": "https://huggingface.co/bartowski/Meta-Llama-3.1-8B-Instruct-GGUF",
      "files": [
        "Meta-Llama-3.1-8B-Instruct-Q4_K_M.gguf"
      ]
    },
    {
      "id": "gemma-4-e4b",
      "name": "Gemma 4 E4B",
      "quant": "Q4_0",
      "paramsB": 7.5,
      "arch": "gemma4",
      "moe": false,
      "fileGb": 4.59,
      "activeGb": 2.28,
      "sharedGb": 2.28,
      "expertGb": 0,
      "kvKibPerToken": 16,
      "kvFixedMib": 20,
      "fullAttnLayers": 4,
      "layers": 42,
      "contextMax": 131072,
      "speedFactor": 0.7,
      "tests": [
        "T0001",
        "T0002"
      ],
      "source": "https://huggingface.co/ggml-org/gemma-4-E4B-it-GGUF",
      "files": [
        "gemma-4-E4B-it-Q4_0.gguf"
      ]
    },
    {
      "id": "qwen3.5-9b",
      "name": "Qwen3.5 9B",
      "quant": "Q4_K_M",
      "paramsB": 9,
      "arch": "qwen35",
      "moe": false,
      "fileGb": 5.68,
      "activeGb": 5.1,
      "sharedGb": 5.1,
      "expertGb": 0,
      "kvKibPerToken": 32,
      "kvFixedMib": 0,
      "fullAttnLayers": 8,
      "layers": 32,
      "contextMax": 262144,
      "source": "https://huggingface.co/unsloth/Qwen3.5-9B-GGUF",
      "files": [
        "Qwen3.5-9B-Q4_K_M.gguf"
      ]
    },
    {
      "id": "gemma-4-12b",
      "name": "Gemma 4 12B",
      "quant": "Q4_K_M",
      "paramsB": 12,
      "arch": "gemma4",
      "moe": false,
      "fileGb": 7.12,
      "activeGb": 6.54,
      "sharedGb": 6.54,
      "expertGb": 0,
      "kvKibPerToken": 16,
      "kvFixedMib": 320,
      "fullAttnLayers": 8,
      "layers": 48,
      "contextMax": 262144,
      "source": "https://huggingface.co/unsloth/gemma-4-12b-it-GGUF",
      "files": [
        "gemma-4-12b-it-Q4_K_M.gguf"
      ]
    },
    {
      "id": "gpt-oss-20b",
      "name": "gpt-oss 20B",
      "quant": "MXFP4",
      "paramsB": 21,
      "arch": "gpt-oss",
      "moe": true,
      "experts": 32,
      "expertsUsed": 4,
      "fileGb": 12.11,
      "activeGb": 2.57,
      "sharedGb": 1.3,
      "expertGb": 10.18,
      "kvKibPerToken": 24,
      "kvFixedMib": 3,
      "fullAttnLayers": 12,
      "layers": 24,
      "contextMax": 131072,
      "source": "https://huggingface.co/ggml-org/gpt-oss-20b-GGUF",
      "files": [
        "gpt-oss-20b-MXFP4.gguf"
      ]
    },
    {
      "id": "qwen3.8-27b",
      "name": "Qwen3.8 27B",
      "quant": "Q4_K_M",
      "paramsB": 27,
      "arch": "qwen35",
      "moe": false,
      "fileGb": 16.46,
      "activeGb": 15.74,
      "sharedGb": 15.74,
      "expertGb": 0,
      "kvKibPerToken": 64,
      "kvFixedMib": 0,
      "fullAttnLayers": 16,
      "layers": 65,
      "contextMax": 262144,
      "source": "https://huggingface.co/unsloth/Qwen3.8-27B-GGUF",
      "files": [
        "Qwen3.8-27B-UD-Q4_K_M.gguf"
      ]
    },
    {
      "id": "qwen3-coder-30b-a3b",
      "name": "Qwen3 Coder 30B-A3B",
      "quant": "Q4_K_M",
      "paramsB": 30.5,
      "arch": "qwen3moe",
      "moe": true,
      "experts": 128,
      "expertsUsed": 8,
      "fileGb": 18.56,
      "activeGb": 1.92,
      "sharedGb": 0.82,
      "expertGb": 17.55,
      "kvKibPerToken": 96,
      "kvFixedMib": 0,
      "fullAttnLayers": 48,
      "layers": 48,
      "contextMax": 262144,
      "source": "https://huggingface.co/unsloth/Qwen3-Coder-30B-A3B-Instruct-GGUF",
      "files": [
        "Qwen3-Coder-30B-A3B-Instruct-Q4_K_M.gguf"
      ]
    },
    {
      "id": "qwen3.6-35b-a3b",
      "name": "Qwen3.6 35B-A3B",
      "quant": "Q4_K_M",
      "paramsB": 35,
      "arch": "qwen35moe",
      "moe": true,
      "experts": 256,
      "expertsUsed": 8,
      "fileGb": 22.13,
      "activeGb": 2.63,
      "sharedGb": 2.01,
      "expertGb": 19.57,
      "kvKibPerToken": 20,
      "kvFixedMib": 0,
      "fullAttnLayers": 10,
      "layers": 40,
      "contextMax": 262144,
      "source": "https://huggingface.co/unsloth/Qwen3.6-35B-A3B-GGUF",
      "files": [
        "Qwen3.6-35B-A3B-UD-Q4_K_M.gguf"
      ]
    },
    {
      "id": "llama-3.3-70b",
      "name": "Llama 3.3 70B",
      "quant": "Q4_K_M",
      "paramsB": 70.6,
      "arch": "llama",
      "moe": false,
      "fileGb": 42.52,
      "activeGb": 41.92,
      "sharedGb": 41.92,
      "expertGb": 0,
      "kvKibPerToken": 320,
      "kvFixedMib": 0,
      "fullAttnLayers": 80,
      "layers": 80,
      "contextMax": 131072,
      "source": "https://huggingface.co/bartowski/Llama-3.3-70B-Instruct-GGUF",
      "files": [
        "Llama-3.3-70B-Instruct-Q4_K_M.gguf"
      ]
    },
    {
      "id": "gpt-oss-120b",
      "name": "gpt-oss 120B",
      "quant": "MXFP4",
      "paramsB": 117,
      "arch": "gpt-oss",
      "moe": true,
      "experts": 128,
      "expertsUsed": 4,
      "fileGb": 63.39,
      "activeGb": 3.59,
      "sharedGb": 1.69,
      "expertGb": 61.07,
      "kvKibPerToken": 36,
      "kvFixedMib": 4.5,
      "fullAttnLayers": 18,
      "layers": 36,
      "contextMax": 131072,
      "source": "https://huggingface.co/ggml-org/gpt-oss-120b-GGUF",
      "files": [
        "gpt-oss-120b-MXFP4.gguf"
      ]
    }
  ]
}