{
  "name": "AIPCs local AI benchmark results",
  "license": "CC BY 4.0 (https://creativecommons.org/licenses/by/4.0/)",
  "attribution": "AIPCs (https://www.aipcs.store/benchmarks/)",
  "units": {
    "pp": "prompt processing, tokens/s",
    "tg": "token generation, tokens/s"
  },
  "tests": [
    {
      "id": "T0001",
      "date": "2026-09-25",
      "machine": {
        "id": "bench-desktop",
        "name": "Reference desktop",
        "cpu": "AMD Ryzen 9 7900X (12 cores / 24 threads, Zen 4)",
        "ram": "32 GB DDR5-6400 (2 x 16 GB, dual channel)",
        "ramBandwidthGBs": 102.4,
        "gpus": [
          {
            "name": "NVIDIA GeForce RTX 4060 8 GB",
            "vramGb": 8,
            "bandwidthGBs": 272
          }
        ],
        "os": "Windows 11 Home",
        "notes": "The RTX 4060 runs on a PCIe 3.0 x1 link in this machine (the card supports PCIe 4.0 x8). That does not change results for models that fit entirely in its 8 GB, but it slows runs that split a model between GPU and CPU. We label those results."
      },
      "device": "AMD Ryzen 9 7900X (CPU only)",
      "backend": "CPU",
      "llamaCpp": "build 11191 (4b1a27fa0)",
      "raw": "bench/results/bench-desktop-20260925-184928/",
      "results": [
        {
          "model": "Gemma 4 E4B",
          "quant": "Q4_0",
          "fileGb": 4.59,
          "ngl": 0,
          "threads": 12,
          "pp": 197.58,
          "tg": 17.75,
          "ppTokens": 512,
          "tgTokens": 128,
          "depth": 0,
          "repetitions": 5,
          "ok": true
        },
        {
          "model": "gpt-oss 20B",
          "quant": "MXFP4",
          "fileGb": 12.11,
          "ngl": 0,
          "threads": 12,
          "pp": 109.25,
          "tg": 21.84,
          "ppTokens": 512,
          "tgTokens": 128,
          "depth": 0,
          "repetitions": 5,
          "ok": true
        },
        {
          "model": "Llama 2 7B",
          "quant": "Q4_0",
          "fileGb": 3.83,
          "ngl": 0,
          "threads": 12,
          "pp": 133.88,
          "tg": 15.5,
          "ppTokens": 512,
          "tgTokens": 128,
          "depth": 0,
          "repetitions": 5,
          "ok": true
        },
        {
          "model": "Qwen3.5 9B",
          "quant": "Q4_K_M",
          "fileGb": 5.68,
          "ngl": 0,
          "threads": 12,
          "pp": 92.49,
          "tg": 10.37,
          "ppTokens": 512,
          "tgTokens": 128,
          "depth": 0,
          "repetitions": 5,
          "ok": true
        }
      ],
      "url": "https://www.aipcs.store/benchmarks/#T0001"
    },
    {
      "id": "T0002",
      "date": "2026-09-25",
      "machine": {
        "id": "bench-desktop",
        "name": "Reference desktop",
        "cpu": "AMD Ryzen 9 7900X (12 cores / 24 threads, Zen 4)",
        "ram": "32 GB DDR5-6400 (2 x 16 GB, dual channel)",
        "ramBandwidthGBs": 102.4,
        "gpus": [
          {
            "name": "NVIDIA GeForce RTX 4060 8 GB",
            "vramGb": 8,
            "bandwidthGBs": 272
          }
        ],
        "os": "Windows 11 Home",
        "notes": "The RTX 4060 runs on a PCIe 3.0 x1 link in this machine (the card supports PCIe 4.0 x8). That does not change results for models that fit entirely in its 8 GB, but it slows runs that split a model between GPU and CPU. We label those results."
      },
      "device": "NVIDIA GeForce RTX 4060 8 GB",
      "backend": "CUDA 12.4",
      "llamaCpp": "build 11191 (4b1a27fa0)",
      "driver": "NVIDIA 610.88",
      "notes": "In this machine the RTX 4060 runs on a PCIe 3.0 x1 link (the card supports PCIe 4.0 x8). Rows where the model is split between the GPU and system RAM (gpt-oss 20B) are slowed by that link, prompt reading most of all; rows with the whole model on the GPU are not affected.",
      "raw": "bench/results/bench-desktop-20260925-184928/",
      "results": [
        {
          "model": "Gemma 4 E4B",
          "quant": "Q4_0",
          "fileGb": 4.59,
          "ngl": 99,
          "threads": 12,
          "pp": 2964.33,
          "tg": 67.66,
          "ppTokens": 512,
          "tgTokens": 128,
          "depth": 0,
          "repetitions": 5,
          "ok": true
        },
        {
          "model": "Gemma 4 E4B",
          "quant": "Q4_0",
          "fileGb": 4.59,
          "ngl": 99,
          "threads": 12,
          "pp": 2722.31,
          "tg": 65.06,
          "ppTokens": 512,
          "tgTokens": 128,
          "depth": 4096,
          "repetitions": 5,
          "ok": true
        },
        {
          "model": "gpt-oss 20B",
          "quant": "MXFP4",
          "fileGb": 12.11,
          "ngl": 99,
          "nCpuMoe": 24,
          "threads": 12,
          "pp": 51.67,
          "tg": 25.21,
          "ppTokens": 512,
          "tgTokens": 128,
          "depth": 0,
          "repetitions": 5,
          "ok": true
        },
        {
          "model": "gpt-oss 20B",
          "quant": "MXFP4",
          "fileGb": 12.11,
          "ngl": -1,
          "threads": 12,
          "pp": 106.28,
          "tg": 39.55,
          "ppTokens": 512,
          "tgTokens": 128,
          "depth": 0,
          "repetitions": 5,
          "ok": true
        },
        {
          "model": "Llama 2 7B",
          "quant": "Q4_0",
          "fileGb": 3.83,
          "ngl": 99,
          "threads": 12,
          "pp": 2715.73,
          "tg": 65.5,
          "ppTokens": 512,
          "tgTokens": 128,
          "depth": 0,
          "repetitions": 5,
          "ok": true
        },
        {
          "model": "Llama 2 7B",
          "quant": "Q4_0",
          "fileGb": 3.83,
          "ngl": 99,
          "threads": 12,
          "pp": 1874.07,
          "tg": 43.34,
          "ppTokens": 512,
          "tgTokens": 128,
          "depth": 4096,
          "repetitions": 5,
          "ok": true
        },
        {
          "model": "Qwen3.5 9B",
          "quant": "Q4_K_M",
          "fileGb": 5.68,
          "ngl": 99,
          "threads": 12,
          "pp": 1897.22,
          "tg": 45.53,
          "ppTokens": 512,
          "tgTokens": 128,
          "depth": 0,
          "repetitions": 5,
          "ok": true
        },
        {
          "model": "Qwen3.5 9B",
          "quant": "Q4_K_M",
          "fileGb": 5.68,
          "ngl": 99,
          "threads": 12,
          "pp": 1807.96,
          "tg": 44.6,
          "ppTokens": 512,
          "tgTokens": 128,
          "depth": 4096,
          "repetitions": 5,
          "ok": true
        }
      ],
      "url": "https://www.aipcs.store/benchmarks/#T0002"
    },
    {
      "id": "T0003",
      "date": "2026-09-26",
      "machine": {
        "id": "bench-k1",
        "name": "ACEMAGIC K1",
        "product": {
          "id": "acemagic-k1",
          "collection": "products"
        },
        "cpu": "AMD Ryzen 3 4300U (4 cores / 4 threads, Zen 2)",
        "ram": "16 GB DDR4-2666 (one SO-DIMM as shipped, so single channel; the second slot is free)",
        "ramBandwidthGBs": 21.3,
        "gpus": [],
        "os": "Windows 11 Pro",
        "notes": "The integrated Radeon graphics (5 compute units) share system memory, so they run at the same 21.3 GB/s and have no separate VRAM figure. Our tests on it ran over a Remote Desktop session."
      },
      "device": "AMD Ryzen 3 4300U (CPU only)",
      "backend": "CPU",
      "llamaCpp": "build 11191 (4b1a27fa0)",
      "notes": "Run over a Remote Desktop session. The K1 has one 16 GB DDR4-2666 module, so its memory runs single-channel at 21.3 GB/s (theoretical peak). Windows power mode: Best performance.",
      "raw": "bench/results/bench-k1-20260926-130154/",
      "results": [
        {
          "model": "Gemma 4 E4B",
          "quant": "Q4_0",
          "fileGb": 4.59,
          "ngl": 0,
          "threads": 4,
          "pp": 31.76,
          "tg": 4.79,
          "ppTokens": 512,
          "tgTokens": 128,
          "depth": 0,
          "repetitions": 5,
          "ok": true
        },
        {
          "model": "gpt-oss 20B",
          "quant": "MXFP4",
          "fileGb": 12.11,
          "ngl": 0,
          "threads": 4,
          "pp": 27.38,
          "tg": 5.5,
          "ppTokens": 512,
          "tgTokens": 128,
          "depth": 0,
          "repetitions": 5,
          "ok": true
        },
        {
          "model": "Llama 2 7B",
          "quant": "Q4_0",
          "fileGb": 3.83,
          "ngl": 0,
          "threads": 4,
          "pp": 20.47,
          "tg": 3.94,
          "ppTokens": 512,
          "tgTokens": 128,
          "depth": 0,
          "repetitions": 5,
          "ok": true
        },
        {
          "model": "Qwen3.5 9B",
          "quant": "Q4_K_M",
          "fileGb": 5.68,
          "ngl": 0,
          "threads": 4,
          "pp": 18.07,
          "tg": 2.7,
          "ppTokens": 512,
          "tgTokens": 128,
          "depth": 0,
          "repetitions": 5,
          "ok": true
        }
      ],
      "url": "https://www.aipcs.store/benchmarks/#T0003"
    },
    {
      "id": "T0004",
      "date": "2026-09-26",
      "machine": {
        "id": "bench-k1",
        "name": "ACEMAGIC K1",
        "product": {
          "id": "acemagic-k1",
          "collection": "products"
        },
        "cpu": "AMD Ryzen 3 4300U (4 cores / 4 threads, Zen 2)",
        "ram": "16 GB DDR4-2666 (one SO-DIMM as shipped, so single channel; the second slot is free)",
        "ramBandwidthGBs": 21.3,
        "gpus": [],
        "os": "Windows 11 Pro",
        "notes": "The integrated Radeon graphics (5 compute units) share system memory, so they run at the same 21.3 GB/s and have no separate VRAM figure. Our tests on it ran over a Remote Desktop session."
      },
      "device": "AMD Radeon Graphics, integrated (Ryzen 3 4300U)",
      "backend": "Vulkan",
      "llamaCpp": "build 11191 (4b1a27fa0)",
      "notes": "Run over a Remote Desktop session. The K1 has one 16 GB DDR4-2666 module, so its memory runs single-channel at 21.3 GB/s (theoretical peak). Windows power mode: Best performance. The integrated graphics share that memory. gpt-oss 20B (12.1 GB) did not fit in the memory Windows lets the integrated graphics use, and failed to load.",
      "raw": "bench/results/bench-k1-20260926-130154/",
      "results": [
        {
          "model": "Gemma 4 E4B",
          "quant": "Q4_0",
          "fileGb": 4.59,
          "ngl": 99,
          "threads": 4,
          "pp": 54.09,
          "tg": 5.87,
          "ppTokens": 512,
          "tgTokens": 128,
          "depth": 0,
          "repetitions": 5,
          "ok": true
        },
        {
          "model": "Gemma 4 E4B",
          "quant": "Q4_0",
          "fileGb": 4.59,
          "ngl": 99,
          "threads": 4,
          "pp": 9.38,
          "tg": 5.08,
          "ppTokens": 512,
          "tgTokens": 128,
          "depth": 4096,
          "repetitions": 5,
          "ok": true
        },
        {
          "model": "gpt-oss 20B",
          "quant": "MXFP4",
          "fileGb": 12.11,
          "ngl": 99,
          "pp": null,
          "tg": null,
          "ppTokens": 512,
          "tgTokens": 128,
          "depth": 0,
          "repetitions": 5,
          "ok": false,
          "error": "out of graphics memory while loading the model"
        },
        {
          "model": "Llama 2 7B",
          "quant": "Q4_0",
          "fileGb": 3.83,
          "ngl": 99,
          "threads": 4,
          "pp": 38.87,
          "tg": 4.82,
          "ppTokens": 512,
          "tgTokens": 128,
          "depth": 0,
          "repetitions": 5,
          "ok": true
        },
        {
          "model": "Llama 2 7B",
          "quant": "Q4_0",
          "fileGb": 3.83,
          "ngl": 99,
          "threads": 4,
          "pp": 17.22,
          "tg": 2.79,
          "ppTokens": 512,
          "tgTokens": 128,
          "depth": 4096,
          "repetitions": 5,
          "ok": true
        },
        {
          "model": "Qwen3.5 9B",
          "quant": "Q4_K_M",
          "fileGb": 5.68,
          "ngl": 99,
          "threads": 4,
          "pp": 33.61,
          "tg": 3.37,
          "ppTokens": 512,
          "tgTokens": 128,
          "depth": 0,
          "repetitions": 5,
          "ok": true
        },
        {
          "model": "Qwen3.5 9B",
          "quant": "Q4_K_M",
          "fileGb": 5.68,
          "ngl": 99,
          "threads": 4,
          "pp": 14.78,
          "tg": 3.1,
          "ppTokens": 512,
          "tgTokens": 128,
          "depth": 4096,
          "repetitions": 5,
          "ok": true
        }
      ],
      "url": "https://www.aipcs.store/benchmarks/#T0004"
    }
  ]
}