[
  {
    "build_commit": "4b1a27fa0",
    "build_number": 11191,
    "cpu_info": "AMD Ryzen 9 7900X 12-Core Processor            ",
    "gpu_info": "NVIDIA GeForce RTX 4060",
    "backends": "CUDA",
    "model_filename": "D:\\AIpcs-bench\\models\\llama-2-7b.Q4_0.gguf",
    "model_type": "llama 7B Q4_0",
    "model_size": 3825065984,
    "model_n_params": 6738415616,
    "n_batch": 2048,
    "n_ubatch": 512,
    "n_threads": 12,
    "cpu_mask": "0x0",
    "cpu_strict": false,
    "poll": 50,
    "type_k": "f16",
    "type_v": "f16",
    "n_gpu_layers": 99,
    "n_cpu_moe": 0,
    "split_mode": "layer",
    "main_gpu": 0,
    "no_kv_offload": false,
    "flash_attn": -1,
    "devices": "auto",
    "tensor_split": "0.00",
    "tensor_buft_overrides": "none",
    "load_mode": "auto",
    "lazy_mode": "auto",
    "embeddings": false,
    "no_op_offload": 0,
    "no_host": false,
    "fit_target": 0,
    "fit_min_ctx": 0,
    "n_prompt": 512,
    "n_gen": 0,
    "n_depth": 0,
    "test_time": "2026-09-25T22:49:36Z",
    "avg_ns": 188560440,
    "stddev_ns": 2639173,
    "avg_ts": 2715.733589,
    "stddev_ts": 37.854465,
    "samples_ns": [ 190194600, 188205900, 186102800, 186122000, 192176900 ],
    "samples_ts": [ 2691.98, 2720.42, 2751.17, 2750.88, 2664.21 ]
  },
  {
    "build_commit": "4b1a27fa0",
    "build_number": 11191,
    "cpu_info": "AMD Ryzen 9 7900X 12-Core Processor            ",
    "gpu_info": "NVIDIA GeForce RTX 4060",
    "backends": "CUDA",
    "model_filename": "D:\\AIpcs-bench\\models\\llama-2-7b.Q4_0.gguf",
    "model_type": "llama 7B Q4_0",
    "model_size": 3825065984,
    "model_n_params": 6738415616,
    "n_batch": 2048,
    "n_ubatch": 512,
    "n_threads": 12,
    "cpu_mask": "0x0",
    "cpu_strict": false,
    "poll": 50,
    "type_k": "f16",
    "type_v": "f16",
    "n_gpu_layers": 99,
    "n_cpu_moe": 0,
    "split_mode": "layer",
    "main_gpu": 0,
    "no_kv_offload": false,
    "flash_attn": -1,
    "devices": "auto",
    "tensor_split": "0.00",
    "tensor_buft_overrides": "none",
    "load_mode": "auto",
    "lazy_mode": "auto",
    "embeddings": false,
    "no_op_offload": 0,
    "no_host": false,
    "fit_target": 0,
    "fit_min_ctx": 0,
    "n_prompt": 0,
    "n_gen": 128,
    "n_depth": 0,
    "test_time": "2026-09-25T22:49:37Z",
    "avg_ns": 1954078580,
    "stddev_ns": 4160265,
    "avg_ts": 65.504256,
    "stddev_ts": 0.139285,
    "samples_ns": [ 1952329600, 1956222300, 1960323700, 1951274100, 1950243200 ],
    "samples_ts": [ 65.5627, 65.4322, 65.2953, 65.5982, 65.6328 ]
  },
  {
    "build_commit": "4b1a27fa0",
    "build_number": 11191,
    "cpu_info": "AMD Ryzen 9 7900X 12-Core Processor            ",
    "gpu_info": "NVIDIA GeForce RTX 4060",
    "backends": "CUDA",
    "model_filename": "D:\\AIpcs-bench\\models\\llama-2-7b.Q4_0.gguf",
    "model_type": "llama 7B Q4_0",
    "model_size": 3825065984,
    "model_n_params": 6738415616,
    "n_batch": 2048,
    "n_ubatch": 512,
    "n_threads": 12,
    "cpu_mask": "0x0",
    "cpu_strict": false,
    "poll": 50,
    "type_k": "f16",
    "type_v": "f16",
    "n_gpu_layers": 99,
    "n_cpu_moe": 0,
    "split_mode": "layer",
    "main_gpu": 0,
    "no_kv_offload": false,
    "flash_attn": -1,
    "devices": "auto",
    "tensor_split": "0.00",
    "tensor_buft_overrides": "none",
    "load_mode": "auto",
    "lazy_mode": "auto",
    "embeddings": false,
    "no_op_offload": 0,
    "no_host": false,
    "fit_target": 0,
    "fit_min_ctx": 0,
    "n_prompt": 512,
    "n_gen": 0,
    "n_depth": 4096,
    "test_time": "2026-09-25T22:49:47Z",
    "avg_ns": 273211680,
    "stddev_ns": 1779090,
    "avg_ts": 1874.068513,
    "stddev_ts": 12.230208,
    "samples_ns": [ 274902700, 274692500, 271918300, 270828300, 273716600 ],
    "samples_ts": [ 1862.48, 1863.9, 1882.92, 1890.5, 1870.55 ]
  },
  {
    "build_commit": "4b1a27fa0",
    "build_number": 11191,
    "cpu_info": "AMD Ryzen 9 7900X 12-Core Processor            ",
    "gpu_info": "NVIDIA GeForce RTX 4060",
    "backends": "CUDA",
    "model_filename": "D:\\AIpcs-bench\\models\\llama-2-7b.Q4_0.gguf",
    "model_type": "llama 7B Q4_0",
    "model_size": 3825065984,
    "model_n_params": 6738415616,
    "n_batch": 2048,
    "n_ubatch": 512,
    "n_threads": 12,
    "cpu_mask": "0x0",
    "cpu_strict": false,
    "poll": 50,
    "type_k": "f16",
    "type_v": "f16",
    "n_gpu_layers": 99,
    "n_cpu_moe": 0,
    "split_mode": "layer",
    "main_gpu": 0,
    "no_kv_offload": false,
    "flash_attn": -1,
    "devices": "auto",
    "tensor_split": "0.00",
    "tensor_buft_overrides": "none",
    "load_mode": "auto",
    "lazy_mode": "auto",
    "embeddings": false,
    "no_op_offload": 0,
    "no_host": false,
    "fit_target": 0,
    "fit_min_ctx": 0,
    "n_prompt": 0,
    "n_gen": 128,
    "n_depth": 4096,
    "test_time": "2026-09-25T22:50:05Z",
    "avg_ns": 2953598140,
    "stddev_ns": 5089384,
    "avg_ts": 43.337075,
    "stddev_ts": 0.074649,
    "samples_ns": [ 2958587800, 2951746700, 2948144000, 2950127200, 2959385000 ],
    "samples_ts": [ 43.2639, 43.3642, 43.4171, 43.388, 43.2522 ]
  }
]
