{
 "gpu": "NVIDIA GeForce RTX 5070 Ti",
 "gpu_gib": 15.92,
 "run_dates": [
  "2026-10-06 22:38 UTC",
  "2026-10-07 13:00 UTC",
  "2026-10-07 13:02 UTC",
  "2026-10-07 13:12 UTC",
  "2026-10-07 13:15 UTC",
  "2026-10-07 13:27 UTC",
  "2026-10-07 13:30 UTC",
  "2026-10-07 13:42 UTC",
  "2026-10-07 13:45 UTC"
 ],
 "runs": [
  {
   "label": "llamacpp-f16kv-chat",
   "backend": "llamacpp",
   "model": "/models/Qwen2.5-7B-Instruct-Q4_K_M.gguf",
   "scenario": "chat",
   "prompt_tokens": 512,
   "max_tokens": 256,
   "started_at": "2026-10-07 13:27 UTC",
   "gpu": "NVIDIA GeForce RTX 5070 Ti",
   "gpu_gib": 15.92,
   "kv_cache": {
    "dtype": null,
    "blocks": null,
    "block_size": null,
    "gpu_memory_utilization": null
   },
   "levels": [
    {
     "concurrency": 1,
     "requests": 48,
     "errors": 0,
     "throughput_tok_s": 128.0,
     "ttft_p50_ms": 161.2,
     "ttft_p95_ms": 164.4,
     "tpot_p50_ms": 7.22,
     "peak_vram_gib": 9.28,
     "peak_kv_cache_usage": null,
     "preemptions": null,
     "mean_prompt_tokens": 481.8
    },
    {
     "concurrency": 4,
     "requests": 48,
     "errors": 0,
     "throughput_tok_s": 198.9,
     "ttft_p50_ms": 454.2,
     "ttft_p95_ms": 461.7,
     "tpot_p50_ms": 14.27,
     "peak_vram_gib": 9.29,
     "peak_kv_cache_usage": null,
     "preemptions": null,
     "mean_prompt_tokens": 482.0
    },
    {
     "concurrency": 16,
     "requests": 48,
     "errors": 0,
     "throughput_tok_s": 527.4,
     "ttft_p50_ms": 4363.4,
     "ttft_p95_ms": 4771.7,
     "tpot_p50_ms": 13.56,
     "peak_vram_gib": 9.29,
     "peak_kv_cache_usage": null,
     "preemptions": null,
     "mean_prompt_tokens": 482.9
    }
   ]
  },
  {
   "label": "llamacpp-f16kv-long",
   "backend": "llamacpp",
   "model": "/models/Qwen2.5-7B-Instruct-Q4_K_M.gguf",
   "scenario": "long",
   "prompt_tokens": 4096,
   "max_tokens": 256,
   "started_at": "2026-10-07 13:30 UTC",
   "gpu": "NVIDIA GeForce RTX 5070 Ti",
   "gpu_gib": 15.92,
   "kv_cache": {
    "dtype": null,
    "blocks": null,
    "block_size": null,
    "gpu_memory_utilization": null
   },
   "levels": [
    {
     "concurrency": 8,
     "requests": 144,
     "errors": 0,
     "throughput_tok_s": 231.2,
     "ttft_p50_ms": 2096.5,
     "ttft_p95_ms": 2213.2,
     "tpot_p50_ms": 26.75,
     "peak_vram_gib": 9.34,
     "peak_kv_cache_usage": null,
     "preemptions": null,
     "mean_prompt_tokens": 3511.2
    },
    {
     "concurrency": 16,
     "requests": 144,
     "errors": 0,
     "throughput_tok_s": 225.8,
     "ttft_p50_ms": 9791.4,
     "ttft_p95_ms": 10453.6,
     "tpot_p50_ms": 32.69,
     "peak_vram_gib": 9.32,
     "peak_kv_cache_usage": null,
     "preemptions": null,
     "mean_prompt_tokens": 3512.0
    },
    {
     "concurrency": 32,
     "requests": 144,
     "errors": 0,
     "throughput_tok_s": 223.2,
     "ttft_p50_ms": 28271.3,
     "ttft_p95_ms": 28530.3,
     "tpot_p50_ms": 33.14,
     "peak_vram_gib": 9.32,
     "peak_kv_cache_usage": null,
     "preemptions": null,
     "mean_prompt_tokens": 3512.0
    },
    {
     "concurrency": 48,
     "requests": 144,
     "errors": 0,
     "throughput_tok_s": 219.2,
     "ttft_p50_ms": 47318.8,
     "ttft_p95_ms": 48432.6,
     "tpot_p50_ms": 33.55,
     "peak_vram_gib": 9.5,
     "peak_kv_cache_usage": null,
     "preemptions": null,
     "mean_prompt_tokens": 3512.0
    }
   ]
  },
  {
   "label": "llamacpp-q8kv-chat",
   "backend": "llamacpp",
   "model": "/models/Qwen2.5-7B-Instruct-Q4_K_M.gguf",
   "scenario": "chat",
   "prompt_tokens": 512,
   "max_tokens": 256,
   "started_at": "2026-10-07 13:42 UTC",
   "gpu": "NVIDIA GeForce RTX 5070 Ti",
   "gpu_gib": 15.92,
   "kv_cache": {
    "dtype": null,
    "blocks": null,
    "block_size": null,
    "gpu_memory_utilization": null
   },
   "levels": [
    {
     "concurrency": 1,
     "requests": 48,
     "errors": 0,
     "throughput_tok_s": 124.5,
     "ttft_p50_ms": 143.3,
     "ttft_p95_ms": 146.3,
     "tpot_p50_ms": 7.5,
     "peak_vram_gib": 7.68,
     "peak_kv_cache_usage": null,
     "preemptions": null,
     "mean_prompt_tokens": 481.8
    },
    {
     "concurrency": 4,
     "requests": 48,
     "errors": 0,
     "throughput_tok_s": 192.9,
     "ttft_p50_ms": 385.3,
     "ttft_p95_ms": 404.3,
     "tpot_p50_ms": 13.94,
     "peak_vram_gib": 7.68,
     "peak_kv_cache_usage": null,
     "preemptions": null,
     "mean_prompt_tokens": 482.0
    },
    {
     "concurrency": 16,
     "requests": 48,
     "errors": 0,
     "throughput_tok_s": 516.9,
     "ttft_p50_ms": 4373.4,
     "ttft_p95_ms": 4748.3,
     "tpot_p50_ms": 14.12,
     "peak_vram_gib": 7.68,
     "peak_kv_cache_usage": null,
     "preemptions": null,
     "mean_prompt_tokens": 482.9
    }
   ]
  },
  {
   "label": "llamacpp-q8kv-long",
   "backend": "llamacpp",
   "model": "/models/Qwen2.5-7B-Instruct-Q4_K_M.gguf",
   "scenario": "long",
   "prompt_tokens": 4096,
   "max_tokens": 256,
   "started_at": "2026-10-07 13:45 UTC",
   "gpu": "NVIDIA GeForce RTX 5070 Ti",
   "gpu_gib": 15.92,
   "kv_cache": {
    "dtype": null,
    "blocks": null,
    "block_size": null,
    "gpu_memory_utilization": null
   },
   "levels": [
    {
     "concurrency": 8,
     "requests": 144,
     "errors": 0,
     "throughput_tok_s": 226.8,
     "ttft_p50_ms": 1707.4,
     "ttft_p95_ms": 2085.3,
     "tpot_p50_ms": 28.62,
     "peak_vram_gib": 7.83,
     "peak_kv_cache_usage": null,
     "preemptions": null,
     "mean_prompt_tokens": 3511.2
    },
    {
     "concurrency": 16,
     "requests": 144,
     "errors": 0,
     "throughput_tok_s": 216.6,
     "ttft_p50_ms": 10218.8,
     "ttft_p95_ms": 10639.1,
     "tpot_p50_ms": 34.42,
     "peak_vram_gib": 7.94,
     "peak_kv_cache_usage": null,
     "preemptions": null,
     "mean_prompt_tokens": 3512.0
    },
    {
     "concurrency": 32,
     "requests": 144,
     "errors": 0,
     "throughput_tok_s": 211.7,
     "ttft_p50_ms": 29455.0,
     "ttft_p95_ms": 30649.5,
     "tpot_p50_ms": 35.05,
     "peak_vram_gib": 8.03,
     "peak_kv_cache_usage": null,
     "preemptions": null,
     "mean_prompt_tokens": 3512.0
    },
    {
     "concurrency": 48,
     "requests": 144,
     "errors": 0,
     "throughput_tok_s": 202.8,
     "ttft_p50_ms": 50434.0,
     "ttft_p95_ms": 53126.6,
     "tpot_p50_ms": 36.42,
     "peak_vram_gib": 7.94,
     "peak_kv_cache_usage": null,
     "preemptions": null,
     "mean_prompt_tokens": 3512.0
    }
   ]
  },
  {
   "label": "ollama-q4km",
   "backend": "ollama",
   "model": "qwen2.5:7b-instruct",
   "scenario": "chat",
   "prompt_tokens": 512,
   "max_tokens": 256,
   "started_at": "2026-10-06 22:38 UTC",
   "gpu": "NVIDIA GeForce RTX 5070 Ti",
   "gpu_gib": 15.92,
   "kv_cache": {
    "dtype": null,
    "blocks": null,
    "block_size": null,
    "gpu_memory_utilization": null
   },
   "levels": [
    {
     "concurrency": 1,
     "requests": 48,
     "errors": 0,
     "throughput_tok_s": 118.4,
     "ttft_p50_ms": 110.0,
     "ttft_p95_ms": 135.3,
     "tpot_p50_ms": 7.92,
     "peak_vram_gib": 8.11,
     "peak_kv_cache_usage": null,
     "preemptions": null,
     "mean_prompt_tokens": 481.8
    },
    {
     "concurrency": 4,
     "requests": 48,
     "errors": 0,
     "throughput_tok_s": 112.8,
     "ttft_p50_ms": 6313.3,
     "ttft_p95_ms": 8385.5,
     "tpot_p50_ms": 7.89,
     "peak_vram_gib": 8.12,
     "peak_kv_cache_usage": null,
     "preemptions": null,
     "mean_prompt_tokens": 482.0
    },
    {
     "concurrency": 16,
     "requests": 48,
     "errors": 0,
     "throughput_tok_s": 99.1,
     "ttft_p50_ms": 37130.9,
     "ttft_p95_ms": 42507.2,
     "tpot_p50_ms": 10.44,
     "peak_vram_gib": 8.07,
     "peak_kv_cache_usage": null,
     "preemptions": null,
     "mean_prompt_tokens": 482.9
    }
   ]
  },
  {
   "label": "vllm-fp16kv-chat",
   "backend": "vllm",
   "model": "Qwen/Qwen2.5-7B-Instruct-AWQ",
   "scenario": "chat",
   "prompt_tokens": 512,
   "max_tokens": 256,
   "started_at": "2026-10-07 13:00 UTC",
   "gpu": "NVIDIA GeForce RTX 5070 Ti",
   "gpu_gib": 15.92,
   "kv_cache": {
    "dtype": "auto",
    "blocks": 9878,
    "block_size": 16,
    "gpu_memory_utilization": 0.9
   },
   "levels": [
    {
     "concurrency": 1,
     "requests": 48,
     "errors": 0,
     "throughput_tok_s": 114.9,
     "ttft_p50_ms": 144.3,
     "ttft_p95_ms": 148.6,
     "tpot_p50_ms": 8.15,
     "peak_vram_gib": 15.66,
     "peak_kv_cache_usage": 0.005,
     "preemptions": 0.0,
     "mean_prompt_tokens": 481.8
    },
    {
     "concurrency": 4,
     "requests": 48,
     "errors": 0,
     "throughput_tok_s": 408.8,
     "ttft_p50_ms": 275.9,
     "ttft_p95_ms": 375.5,
     "tpot_p50_ms": 8.72,
     "peak_vram_gib": 15.75,
     "peak_kv_cache_usage": 0.018,
     "preemptions": 0.0,
     "mean_prompt_tokens": 482.0
    },
    {
     "concurrency": 16,
     "requests": 48,
     "errors": 0,
     "throughput_tok_s": 1092.4,
     "ttft_p50_ms": 888.8,
     "ttft_p95_ms": 1393.9,
     "tpot_p50_ms": 10.65,
     "peak_vram_gib": 15.81,
     "peak_kv_cache_usage": 0.072,
     "preemptions": 0.0,
     "mean_prompt_tokens": 482.9
    }
   ]
  },
  {
   "label": "vllm-fp16kv-long",
   "backend": "vllm",
   "model": "Qwen/Qwen2.5-7B-Instruct-AWQ",
   "scenario": "long",
   "prompt_tokens": 4096,
   "max_tokens": 256,
   "started_at": "2026-10-07 13:02 UTC",
   "gpu": "NVIDIA GeForce RTX 5070 Ti",
   "gpu_gib": 15.92,
   "kv_cache": {
    "dtype": "auto",
    "blocks": 9878,
    "block_size": 16,
    "gpu_memory_utilization": 0.9
   },
   "levels": [
    {
     "concurrency": 8,
     "requests": 144,
     "errors": 0,
     "throughput_tok_s": 257.2,
     "ttft_p50_ms": 2350.0,
     "ttft_p95_ms": 2962.6,
     "tpot_p50_ms": 22.26,
     "peak_vram_gib": 15.81,
     "peak_kv_cache_usage": 0.19,
     "preemptions": 0.0,
     "mean_prompt_tokens": 3511.2
    },
    {
     "concurrency": 16,
     "requests": 144,
     "errors": 0,
     "throughput_tok_s": 285.6,
     "ttft_p50_ms": 2423.2,
     "ttft_p95_ms": 6295.7,
     "tpot_p50_ms": 46.03,
     "peak_vram_gib": 15.85,
     "peak_kv_cache_usage": 0.378,
     "preemptions": 0.0,
     "mean_prompt_tokens": 3512.0
    },
    {
     "concurrency": 32,
     "requests": 144,
     "errors": 0,
     "throughput_tok_s": 294.0,
     "ttft_p50_ms": 2140.6,
     "ttft_p95_ms": 17341.4,
     "tpot_p50_ms": 101.06,
     "peak_vram_gib": 15.84,
     "peak_kv_cache_usage": 0.755,
     "preemptions": 0.0,
     "mean_prompt_tokens": 3512.0
    },
    {
     "concurrency": 48,
     "requests": 144,
     "errors": 0,
     "throughput_tok_s": 296.8,
     "ttft_p50_ms": 5877.1,
     "ttft_p95_ms": 27536.5,
     "tpot_p50_ms": 136.83,
     "peak_vram_gib": 15.83,
     "peak_kv_cache_usage": 1.0,
     "preemptions": 6.0,
     "mean_prompt_tokens": 3512.0
    }
   ]
  },
  {
   "label": "vllm-fp8kv-chat",
   "backend": "vllm",
   "model": "Qwen/Qwen2.5-7B-Instruct-AWQ",
   "scenario": "chat",
   "prompt_tokens": 512,
   "max_tokens": 256,
   "started_at": "2026-10-07 13:12 UTC",
   "gpu": "NVIDIA GeForce RTX 5070 Ti",
   "gpu_gib": 15.92,
   "kv_cache": {
    "dtype": "fp8",
    "blocks": 17019,
    "block_size": 16,
    "gpu_memory_utilization": 0.9
   },
   "levels": [
    {
     "concurrency": 1,
     "requests": 48,
     "errors": 0,
     "throughput_tok_s": 120.0,
     "ttft_p50_ms": 141.4,
     "ttft_p95_ms": 143.2,
     "tpot_p50_ms": 7.82,
     "peak_vram_gib": 15.27,
     "peak_kv_cache_usage": 0.003,
     "preemptions": 0.0,
     "mean_prompt_tokens": 481.8
    },
    {
     "concurrency": 4,
     "requests": 48,
     "errors": 0,
     "throughput_tok_s": 437.8,
     "ttft_p50_ms": 265.8,
     "ttft_p95_ms": 353.8,
     "tpot_p50_ms": 8.11,
     "peak_vram_gib": 15.42,
     "peak_kv_cache_usage": 0.011,
     "preemptions": 0.0,
     "mean_prompt_tokens": 482.0
    },
    {
     "concurrency": 16,
     "requests": 48,
     "errors": 0,
     "throughput_tok_s": 1209.9,
     "ttft_p50_ms": 865.2,
     "ttft_p95_ms": 1294.8,
     "tpot_p50_ms": 9.81,
     "peak_vram_gib": 15.57,
     "peak_kv_cache_usage": 0.042,
     "preemptions": 0.0,
     "mean_prompt_tokens": 482.9
    }
   ]
  },
  {
   "label": "vllm-fp8kv-long",
   "backend": "vllm",
   "model": "Qwen/Qwen2.5-7B-Instruct-AWQ",
   "scenario": "long",
   "prompt_tokens": 4096,
   "max_tokens": 256,
   "started_at": "2026-10-07 13:15 UTC",
   "gpu": "NVIDIA GeForce RTX 5070 Ti",
   "gpu_gib": 15.92,
   "kv_cache": {
    "dtype": "fp8",
    "blocks": 17019,
    "block_size": 16,
    "gpu_memory_utilization": 0.9
   },
   "levels": [
    {
     "concurrency": 8,
     "requests": 144,
     "errors": 0,
     "throughput_tok_s": 296.2,
     "ttft_p50_ms": 2179.6,
     "ttft_p95_ms": 2938.1,
     "tpot_p50_ms": 18.12,
     "peak_vram_gib": 15.59,
     "peak_kv_cache_usage": 0.11,
     "preemptions": 0.0,
     "mean_prompt_tokens": 3511.2
    },
    {
     "concurrency": 16,
     "requests": 144,
     "errors": 0,
     "throughput_tok_s": 326.9,
     "ttft_p50_ms": 2546.8,
     "ttft_p95_ms": 5828.3,
     "tpot_p50_ms": 39.11,
     "peak_vram_gib": 15.58,
     "peak_kv_cache_usage": 0.22,
     "preemptions": 0.0,
     "mean_prompt_tokens": 3512.0
    },
    {
     "concurrency": 32,
     "requests": 144,
     "errors": 0,
     "throughput_tok_s": 350.7,
     "ttft_p50_ms": 1824.4,
     "ttft_p95_ms": 15628.1,
     "tpot_p50_ms": 84.0,
     "peak_vram_gib": 15.57,
     "peak_kv_cache_usage": 0.438,
     "preemptions": 0.0,
     "mean_prompt_tokens": 3512.0
    },
    {
     "concurrency": 48,
     "requests": 144,
     "errors": 0,
     "throughput_tok_s": 359.0,
     "ttft_p50_ms": 1840.0,
     "ttft_p95_ms": 25851.8,
     "tpot_p50_ms": 125.66,
     "peak_vram_gib": 15.58,
     "peak_kv_cache_usage": 0.654,
     "preemptions": 0.0,
     "mean_prompt_tokens": 3512.0
    }
   ]
  }
 ],
 "calibration": [
  {
   "model": "Qwen/Qwen2.5-7B-Instruct-AWQ",
   "preset": "qwen2.5-7b",
   "kv_cache_dtype": "auto",
   "gpu_gib": 15.92,
   "weights_gib": 5.188,
   "gpu_memory_utilization": 0.9,
   "predicted_tokens": 143040,
   "actual_tokens": 158048,
   "error_pct": -9.5
  },
  {
   "model": "Qwen/Qwen2.5-7B-Instruct-AWQ",
   "preset": "qwen2.5-7b",
   "kv_cache_dtype": "fp8",
   "gpu_gib": 15.92,
   "weights_gib": 5.188,
   "gpu_memory_utilization": 0.9,
   "predicted_tokens": 286096,
   "actual_tokens": 272304,
   "error_pct": 5.1
  }
 ],
 "planner": {
  "gib": 1073741824,
  "kv_bytes": {
   "auto": 2.0,
   "fp8": 1.0
  },
  "gpu_memory_utilization": 0.9,
  "overhead_gib": 1.5,
  "block_size": 16,
  "models": {
   "qwen2.5-7b": {
    "n_layers": 28,
    "n_kv_heads": 4,
    "head_dim": 128,
    "params_b": 7.62
   },
   "qwen2.5-14b": {
    "n_layers": 48,
    "n_kv_heads": 8,
    "head_dim": 128,
    "params_b": 14.7
   },
   "qwen2.5-32b": {
    "n_layers": 64,
    "n_kv_heads": 8,
    "head_dim": 128,
    "params_b": 32.5
   },
   "qwen3-8b": {
    "n_layers": 36,
    "n_kv_heads": 8,
    "head_dim": 128,
    "params_b": 8.19
   },
   "qwen3-14b": {
    "n_layers": 40,
    "n_kv_heads": 8,
    "head_dim": 128,
    "params_b": 14.8
   },
   "llama-3.1-8b": {
    "n_layers": 32,
    "n_kv_heads": 8,
    "head_dim": 128,
    "params_b": 8.03
   },
   "mistral-7b": {
    "n_layers": 32,
    "n_kv_heads": 8,
    "head_dim": 128,
    "params_b": 7.25
   }
  }
 }
}