{
  "generatedBy": "bench.mjs",
  "startedAt": "2026-10-03T22:11:30.238Z",
  "finishedAt": "2026-10-03T22:11:30.328Z",
  "machine": {
    "cpu": "Apple M1",
    "os": "Darwin 26.3",
    "arch": "arm64",
    "node": "v22.18.0",
    "python": "Python 3.13.5"
  },
  "metricCount": 35,
  "metrics": [
    {
      "metric": "ollama_version",
      "value": "0.35.1",
      "unit": ""
    },
    {
      "metric": "llamacpp_build",
      "value": "b11351",
      "unit": "",
      "note": "llama-b11351-bin-macos-arm64.tar.gz"
    },
    {
      "metric": "mlx_lm_version",
      "value": "0.32.0",
      "unit": "",
      "note": "mlx 0.32.3, installed with uv into a Python 3.12 venv"
    },
    {
      "metric": "model_gguf",
      "value": "qwen2.5:1.5b Ollama blob sha256 183715c4..., Q4_K_M, 986048512 bytes",
      "unit": "",
      "note": "same file for Ollama and llama.cpp"
    },
    {
      "metric": "model_mlx",
      "value": "mlx-community/Qwen2.5-1.5B-Instruct-4bit, model.safetensors 868628559 bytes",
      "unit": "",
      "note": "MLX 4-bit; a different quantization from Q4_K_M"
    },
    {
      "metric": "ollama_prompt_tokens",
      "value": 719,
      "unit": "tokens",
      "note": "chat-templated prompt length"
    },
    {
      "metric": "ollama_cold_load_s",
      "value": 1.82,
      "unit": "s",
      "note": "load_duration, first request",
      "n": 1
    },
    {
      "metric": "ollama_gen_tps_median",
      "value": 49.61,
      "unit": "tok/s",
      "note": "128-token generation, warm requests; min 49.44, max 49.7",
      "n": 3
    },
    {
      "metric": "ollama_ps",
      "value": "1.2 GB, 100% GPU, context 4096",
      "unit": "",
      "note": "`ollama ps` SIZE, PROCESSOR, CONTEXT"
    },
    {
      "metric": "ollama_runner_rss_mib",
      "value": 1178,
      "unit": "MiB",
      "note": "ps RSS of Ollama's llama-server runner process with the model loaded"
    },
    {
      "metric": "llamabench_pp512_tps_mean",
      "value": 559.42,
      "unit": "tok/s",
      "note": "llama-bench -r 3; samples [559.665, 559.694, 558.904]",
      "n": 3
    },
    {
      "metric": "llamabench_tg128_tps_mean",
      "value": 50.57,
      "unit": "tok/s",
      "note": "llama-bench -r 3; samples [50.4309, 50.5139, 50.7789]",
      "n": 3
    },
    {
      "metric": "llamaserver_gen_tps_median",
      "value": 49.01,
      "unit": "tok/s",
      "note": "same chat request via /v1/chat/completions, 128 tokens, warm requests; min 48.49, max 49.24",
      "n": 3
    },
    {
      "metric": "llamaserver_rss_mib",
      "value": 1195,
      "unit": "MiB",
      "note": "ps RSS of llama-server with the model loaded, -c 4096"
    },
    {
      "metric": "ollama_pp_tps_median",
      "value": 563.23,
      "unit": "tok/s",
      "note": "723-token prompt, prompt cache defeated, warm model; min 557.02, max 563.32",
      "n": 3
    },
    {
      "metric": "llamaserver_pp_tps_median",
      "value": 539.14,
      "unit": "tok/s",
      "note": "same prompt, cache_prompt false, warm model; min 538.71, max 542.4",
      "n": 3
    },
    {
      "metric": "mlx_benchmark_pp512_tps_median",
      "value": 604.99,
      "unit": "tok/s",
      "note": "mlx_lm.benchmark -p 512 -g 128 -n 3; min 603.46, max 606.88",
      "n": 3
    },
    {
      "metric": "mlx_benchmark_tg128_tps_median",
      "value": 59.96,
      "unit": "tok/s",
      "note": "mlx_lm.benchmark -p 512 -g 128 -n 3; min 59.62, max 59.99",
      "n": 3
    },
    {
      "metric": "mlx_benchmark_peak_memory_gb",
      "value": 1.432,
      "unit": "GB",
      "note": "mlx_lm.benchmark peak_memory"
    },
    {
      "metric": "mlx_generate_gen_tps_median",
      "value": 58.74,
      "unit": "tok/s",
      "note": "mlx_lm.generate, same prompt, 128 tokens, temp 0, separate process each run; min 57.96, max 58.93",
      "n": 3
    },
    {
      "metric": "mlx_generate_pp_tps_median",
      "value": 585.34,
      "unit": "tok/s",
      "note": "mlx_lm.generate prompt rate (no cache across processes); min 583.34, max 586.12",
      "n": 3
    },
    {
      "metric": "mlx_generate_peak_footprint_mib",
      "value": 1573,
      "unit": "MiB",
      "note": "/usr/bin/time -l peak memory footprint of the whole Python process",
      "n": 3
    },
    {
      "metric": "lmstudio_llmster_version",
      "value": "0.0.25-1",
      "unit": "",
      "note": "llmster-0.0.25-1-darwin-arm64.full.tar.gz from the official install.sh, bootstrapped with HOME pointed at a lab folder"
    },
    {
      "metric": "lmstudio_engine",
      "value": "llama.cpp-mac-arm64-apple-metal-advsimd 2.41.0",
      "unit": "",
      "note": "runtime field of the /api/v0 response"
    },
    {
      "metric": "lmstudio_load_s",
      "value": 26.46,
      "unit": "s",
      "note": "lms load, first load after install",
      "n": 1
    },
    {
      "metric": "lmstudio_gen_tps_median",
      "value": 48.16,
      "unit": "tok/s",
      "note": "stats.tokens_per_second, same prompt, 128 tokens, warm; min 47.76, max 48.42",
      "n": 3
    },
    {
      "metric": "lmstudio_pp_tps_median",
      "value": 538.49,
      "unit": "tok/s",
      "note": "prompt_tokens / time_to_first_token with a run-number prefix (cache defeated); TTFT includes request overhead; min 524.93, max 539.34",
      "n": 3
    },
    {
      "metric": "lmstudio_rss_mib",
      "value": {
        "llmster": 373,
        "node": 63,
        "llama-server": 1246
      },
      "unit": "MiB",
      "note": "ps RSS per process with the model loaded"
    },
    {
      "metric": "lmstudio_download_mb",
      "value": 615.4,
      "unit": "MB",
      "note": "llmster full bundle tarball"
    },
    {
      "metric": "lmstudio_installed_size",
      "value": "2.1G",
      "unit": "",
      "note": "du -sh of the isolated LM Studio home, model file excluded"
    },
    {
      "metric": "mlx_install_packages",
      "value": 34,
      "unit": "packages",
      "note": "uv pip install mlx-lm into a fresh venv"
    },
    {
      "metric": "mlx_install_seconds",
      "value": 1503,
      "unit": "s",
      "note": "wall time on this connection (download-bound; 'Prepared 14 packages in 21m 55s')"
    },
    {
      "metric": "mlx_venv_size",
      "value": "345M",
      "unit": "",
      "note": "du -sh of the venv"
    },
    {
      "metric": "ollama_download_mb",
      "value": 159.6,
      "unit": "MB",
      "note": "ollama-darwin.tgz"
    },
    {
      "metric": "llamacpp_download_mb",
      "value": 11.8,
      "unit": "MB",
      "note": "llama-b11351-bin-macos-arm64.tar.gz"
    }
  ],
  "log": [
    ""
  ]
}
