{
  "schema_version": "1.0.0",
  "reported_on": "2026-09-30",
  "provenance": {
    "kind": "user-supplied rerun summary and updated skill documents",
    "raw_response_counters_supplied": false,
    "streamed_arrival_timestamps_supplied": false,
    "inference_rerun_by_blog_author": false,
    "independently_recomputed": false
  },
  "hardware_reported": {
    "chip": "Apple M5 Max",
    "unified_memory_gb": 64,
    "memory_type": "LPDDR5X",
    "bandwidth_gb_per_second": 400
  },
  "method_reported": {
    "prefill_prompt_tokens": 114,
    "burst_prompt_tokens": 35,
    "sustained_generated_tokens_minimum": 2600,
    "kv_estimate_context_tokens": 16384,
    "repetition_count": null,
    "runtime_version": null,
    "exact_requests": null,
    "sampling_settings": null,
    "requested_context_tokens": null,
    "thinking_mode": null,
    "speculative_decoding_activation": null
  },
  "provided_script_audit": {
    "filename": "benchmark_suite.py",
    "stream": false,
    "ttft_sec_formula": "load_duration / 1e9 + prompt_eval_duration / 1e9",
    "client_observed_first_token": false,
    "note": "The summary does not identify its precise executing script. Its TTFT method remains unconfirmed; the supplied script computes an estimate."
  },
  "models": [
    {
      "model": "gemma4:12b-mlx",
      "cold_load_seconds": 1.424,
      "warm_load_seconds_range": [
        0.017,
        0.064
      ],
      "prompt_eval_tokens_per_second_range": [
        390.58,
        390.58
      ],
      "reported_ttft_seconds": 0.309,
      "short_generation_tokens_per_second_range": [
        67.42,
        96.5
      ],
      "sustained_generation_tokens_per_second": 63.75,
      "reported_weight_footprint_gb": 7.7,
      "estimated_footprint_with_16k_kv_gb": 8.9,
      "sustained_generated_tokens_note": "2,600+ output tokens reported"
    },
    {
      "model": "gemma4:26b-mlx",
      "cold_load_seconds": 2.633,
      "warm_load_seconds_range": [
        0.01,
        0.019
      ],
      "prompt_eval_tokens_per_second_range": [
        78.86,
        83.87
      ],
      "reported_ttft_seconds": 1.465,
      "short_generation_tokens_per_second_range": [
        126.97,
        130.27
      ],
      "sustained_generation_tokens_per_second": 89.42,
      "reported_weight_footprint_gb": 18.0,
      "estimated_footprint_with_16k_kv_gb": 20.8,
      "sustained_generated_tokens_note": "2,600+ output tokens reported"
    },
    {
      "model": "llmsx-research-gemma31-mlx",
      "cold_load_seconds": 4.635,
      "warm_load_seconds_range": [
        0.02,
        0.064
      ],
      "prompt_eval_tokens_per_second_range": [
        123.59,
        123.59
      ],
      "reported_ttft_seconds": 0.944,
      "short_generation_tokens_per_second_range": [
        25.37,
        29.07
      ],
      "sustained_generation_tokens_per_second": 27.39,
      "reported_weight_footprint_gb": 19.0,
      "estimated_footprint_with_16k_kv_gb": 22.0,
      "sustained_generated_tokens_note": "2,758+ output tokens reported"
    }
  ],
  "interpretation_limits": [
    "TTFT values are preserved as reported early-response estimates pending client-observed streamed timestamps.",
    "Decode workload, loading and cache state differ from the native two-tool probes. Do not pool the measurements.",
    "Long generated output does not establish long-input-context throughput.",
    "Weight and KV figures are sizing estimates, not peak resident memory traces.",
    "No NVIDIA or RTX 5080 deployment was demonstrated by these trials.",
    "Gemma26 is a mixture-of-experts model. The reported trial does not isolate layer pruning, kernels or speculative decoding.",
    "Apple specifies 460 or 614 GB/s for M5 Max variants. The reported 400 GB/s is not an independently measured bandwidth ceiling.",
    "Short generation performance does not establish research correctness. Gemma31 remains the canonical model selected from reviewed workflow results."
  ],
  "primary_references": {
    "api_counters": "https://docs.ollama.com/api/generate",
    "model_architecture": "https://ollama.com/library/gemma4:26b-mlx",
    "hardware_specification": "https://www.apple.com/macbook-pro/specs/"
  },
  "supplied_document_sha256": {
    "SKILL.md": "fee0efc72106ac6656f267560a6a3885eef47e632e99a1922c30a0ed5a88c65a",
    "RABBITHOLE.md": "0ee267f4ed0c089a9ee0ee7d7eeae042d4c7953d534e91c5bb2dbfa83187e385",
    "scripts/benchmark_suite.py": "9a5717d03cc5a97267d9f7dec5a8546a8a8883a5d90e4d7c7771a501120fa879"
  },
  "export_note": "Summary only. No private paths, credentials, session identifiers or raw model reasoning."
}
