{
  "schema_version": 2,
  "status": "correctness_failed_no_promotion",
  "sources": {
    "short-results.json": "0bea6816982b66ddba1cb35a8f62dd6aea4aa61c75587076d0e0efcfd8c6f112",
    "ollama-control.json": "9932de9b1c701cc32eb8a74ee470f70fc4b30564c94f04ff6098022eab8d2380",
    "cache-results.json": "5343680f12fa7572a4ac5f69beb85c77255910967f13679c5b7261b3663dbaa9",
    "logit-diagnostic.json": "801c4affb8c1e19c9a4efae7d66646e9fc7649adfcf87dfb73e24862bc953db7",
    "logit-diagnostic-position60.json": "bcf27927afa6c63a2c85c35023eea1f8e2181e3be56b2ec881de39f7b843ba76"
  },
  "strict_trial_comparisons": 30,
  "strict_failures": [
    {
      "prompt_index": 1,
      "repeat": 0,
      "source": "speculative",
      "case": "8",
      "draft_length": 8,
      "first_mismatch_index": 60,
      "baseline_id": 12630,
      "trial_id": 54945
    },
    {
      "prompt_index": 1,
      "repeat": 0,
      "source": "oracle_cases",
      "case": "accepted",
      "draft_length": 8,
      "first_mismatch_index": 60,
      "baseline_id": 12630,
      "trial_id": 54945
    },
    {
      "prompt_index": 1,
      "repeat": 0,
      "source": "oracle_cases",
      "case": "middle",
      "draft_length": 8,
      "first_mismatch_index": 60,
      "baseline_id": 12630,
      "trial_id": 54945
    },
    {
      "prompt_index": 1,
      "repeat": 1,
      "source": "speculative",
      "case": "8",
      "draft_length": 8,
      "first_mismatch_index": 60,
      "baseline_id": 12630,
      "trial_id": 54945
    }
  ],
  "acceptance": {
    "2": {
      "accepted": 30,
      "proposed": 426,
      "fraction": 0.07042253521126761
    },
    "4": {
      "accepted": 50,
      "proposed": 772,
      "fraction": 0.06476683937823834
    },
    "8": {
      "accepted": 74,
      "proposed": 1374,
      "fraction": 0.053857350800582245
    }
  },
  "native_summary": {
    "baseline": {
      "samples": 6,
      "median_tokens_per_second": 21.131671546556824,
      "median_ttft_seconds": 0.1711041665,
      "p95_ttft_seconds": 0.870802166,
      "p95_is_exploratory": true
    },
    "draft_lengths": {
      "2": {
        "samples": 6,
        "median_tokens_per_second": 2.5063639555477595,
        "median_ttft_seconds": 1.0243649375,
        "p95_ttft_seconds": 3.7452225,
        "p95_is_exploratory": true,
        "paired_speedups": [
          0.1627237536956641,
          0.13442377931986516,
          0.11555095368240668,
          0.1440394745072047,
          0.1047864872283585,
          0.10896547917980298
        ],
        "speedup_vs_native_plain_median": 0.1186069899877913
      },
      "4": {
        "samples": 6,
        "median_tokens_per_second": 1.868681651948949,
        "median_ttft_seconds": 0.8346532289999999,
        "p95_ttft_seconds": 0.842707708,
        "p95_is_exploratory": true,
        "paired_speedups": [
          0.2024834902596947,
          0.10998947509103008,
          0.08194809925357047,
          0.09787601508636709,
          0.06768281169905882,
          0.06798605493535927
        ],
        "speedup_vs_native_plain_median": 0.08843037560147154
      },
      "8": {
        "samples": 6,
        "median_tokens_per_second": 1.2057910081239238,
        "median_ttft_seconds": 1.427953854,
        "p95_ttft_seconds": 1.551711375,
        "p95_is_exploratory": true,
        "paired_speedups": [
          0.16495979922501028,
          0.08934049662621371,
          0.05300395363219988,
          0.06300579752522639,
          0.03936981706704357,
          0.03950757862415118
        ],
        "speedup_vs_native_plain_median": 0.05706084374193268
      }
    }
  },
  "ordinary_comparison": [
    {
      "prompt_index": 0,
      "repeat": 0,
      "native_count_including_terminal_eos": 17,
      "native_terminal_eos_removed": true,
      "native_count_excluding_terminal_eos": 16,
      "ordinary_eval_count": 16,
      "normalized_count_match": true,
      "output_bytes_match": true
    },
    {
      "prompt_index": 0,
      "repeat": 1,
      "native_count_including_terminal_eos": 17,
      "native_terminal_eos_removed": true,
      "native_count_excluding_terminal_eos": 16,
      "ordinary_eval_count": 16,
      "normalized_count_match": true,
      "output_bytes_match": true
    },
    {
      "prompt_index": 1,
      "repeat": 0,
      "native_count_including_terminal_eos": 64,
      "native_terminal_eos_removed": false,
      "native_count_excluding_terminal_eos": 64,
      "ordinary_eval_count": 64,
      "normalized_count_match": true,
      "output_bytes_match": false
    },
    {
      "prompt_index": 1,
      "repeat": 1,
      "native_count_including_terminal_eos": 64,
      "native_terminal_eos_removed": false,
      "native_count_excluding_terminal_eos": 64,
      "ordinary_eval_count": 64,
      "normalized_count_match": true,
      "output_bytes_match": false
    },
    {
      "prompt_index": 2,
      "repeat": 0,
      "native_count_including_terminal_eos": 58,
      "native_terminal_eos_removed": true,
      "native_count_excluding_terminal_eos": 57,
      "ordinary_eval_count": 57,
      "normalized_count_match": true,
      "output_bytes_match": true
    },
    {
      "prompt_index": 2,
      "repeat": 1,
      "native_count_including_terminal_eos": 58,
      "native_terminal_eos_removed": true,
      "native_count_excluding_terminal_eos": 57,
      "ordinary_eval_count": 57,
      "normalized_count_match": true,
      "output_bytes_match": true
    }
  ],
  "ordinary_resident_median_tokens_per_wall_second": 46.57067096745669,
  "ordinary_first_cold_load_seconds": 3.717410417,
  "limits": [
    "Counts normalized by removing native terminal EOS only; ordinary token IDs are unavailable.",
    "Native visible bytes were decoded after the baseline run, with terminal EOS removed; original timing is unchanged.",
    "Six exploratory short samples per engine are not a balanced performance qualification.",
    "Divergent outputs cannot qualify a speedup.",
    "Ordinary assistant activation is uninstrumented.",
    "Same-prefix logit samples do not identify an exact kernel or prove all cache paths correct."
  ],
  "promotion": false,
  "experiment_date": "2026-09-30",
  "strict_failure_count": 4,
  "ordinary_aggregate": {
    "comparisons": 6,
    "normalized_token_count_matches": 6,
    "visible_output_byte_matches": 4,
    "sampled_token_ids_available": false
  },
  "corrected_workload": {
    "synthetic_prompt_count": 3,
    "repeats": 2,
    "generation_cap_tokens": 64,
    "draft_lengths": [
      2,
      4,
      8
    ],
    "actual_rtx_trial_comparisons": 18,
    "forced_oracle_trial_comparisons": 12,
    "initial_native_bos_policy": false,
    "raw_fixture_requires_literal_bos": true,
    "json_only_format_compliance": false,
    "code_explanation_reaches_generation_cap": true
  },
  "rate_scope": {
    "native_and_rtx": "Client wall generation includes prefill, target RPC, draft/tokenizer requests, commits and final decode; excludes model startup; count includes native terminal EOS when emitted.",
    "ordinary_resident": "Five resident calls after the first cold call; actual engine eval_count per client wall second. Ordinary eval_count excludes terminal EOS; cache and prefill behavior differ.",
    "ratio_of_medians": "speedup_vs_native_plain_median is each draft-path median divided by native plain median, not the median paired ratio.",
    "qualified_speedup": false,
    "gpu_kernel_time": "Not synchronously instrumented; null fields are not measured zero.",
    "first_token": "First committed token delivery as observed by the client; buffered committed IDs share arrival timestamps."
  },
  "public_sources": {
    "validation_plan": "https://llmsx.org/downloads/benchmarks/speculative-decoding-2026-09-30/validation-plan.md",
    "block_parity": "https://llmsx.org/downloads/benchmarks/speculative-decoding-2026-09-30/block-parity.md",
    "experiment_bundle": "https://llmsx.org/downloads/benchmarks/speculative-decoding-2026-09-30/experiment-bundle.zip",
    "article": "https://llmsx.org/blog/speculative-decoding-apple-silicon-rtx-5080/"
  },
  "source_provenance": {
    "repository": "https://github.com/mithudso/llms-explorer",
    "local_record_commit": "239f4ed2c88c3c334e41978d9656de2bd3e1d07e",
    "commit_publication": "Local source snapshot identifier, not a claim that this commit is present on the public remote. Complete scoped source overlay is supplied in the ZIP.",
    "llmsx_version": "0.2.4",
    "native_sidecar_version": "0.1.0",
    "canonical_model": "llmsx-research-gemma31-mlx",
    "imported_ollama_source_revision": "cc4069396f3ad2c370c53eed2e4a42ac13adab84",
    "canonical_manifest_sha256": "559e7d6999dbf031c7f842e84257eea6d43a8adcb6ff90a727baa05d4a5e6b0d",
    "canonical_tokenizer_fingerprint": "5f29b160b4d08cc73236f10311cf16d766edd54421d54617ca5ffe74546a6e4f",
    "recorded_loaded_mlx_version": "0.32.2-65-g59d600b"
  },
  "hash_semantics": {
    "sources_field": "SHA256 of the original committed source receipt bytes, before publication sanitization. These values do not validate sanitized public files.",
    "manifest": "The ZIP manifest supplies raw_source_sha256 and public_sha256 per file. public_sha256 validates extracted bytes.",
    "excluded": "Per-run and session identifiers, raw thinking/reasoning fields, credential fields and private runtime paths. Numeric thinking_bytes counts remain as timing/byte-count metadata.",
    "model_fingerprints": "Model/tokenizer hashes are artifact and token-semantics identifiers; neither proves numerically identical block and single-token execution."
  }
}
