{
  "version": "1.0.0",
  "delta": "Consolidate recorded and reported results through October5; no new inference.",
  "as_of": "2026-10-05",
  "new_model_requests": 0,
  "mlx_recorded": {
    "kind": "mixed-provenance research phases",
    "model": "gemma4:31b-mlx",
    "context": 65536,
    "phases": [
      {
        "phase": "fresh one-concept worker",
        "fixture": "fresh standard-contract worker probe",
        "seconds": 243,
        "outcome": "Six claims recorded; incorrect archival-age bound required correction."
      },
      {
        "phase": "evidence-directed correction",
        "fixture": "same fresh one-concept worker probe",
        "seconds": 293.349,
        "outcome": "All six corrected facts independently reviewed.",
        "actual_tool_calls": {
          "search": 1,
          "scrape": 4,
          "source_extraction": 8,
          "claims_write": 2,
          "concept_done": 2
        }
      },
      {
        "phase": "fresh blind verification gate",
        "fixture": "original five-concept Qwen research artifact",
        "seconds": 315,
        "sampled": 10,
        "counts": {
          "SUPPORTED": 10
        },
        "scrapes": 8,
        "additional_review_fetches": 0,
        "coverage": {
          "date-criteria-field-types-and-formats": 2,
          "date-criteria-date-format-specifications": 2,
          "expireafterdays-data-age-calculation": 2,
          "partition-fields-constraints-for-date-rules": 2,
          "date-vs-custom-criteria-selection": 2
        }
      },
      {
        "phase": "Explorer canonical completion",
        "fixture": "original five-concept Qwen research artifact",
        "seconds": 41.445,
        "turns": 2,
        "status": "ok",
        "returncode": 0,
        "canonical_status": "COMPLETED"
      }
    ],
    "provenance": {
      "original_research_model": "GGUF Qwen27",
      "original_correction_model": "Qwen MLX",
      "original_completed_concepts": 5,
      "original_claims": 39,
      "original_source_urls": 16,
      "gemma31_fresh_research_workers": 1,
      "full_fresh_five_worker_gemma_benchmark": false,
      "phase_seconds_are_not_one_full_run_total": true,
      "model_inference": "local; Claude Code provided the tool loop",
      "internet_retrieval": "Firecrawl MCP; local inference does not imply offline retrieval",
      "separate_source_review": "Codex reviewed full fetched pages outside the timed local research phases."
    },
    "probes": [
      {
        "model": "gemma4:12b-mlx",
        "passed": true,
        "generated_tokens": 88,
        "decode_tokens_per_second": 100.62
      },
      {
        "model": "gemma4:26b-mlx",
        "passed": true,
        "generated_tokens": 81,
        "decode_tokens_per_second": 54.68
      },
      {
        "model": "qwen3.6:35b-mlx",
        "passed": true,
        "generated_tokens": 111,
        "decode_tokens_per_second": 79.91
      },
      {
        "model": "llmsx-research-mlx",
        "passed": true,
        "generated_tokens": 112,
        "decode_tokens_per_second": 113.04
      },
      {
        "model": "llmsx-research-gemma31-mlx",
        "passed": true,
        "generated_tokens": 102,
        "decode_tokens_per_second": 54.76
      }
    ]
  },
  "mlx_reported_rerun": {
    "raw_response_counters_supplied": false,
    "models": [
      {
        "model": "gemma4:12b-mlx",
        "cold_load_seconds": 1.424,
        "warm_load_seconds_range": [
          0.017,
          0.064
        ],
        "prompt_eval_tokens_per_second_range": [
          390.58,
          390.58
        ],
        "reported_ttft_seconds": 0.309,
        "short_generation_tokens_per_second_range": [
          67.42,
          96.5
        ],
        "sustained_generation_tokens_per_second": 63.75,
        "reported_weight_footprint_gb": 7.7,
        "estimated_footprint_with_16k_kv_gb": 8.9,
        "sustained_generated_tokens_note": "2,600+ output tokens reported"
      },
      {
        "model": "gemma4:26b-mlx",
        "cold_load_seconds": 2.633,
        "warm_load_seconds_range": [
          0.01,
          0.019
        ],
        "prompt_eval_tokens_per_second_range": [
          78.86,
          83.87
        ],
        "reported_ttft_seconds": 1.465,
        "short_generation_tokens_per_second_range": [
          126.97,
          130.27
        ],
        "sustained_generation_tokens_per_second": 89.42,
        "reported_weight_footprint_gb": 18.0,
        "estimated_footprint_with_16k_kv_gb": 20.8,
        "sustained_generated_tokens_note": "2,600+ output tokens reported"
      },
      {
        "model": "llmsx-research-gemma31-mlx",
        "cold_load_seconds": 4.635,
        "warm_load_seconds_range": [
          0.02,
          0.064
        ],
        "prompt_eval_tokens_per_second_range": [
          123.59,
          123.59
        ],
        "reported_ttft_seconds": 0.944,
        "short_generation_tokens_per_second_range": [
          25.37,
          29.07
        ],
        "sustained_generation_tokens_per_second": 27.39,
        "reported_weight_footprint_gb": 19.0,
        "estimated_footprint_with_16k_kv_gb": 22.0,
        "sustained_generated_tokens_note": "2,758+ output tokens reported"
      }
    ],
    "limits": [
      "TTFT values are preserved as reported early-response estimates pending client-observed streamed timestamps.",
      "Decode workload, loading and cache state differ from the native two-tool probes. Do not pool the measurements.",
      "Long generated output does not establish long-input-context throughput.",
      "Weight and KV figures are sizing estimates, not peak resident memory traces.",
      "No NVIDIA or RTX 5080 deployment was demonstrated by these trials.",
      "Gemma26 is a mixture-of-experts model. The reported trial does not isolate layer pruning, kernels or speculative decoding.",
      "Apple specifies 460 or 614 GB/s for M5 Max variants. The reported 400 GB/s is not an independently measured bandwidth ceiling.",
      "Short generation performance does not establish research correctness. Gemma31 remains the canonical model selected from reviewed workflow results."
    ]
  },
  "historical_physical": {
    "4b": {
      "stream_decode_range": [
        196.291,
        196.332
      ],
      "stream_wall_range": [
        7.976,
        8.032
      ],
      "stream_scope": "Three1565-token warm responses with43of44prompt tokens reported cached.",
      "temperature_pilot_raw_passed": 2,
      "temperature_pilot_calibrated_passed": 3,
      "temperature_pilot_runs": 4
    },
    "9b": {
      "coding_passed_runs": 2,
      "coding_wall_seconds": [
        48.701,
        51.992
      ],
      "fresh_full_wire_schema_capture": false,
      "standard_dr_accepted": false
    },
    "qwen35_27b": {
      "coding_passed_runs": 2,
      "coding_wall_seconds": [
        94.963,
        107.845
      ],
      "coding_native_decode": 53.95,
      "numerical_max_error": 0.078459829,
      "numerical_limit": 0.05,
      "terminal_all_five_attempted": true,
      "terminal_accepted_concepts": 0,
      "terminal_experiment_seconds": 8776.973867292007,
      "terminal_helper_minutes": 149,
      "terminal_finish": "BLOCKED",
      "terminal_archive_sha256": "400ea2206d22a99660477a4e6ff554c4e0f5d07b1b054b2f38f6a6fe31f195e8"
    }
  },
  "context_projections": {
    "kind": "CPU vocabulary/template projections only",
    "context": 32768,
    "output_reserve": 4096,
    "coding_original": [
      20093,
      20130
    ],
    "coding_compact": [
      7862,
      7899
    ],
    "research_initial_range": [
      7599,
      7625
    ],
    "selected_history_range": [
      14798,
      16731
    ],
    "largest_spare": 11941,
    "exhaustive_turn_maximum": false
  },
  "qwen36_current": {
    "model": "Qwen3.6-27B IQ2_XXS",
    "model_bytes": 9605378560,
    "context": 32768,
    "kv": "f16",
    "mtp": false,
    "f32_result": {
      "version": "1.0.1",
      "passed": false,
      "requests": 1,
      "starts": 0,
      "resets": 0,
      "automatic_retry": false,
      "functional_passed": true,
      "numerical_parity_verified": false,
      "same_owner_retained": true,
      "actual_peak_verified": false,
      "actual_placement_verified": false,
      "coding_verified": false,
      "standard_research_verified": false,
      "full_qualification": false,
      "errors": [],
      "request_wall_seconds": 4.317360582999754,
      "timings": {
        "cache_n": 0,
        "prompt_n": 64,
        "prompt_ms": 3976.881,
        "prompt_per_token_ms": 62.138765625,
        "prompt_per_second": 16.093013595327594,
        "predicted_n": 21,
        "predicted_ms": 334.442,
        "predicted_per_token_ms": 16.7221,
        "predicted_per_second": 59.80110153629029
      },
      "maximum_absolute_error": 0.08756500482559204,
      "elapsed_seconds": 4.7973948329999985
    },
    "numerical": {
      "functional_passed": true,
      "comparison_passed": false,
      "declared_atol": 0.05,
      "compared_token_count": 21,
      "all_token_identities_match": true,
      "content_matches": true,
      "maximum_absolute_error": 0.08756500482559204,
      "mean_absolute_error": 0.006143957763275206
    },
    "previous_non_f32": {
      "prefill_tokens_per_second": 289.9273,
      "decode_tokens_per_second": 59.0978,
      "wall_seconds": 0.561855291,
      "max_error": 0.06316304206848145
    },
    "whole_output_f32_bytes": 5085593600,
    "workspace_bytes": 2147483648,
    "mismatch_cause_proven": false
  },
  "remaining": [
    "precision remediation without changing0.05",
    "aggregate physical peak and placement",
    "fresh current-candidate full coding pair",
    "repeated stability",
    "genuine standard DR and canonical artifacts",
    "final deployment"
  ]
}
