{
  "schema_version": "1",
  "model": {
    "id": "mlx-community/OLMoE-1B-7B-0125-Instruct-4bit",
    "revision": "e23844197887b031e7ddddbb0b8959c5a6853a7b",
    "architecture": "olmoe",
    "quantization": {
      "group_size": 64,
      "bits": 4
    },
    "n_experts": 64,
    "top_k": 8,
    "moe_layers": 16
  },
  "strata": {
    "commit": "6562877",
    "mlx_lm": "0.31.3"
  },
  "mode": "resident",
  "measured_on": {
    "chip": "Apple M5 Max",
    "ram_gb": 137.4,
    "os": "macOS 26.5.1",
    "mlx_lm": "0.31.3"
  },
  "workload": {
    "prompt_set": "bench/mixed-v1",
    "prompt": "Explain how a CPU pipeline works.",
    "generated_tokens": 140,
    "warmup_tokens": 40,
    "decoding": "greedy"
  },
  "memory": {
    "target_ram_gb": 16.0,
    "reserve_gb": 4.0,
    "available_gb": 12.0,
    "enforced_memory_gb": null,
    "full_footprint_gb": 3.89,
    "measured_peak_gb": 7.38,
    "peak_rss_gb": 1.05,
    "swap_growth_gb": 0.0,
    "compression_events": 0,
    "expert_budget": 64,
    "working_set_gb": 3.89,
    "mode": "resident"
  },
  "throughput": {
    "tokens_per_s": 162.49,
    "p50_ms": 5.8,
    "p95_ms": 7.8,
    "ssd_reads_mb_per_tok": 1.5
  },
  "fidelity": {
    "method": "decode-mode teacher-forced, paged vs resident",
    "prompt_set": "arch_fidelity/mixed-v1",
    "budget": 8,
    "positions": 93,
    "agreement_pct": 100.0,
    "perplexity_delta_pct": 0.0,
    "high_confidence_flips": 0,
    "max_abs_dlogit": 0.0,
    "mean_kl": 0.0,
    "bit_exact": true
  },
  "honesty": {
    "device_real": false,
    "physical_constraint_validated": false,
    "measurement_type": "development-host benchmark",
    "memory_peak_scope": "load-through-generation",
    "throughput_is_optimistic": true
  },
  "sample_output": "A CPU pipeline is a fundamental concept in the design of modern microprocessors that allows for the execution of multiple instructions simultaneously, improving the efficiency and performance of the processor. Here's how it works:\n\n1. **Instruction Queue**: At any given time, the CPU has a limited number of instructions that it can process. This queue holds instructions waiting to be executed.\n\n2. **Fetch Cycle**: The CPU starts by fetching instructions from the memory. This process involves reading the instruction from the memory and preparing it for execution.\n\n3. **Decode Cycle**: While the"
}