[
  {
    "hardware": "API",
    "model": "GPT-5.6 Luna (OpenAI API)",
    "date": "2026-08-30",
    "method": "Throughput",
    "result": "93 tokens/s",
    "conditions": "2026-08-30, median of 5 \u00d7 600-token runs via :8010, reasoning disabled; observed range 92\u2013102 tok/s",
    "receipt": "metrics.json",
    "detail": "stack.html"
  },
  {
    "hardware": "CPU / Dell",
    "model": "Qwen3-30B-A3B-Instruct-2507",
    "date": "2026-08-24",
    "method": "Agent fitness",
    "result": "14.5 s median task",
    "conditions": "15 single-shot tasks; manually judged. 7/7 tool rubric. CPU-only llama.cpp, 6 threads, Q4_K_M. Follow-up multi-turn diagnosis was not viable.",
    "receipt": "https://github.com/randomchaos7800-hub/inference-research/blob/master/dell/kato-eval/VERDICT.md",
    "detail": "#cpu-agent-eval"
  },
  {
    "hardware": "CPU / Dell",
    "model": "Ornith-1.5-9B",
    "date": "2026-08-24",
    "method": "Agent fitness",
    "result": "34.3 s median task",
    "conditions": "Same 15-task single-shot evaluation; manually judged. 6/7 tool rubric. CPU-only llama.cpp, 6 threads, Q4_K_M.",
    "receipt": "https://github.com/randomchaos7800-hub/inference-research/blob/master/dell/kato-eval/VERDICT.md",
    "detail": "#cpu-agent-eval"
  },
  {
    "hardware": "Apple Silicon",
    "model": "Ornith-1.0-9B-4bit",
    "date": "2026-07-16",
    "method": "Quality + throughput",
    "receipt_label": "Published summary",
    "result": "18.66 tokens/s",
    "conditions": "Mac mini M4, 16GB unified memory, MLX. Internal domain-suite mean 3.0/5; protocol differs from GPU throughput runs. Historical model selection. Original external artifact unavailable; source link is the published summary.",
    "receipt": "https://boundarylabs.org/benchmarks.html#mac-mini-mlx",
    "detail": "#mac-mini-mlx"
  },
  {
    "hardware": "GPU / Blackwell",
    "model": "GPT-OSS-20B",
    "date": "2026-07-14",
    "method": "Quality + throughput",
    "result": "133.05 tokens/s peak",
    "conditions": "Native MXFP4, llama.cpp, all-GPU. Internal 15-scenario quality suite: 2.2/5. Peak speed is not a quality ranking. Tower-era result.",
    "receipt": "https://github.com/randomchaos7800-hub/inference-research/blob/master/tower/moe/FINAL-REPORT-2026-07-14.md",
    "detail": "#moe"
  },
  {
    "hardware": "GPU / Blackwell",
    "model": "Ornith-1.0-35B",
    "date": "2026-06-25",
    "method": "Throughput",
    "result": "129.9 tokens/s peak",
    "conditions": "GGUF Q4_K_M, short context, dual RTX 5060 Ti. Dated campaign peak; not current production or an all-time record across later campaigns.",
    "receipt": "https://github.com/randomchaos7800-hub/inference-research/blob/master/tower/ornith/suite-20260625-170030.json",
    "detail": "#campaigns"
  }
]
