{
  "generated_at": "2026-07-27T20:03:52Z",
  "source_generated_at": "2026-07-27T20:01:06Z",
  "article_slug": "spark-open-model-bakeoff-july-2026",
  "article_url": "https://chat.neonflux.co/mission-control/spark-open-model-bakeoff-july-2026/",
  "pdf_url": "https://chat.neonflux.co/downloads/mission-control/spark-open-model-bakeoff-20260727/committee-review-spark-open-model-bakeoff-20260727.pdf",
  "artifact_zip_url": "https://chat.neonflux.co/downloads/mission-control/spark-open-model-bakeoff-20260727/spark-open-model-bakeoff-20260727-artifacts.zip",
  "rankings": {
    "recommended_default": "Nemotron Super",
    "recommended_secondary": "GLM-4.5-Air Q6_K",
    "recommended_specialty": "Kimi Dev 72B Q4_0",
    "fastest_cold_start_model": "Nemotron Super",
    "fastest_cold_start_s": 2.654289750382304,
    "slowest_cold_start_model": "GLM-4.5-Air Q6_K",
    "slowest_cold_start_s": 358.9872544594109
  },
  "dual_probe": {
    "model_key": "dual_ollama",
    "model_label": "GLM-4.5-Air Q6_K + Kimi Dev 72B Q4_0",
    "model_family": "Dual Ollama Probe",
    "scenario_key": "dual_ollama_cold_load",
    "scenario_label": "Dual Ollama Cold Load",
    "concurrency": 2,
    "request_count": 2,
    "success_count": 2,
    "failure_count": 0,
    "throughput_rps": 0.0056454044733282045,
    "latency_avg_s": 191.61148900026456,
    "latency_p50_s": 191.61148900026456,
    "latency_p95_s": 338.006921275286,
    "latency_max_s": 354.27308041695505,
    "wall_time_s": 354.2704529762268,
    "avg_completion_tokens": 128.0,
    "avg_prompt_tokens": 34.5,
    "avg_tokens_per_s": 2.3913675416414693,
    "aggregate_tokens_per_s": 0.7226117725860102,
    "peak_cpu_busy_pct": 51.7,
    "min_cpu_busy_pct": 0.9000000000000057,
    "avg_cpu_busy_pct": 12.57927927927928,
    "peak_load1": 2.86,
    "min_load1": 1.25,
    "avg_load1": 2.094324324324324,
    "peak_mem_used_mb": 97129,
    "min_mem_used_mb": 3111,
    "avg_mem_used_mb": 74713.05405405405,
    "peak_mem_available_mb": 121499,
    "min_mem_available_mb": 27481,
    "avg_mem_available_mb": 49896.71171171171,
    "peak_gpu_temp_c": 71.0,
    "min_gpu_temp_c": 52.0,
    "avg_gpu_temp_c": 56.8018018018018,
    "peak_gpu_util_pct": 96.0,
    "min_gpu_util_pct": 0.0,
    "avg_gpu_util_pct": 12.18018018018018,
    "peak_gpu_power_w": 41.61,
    "min_gpu_power_w": 14.64,
    "avg_gpu_power_w": 18.47936936936937
  },
  "co_resident_failure": {
    "at": "2026-07-27T19:04:26Z",
    "model": "glm45air-q6",
    "error": "CUDA error: out of memory while TRT services were resident",
    "details": "GLM failed to start co-resident on Spark while trtllm-qwen3-8b and trtllm-nemotron3-super were active. GLM and Kimi were rerun in an isolated Ollama window after those TRT services were stopped."
  },
  "cleanup": {
    "removed_model": "hf.co/mradermacher/Kimi-Linear-48B-A3B-Instruct-GGUF:Q8_0",
    "reclaimed_gb": 52
  }
}