{
  "label": "GLM-5.3-Flash UD-IQ3_XXS (120.4 GB) reasoning HIGH",
  "tests": {
    "01-blockfall": {
      "error": "DNF: exceeded the 60 minute time limit (partial output kept: 22942 chunks, 63743 thinking chars, 14460 answer chars)",
      "partial_thinking_chars": 63743,
      "partial_answer_chars": 14460,
      "partial_chunks": 22942
    },
    "03-eruption": {
      "error": "DNF: exceeded the 60 minute time limit (partial output kept: 23010 chunks, 78082 thinking chars, 0 answer chars)",
      "partial_thinking_chars": 78082,
      "partial_answer_chars": 0,
      "partial_chunks": 23010
    },
    "04-the-ledger": {
      "error": "DNF: exceeded the 60 minute time limit (partial output kept: 22790 chunks, 88665 thinking chars, 0 answer chars)",
      "partial_thinking_chars": 88665,
      "partial_answer_chars": 0,
      "partial_chunks": 22790
    },
    "05-blind-artist": {
      "error": "DNF: exceeded the 60 minute time limit (partial output kept: 23199 chunks, 84663 thinking chars, 0 answer chars)",
      "partial_thinking_chars": 84663,
      "partial_answer_chars": 0,
      "partial_chunks": 23199
    }
  },
  "vision": {},
  "size_gb_on_disk": 120.4,
  "load_seconds": 35.0,
  "warmup_seconds": 1.7,
  "mem_baseline_gb": 4.52,
  "mem_peak_gb": 119.2,
  "mem_model_gb_est": 114.68,
  "backend": "llama-server direct: ~/repo/tools/llama.cpp-glm5next/build/bin/llama-server",
  "server_timings": [
    {
      "task": 0,
      "prompt_ms": 867.7,
      "prompt_tokens": 15,
      "prefill_tok_s": 17.29,
      "eval_ms": 787.67,
      "eval_tokens": 8,
      "gen_tok_s": 8.89
    },
    {
      "task": 11
    },
    {
      "task": 23035
    },
    {
      "task": 46100
    },
    {
      "task": 69000
    }
  ]
}