{
 "schema": "local-ai-bench/result/v1",
 "run_id": "20261004T072115Z-ik_llama-glm-5.3-flash-iq2-m-128k-3x-g210-fm4096-gpu012-stability",
 "ticket": "T22c",
 "kind": "stability",
 "model": "glm-5.3-flash",
 "config_id": "ik_llama-glm-5.3-flash-iq2-m-128k-3x-g210-fm4096-gpu012",
 "outcome": "fail",
 "failed_stage": null,
 "exit_status": 0,
 "started_at": "2026-10-04T07:21:15.000857+00:00",
 "finished_at": "2026-10-04T07:31:18.888295+00:00",
 "launch": {
  "file": "launches/ik_llama-glm-5.3-flash-iq2-m-128k-3x-g210-fm4096.json",
  "sha256": "fe1c037653b578219259df5ac2f7fce9deed7442d4759cd749f0001ffe3df341",
  "image": "local-ai-bench/ik-llama-cuda@sha256:d30fa2796c72509c9158a6202fed46c8ce159f3344f3735120be65a73a8137b1",
  "argv": [
   "/app/llama-server",
   "--model",
   "/models/GLM-5.3-Flash-BF16-IQ2_M/GLM-5.3-Flash-BF16-IQ2_M-00001-of-00004.gguf",
   "--ctx-size",
   "131072",
   "--flash-attn",
   "on",
   "--mla-use",
   "3",
   "--dsa",
   "--cache-type-k",
   "q8_0",
   "--n-gpu-layers",
   "999",
   "--fit",
   "--split-mode",
   "layer",
   "--batch-size",
   "2048",
   "--ubatch-size",
   "2048",
   "--parallel",
   "1",
   "--threads",
   "15",
   "--jinja",
   "--metrics",
   "--host",
   "0.0.0.0",
   "--port",
   "8080",
   "--fit-margin",
   "4096"
  ],
  "env": {
   "CUDA_DEVICE_ORDER": "PCI_BUS_ID",
   "CUDA_VISIBLE_DEVICES": "2,1,0"
  },
  "engine": "ik_llama.cpp",
  "engine_source_commit": "https://github.com/ikawrakow/ik_llama.cpp/tree/5bf8f0fe4db98e9560de0f495232f8d3a733c678",
  "mtp": null,
  "entrypoint": "/app/llama-server",
  "port": 8080,
  "shm": "16g",
  "flags": [],
  "weights_at": [
   "/models"
  ]
 },
 "weights": [
  {
   "repo": "bartowski/GLM-5.3-Flash-BF16-GGUF",
   "revision": "66e5e9f0e337470e8801b99bae6c45779b042cc2",
   "manifest": "models/bartowski-GLM-5.3-Flash-IQ2_M.json"
  }
 ],
 "host": {
  "hostname": "omarchy-gpu",
  "cpu": "AMD Ryzen 9 5950X 16-Core Processor",
  "ram_total_gib": 125.7,
  "ram_layout": "DIMM_A1 32 GiB DDR4 3200 MT/s, DIMM_A2 32 GiB DDR4 3200 MT/s, DIMM_B1 32 GiB DDR4 3200 MT/s, DIMM_B2 32 GiB DDR4 3200 MT/s (AM4: 2 channels)",
  "kernel": "7.2.5-3-omarchy",
  "driver": "610.57.04",
  "cuda": "13.3",
  "cpus_online": "0-1,3-17,19-31",
  "cpus_offline": "2,18",
  "machine_checks_this_boot": 6
 },
 "gpus": [
  {
   "index": 0,
   "uuid": "GPU-816b43e4-8d65-dfd7-2234-8517b3dfbf2d",
   "name": "NVIDIA GeForce RTX 3090",
   "bus_id": "00000000:04:00.0",
   "pcie_gen_max": 4,
   "pcie_width": 8,
   "display_active": false,
   "used_by_run": true
  },
  {
   "index": 1,
   "uuid": "GPU-c67ac872-3371-a88c-aa92-d969daf6405e",
   "name": "NVIDIA GeForce RTX 3090",
   "bus_id": "00000000:0B:00.0",
   "pcie_gen_max": 4,
   "pcie_width": 8,
   "display_active": false,
   "used_by_run": true
  },
  {
   "index": 2,
   "uuid": "GPU-76d3c6af-7f18-69d2-3145-899103de1722",
   "name": "NVIDIA GeForce RTX 3090",
   "bus_id": "00000000:0C:00.0",
   "pcie_gen_max": 4,
   "pcie_width": 8,
   "display_active": false,
   "used_by_run": true
  }
 ],
 "layout": {
  "name": "current",
  "cards": 3,
  "display": "off"
 },
 "context": {
  "configured": 131072,
  "occupied_max": {
   "value": 1874,
   "unit": "tokens",
   "detail": {
    "source": "max usage.prompt_tokens over the run"
   }
  },
  "headroom_min": {
   "value": 129132,
   "unit": "tokens",
   "detail": {
    "configured": 131072,
    "configured_source": "env CONFIGURED_CTX",
    "max_prompt_plus_completion": 1940
   }
  }
 },
 "cache_state": {
  "cold": false,
  "reset": []
 },
 "provenance": {
  "bench_commit": "981b077ba6f48c9b94c9bc0a253f95413b94b11e",
  "bench_dirty": true,
  "pins_sha256": "858564b6e07125ac48c2ef6c5178d0a53596f21eb5f0ebda43906472b01b4d78",
  "registry_commit": "d21258dd744e7c78be28177c90060c6af8e10b7e",
  "kit_commit": "ef883d269e50ecbea290f919f095e2f3ca633b42",
  "harness_commit": "04d809ceab9df28f9adaed044884180159172930"
 },
 "metrics": {
  "duration_s": {
   "value": 600.0,
   "unit": "s",
   "detail": {
    "cap_s": 600.0,
    "driver_capped": true
   }
  },
  "tool_calls": {
   "value": 0,
   "unit": "calls",
   "detail": {
    "per_min": 0.0,
    "requests": 68
   }
  },
  "parse_failures": {
   "value": 68,
   "unit": "responses",
   "detail": {
    "wire": {
     "unparsed_markup": 68
    },
    "harness_format_errors": {
     "no_tool_call": 68
    },
    "union_rule": "a response counts once even if both the wire check and the harness flag it",
    "counted": "malformed tool calls only (decision 0003); truncations and other format errors are separate metrics"
   }
  },
  "output_truncations": {
   "value": 0,
   "unit": "responses",
   "detail": {
    "rule": "cut off at max_tokens with no tool call"
   }
  },
  "agent_format_errors": {
   "value": 68,
   "unit": "responses",
   "detail": {
    "rule": "agent protocol errors other than malformed tool calls"
   }
  },
  "repetition_hits": {
   "value": 0,
   "unit": "hits",
   "detail": {
    "ngram64x3": 0,
    "same_call_x4": 0,
    "ngram64x3_cross_field_not_counted": 0,
    "tokenizer": "server /tokenize",
    "saved": "artifacts: proxy/repetition/"
   }
  },
  "decode_window_min_tps": {
   "value": 13.255,
   "unit": "tok/s",
   "detail": {
    "window_s": 60,
    "skip_s": 60,
    "rolling_step_s": 1,
    "n_rolling_windows": 299,
    "min_at_generation_s": 112.0,
    "windows_tps": [
     13.31,
     13.29,
     13.29,
     13.35,
     13.34
    ],
    "windows_tps_note": "consecutive non-overlapping 60 s windows; the value is the minimum over rolling windows (1 s step)"
   }
  },
  "decode_whole_run_tps": {
   "value": 13.319,
   "unit": "tok/s",
   "detail": {
    "generation_seconds": 418.663,
    "generated_tokens": 5576.2
   }
  },
  "tasks_attempted": {
   "value": 23,
   "unit": "attempts",
   "detail": {
    "unfinished_at_cap": 1,
    "task_ids": [
     "django__django-11099",
     "sympy__sympy-21612",
     "psf__requests-2317",
     "matplotlib__matplotlib-23299",
     "scikit-learn__scikit-learn-13439",
     "astropy__astropy-12907",
     "pytest-dev__pytest-7373",
     "sympy__sympy-20590"
    ],
    "task_set_sha256": "8e7487ca7d059df6359fc7d9b86c7021256a8a81b8af049616750f8f83400324"
   }
  },
  "tasks_solved": {
   "value": 0,
   "unit": "attempts",
   "detail": {
    "eval_errors": 0,
    "swebench": "5.0.2"
   }
  },
  "crash_or_oom": {
   "value": false,
   "unit": "bool",
   "detail": {
    "upstream_connection_failures": 0,
    "upstream_http_errors": 0,
    "source": "proxy-observed; the run wrapper adds container death/OOM"
   }
  },
  "gates_after": {
   "state": "skipped",
   "reason": "filled by qualify step: the lab run that follows this run (tools/qualify.py)"
  },
  "agent_attempts": {
   "value": [
    {
     "attempt": 1,
     "round": 1,
     "instance_id": "django__django-11099",
     "exit_status": "RepeatedFormatError",
     "submitted": false,
     "resolved": false,
     "seconds": 30.9,
     "requests": 3,
     "tool_calls": 0,
     "completion_tokens": 246,
     "occupied_max": 1559
    },
    {
     "attempt": 2,
     "round": 1,
     "instance_id": "sympy__sympy-21612",
     "exit_status": "RepeatedFormatError",
     "submitted": false,
     "resolved": false,
     "seconds": 27.9,
     "requests": 3,
     "tool_calls": 0,
     "completion_tokens": 217,
     "occupied_max": 1641
    },
    {
     "attempt": 3,
     "round": 1,
     "instance_id": "psf__requests-2317",
     "exit_status": "RepeatedFormatError",
     "submitted": false,
     "resolved": false,
     "seconds": 37.7,
     "requests": 3,
     "tool_calls": 0,
     "completion_tokens": 348,
     "occupied_max": 1609
    },
    {
     "attempt": 4,
     "round": 1,
     "instance_id": "matplotlib__matplotlib-23299",
     "exit_status": "RepeatedFormatError",
     "submitted": false,
     "resolved": false,
     "seconds": 31.6,
     "requests": 3,
     "tool_calls": 0,
     "completion_tokens": 263,
     "occupied_max": 1874
    },
    {
     "attempt": 5,
     "round": 1,
     "instance_id": "scikit-learn__scikit-learn-13439",
     "exit_status": "RepeatedFormatError",
     "submitted": false,
     "resolved": false,
     "seconds": 31.2,
     "requests": 3,
     "tool_calls": 0,
     "completion_tokens": 260,
     "occupied_max": 1770
    },
    {
     "attempt": 6,
     "round": 1,
     "instance_id": "astropy__astropy-12907",
     "exit_status": "RepeatedFormatError",
     "submitted": false,
     "resolved": false,
     "seconds": 25.1,
     "requests": 3,
     "tool_calls": 0,
     "completion_tokens": 181,
     "occupied_max": 1716
    },
    {
     "attempt": 7,
     "round": 1,
     "instance_id": "pytest-dev__pytest-7373",
     "exit_status": "RepeatedFormatError",
     "submitted": false,
     "resolved": false,
     "seconds": 27.6,
     "requests": 3,
     "tool_calls": 0,
     "completion_tokens": 214,
     "occupied_max": 1641
    },
    {
     "attempt": 8,
     "round": 1,
     "instance_id": "sympy__sympy-20590",
     "exit_status": "RepeatedFormatError",
     "submitted": false,
     "resolved": false,
     "seconds": 39.7,
     "requests": 3,
     "tool_calls": 0,
     "completion_tokens": 375,
     "occupied_max": 1578
    },
    {
     "attempt": 9,
     "round": 2,
     "instance_id": "django__django-11099",
     "exit_status": "RepeatedFormatError",
     "submitted": false,
     "resolved": false,
     "seconds": 23.2,
     "requests": 3,
     "tool_calls": 0,
     "completion_tokens": 240,
     "occupied_max": 1559
    },
    {
     "attempt": 10,
     "round": 2,
     "instance_id": "sympy__sympy-21612",
     "exit_status": "RepeatedFormatError",
     "submitted": false,
     "resolved": false,
     "seconds": 21.1,
     "requests": 3,
     "tool_calls": 0,
     "completion_tokens": 212,
     "occupied_max": 1641
    },
    {
     "attempt": 11,
     "round": 2,
     "instance_id": "psf__requests-2317",
     "exit_status": "RepeatedFormatError",
     "submitted": false,
     "resolved": false,
     "seconds": 26.7,
     "requests": 3,
     "tool_calls": 0,
     "completion_tokens": 286,
     "occupied_max": 1609
    },
    {
     "attempt": 12,
     "round": 2,
     "instance_id": "matplotlib__matplotlib-23299",
     "exit_status": "RepeatedFormatError",
     "submitted": false,
     "resolved": false,
     "seconds": 25.0,
     "requests": 3,
     "tool_calls": 0,
     "completion_tokens": 262,
     "occupied_max": 1874
    },
    {
     "attempt": 13,
     "round": 2,
     "instance_id": "scikit-learn__scikit-learn-13439",
     "exit_status": "RepeatedFormatError",
     "submitted": false,
     "resolved": false,
     "seconds": 24.6,
     "requests": 3,
     "tool_calls": 0,
     "completion_tokens": 257,
     "occupied_max": 1770
    },
    {
     "attempt": 14,
     "round": 2,
     "instance_id": "astropy__astropy-12907",
     "exit_status": "RepeatedFormatError",
     "submitted": false,
     "resolved": false,
     "seconds": 19.0,
     "requests": 3,
     "tool_calls": 0,
     "completion_tokens": 184,
     "occupied_max": 1716
    },
    {
     "attempt": 15,
     "round": 2,
     "instance_id": "pytest-dev__pytest-7373",
     "exit_status": "RepeatedFormatError",
     "submitted": false,
     "resolved": false,
     "seconds": 20.8,
     "requests": 3,
     "tool_calls": 0,
     "completion_tokens": 208,
     "occupied_max": 1641
    },
    {
     "attempt": 16,
     "round": 2,
     "instance_id": "sympy__sympy-20590",
     "exit_status": "RepeatedFormatError",
     "submitted": false,
     "resolved": false,
     "seconds": 29.2,
     "requests": 3,
     "tool_calls": 0,
     "completion_tokens": 319,
     "occupied_max": 1578
    },
    {
     "attempt": 17,
     "round": 3,
     "instance_id": "django__django-11099",
     "exit_status": "RepeatedFormatError",
     "submitted": false,
     "resolved": false,
     "seconds": 23.2,
     "requests": 3,
     "tool_calls": 0,
     "completion_tokens": 240,
     "occupied_max": 1559
    },
    {
     "attempt": 18,
     "round": 3,
     "instance_id": "sympy__sympy-21612",
     "exit_status": "RepeatedFormatError",
     "submitted": false,
     "resolved": false,
     "seconds": 21.1,
     "requests": 3,
     "tool_calls": 0,
     "completion_tokens": 212,
     "occupied_max": 1641
    },
    {
     "attempt": 19,
     "round": 3,
     "instance_id": "psf__requests-2317",
     "exit_status": "RepeatedFormatError",
     "submitted": false,
     "resolved": false,
     "seconds": 26.8,
     "requests": 3,
     "tool_calls": 0,
     "completion_tokens": 286,
     "occupied_max": 1609
    },
    {
     "attempt": 20,
     "round": 3,
     "instance_id": "matplotlib__matplotlib-23299",
     "exit_status": "RepeatedFormatError",
     "submitted": false,
     "resolved": false,
     "seconds": 25.0,
     "requests": 3,
     "tool_calls": 0,
     "completion_tokens": 262,
     "occupied_max": 1874
    },
    {
     "attempt": 21,
     "round": 3,
     "instance_id": "scikit-learn__scikit-learn-13439",
     "exit_status": "RepeatedFormatError",
     "submitted": false,
     "resolved": false,
     "seconds": 24.6,
     "requests": 3,
     "tool_calls": 0,
     "completion_tokens": 257,
     "occupied_max": 1770
    },
    {
     "attempt": 22,
     "round": 3,
     "instance_id": "astropy__astropy-12907",
     "exit_status": "RepeatedFormatError",
     "submitted": false,
     "resolved": false,
     "seconds": 19.0,
     "requests": 3,
     "tool_calls": 0,
     "completion_tokens": 184,
     "occupied_max": 1716
    },
    {
     "attempt": 23,
     "round": 3,
     "instance_id": "pytest-dev__pytest-7373",
     "exit_status": "WallClockCap",
     "submitted": false,
     "resolved": false,
     "seconds": 19.0,
     "requests": 3,
     "tool_calls": 0,
     "completion_tokens": 133,
     "occupied_max": 1522
    }
   ],
   "unit": "list"
  },
  "tasks_over_64k": {
   "value": [],
   "unit": "instance ids"
  },
  "completion_tokens": {
   "value": 5646,
   "unit": "tokens",
   "detail": {
    "finish_reasons": {
     "stop": 68
    }
   }
  },
  "mtp_accepted_tps": {
   "state": "skipped",
   "reason": "speculative decoding off in this launch"
  },
  "mtp_acceptance_rate": {
   "state": "skipped",
   "reason": "speculative decoding off in this launch"
  }
 },
 "eligibility": "none",
 "artifacts": [
  "artifacts/runs/20261004T072115Z-lab-ik_llama-glm-5.3-flash-iq2-m-128k-3x-g210-fm4096-gpu012-stability"
 ],
 "notes": "stability run, 10 min cap, floor 15.0 tok/s; mini-swe-agent 04d809ceab9d (benchmarks/swebench.yaml, native tool calls); sampling {\"temperature\": 0.7, \"top_p\": 0.95, \"seed\": 1234, \"max_tokens\": 8192} | repetition tokenizer: server /tokenize | 0 tool calls in this run count toward the config's 100-call minimum (\u00a75) | FAIL: 68 tool-call parse failure(s); decode window min 13.255 < floor 15.0"
}
