{
 "schema": "local-ai-bench/result/v1",
 "run_id": "20261004T110201Z-exllamav3-glm-5.3-flash-exl3-3.05bpw-exact-128k-1x-gpu2-lab",
 "ticket": "T23",
 "kind": "lab",
 "model": "glm-5.3-flash",
 "config_id": "exllamav3-glm-5.3-flash-exl3-3.05bpw-exact-128k-1x-gpu2",
 "outcome": "fail",
 "failed_stage": "gates",
 "exit_status": 0,
 "started_at": "2026-10-04T11:02:01.535096+00:00",
 "finished_at": "2026-10-04T11:12:59.784824+00:00",
 "launch": {
  "file": "launches/exllamav3-glm-5.3-flash-exl3-3.05bpw-exact-128k-1x.json",
  "sha256": "b679beb9863b299d0b6e44b1269b954e5292fb8039caa75a0e4c0d34e6cb17fa",
  "image": "ghcr.io/0xsero/glm53-flash-offload@sha256:bb633b0bcb85573ad40b6c408af5e1036ae062c593e42479f4e4d2ab521e551d",
  "argv": [
   "/opt/glm53/docker/entrypoint.sh",
   "-m",
   "/models/turboderp-GLM-5.3-Flash-exl3-3.05bpw",
   "-cs",
   "131072",
   "--max-batch-size",
   "8",
   "-chunk_size",
   "8192",
   "-ambs",
   "4",
   "--host",
   "0.0.0.0",
   "--port",
   "30000",
   "--served-name",
   "glm-5.3-flash"
  ],
  "env": {
   "GLM53_MODE": "exact",
   "GLM53_MODEL_DIR": "/models/turboderp-GLM-5.3-Flash-exl3-3.05bpw",
   "GLM53_MODEL_DOWNLOAD": "0",
   "HF_HUB_OFFLINE": "1"
  },
  "engine": "exllamav3",
  "engine_source_commit": "https://github.com/0xSero/glm53-flash-offload/tree/fe97bcf846347d035859a0610cebbb707fa36caa",
  "mtp": null,
  "entrypoint": "/opt/glm53/docker/entrypoint.sh",
  "port": 30000,
  "shm": "16g",
  "flags": [],
  "weights_at": [
   "/models/turboderp-GLM-5.3-Flash-exl3-3.05bpw"
  ]
 },
 "weights": [
  {
   "repo": "turboderp/GLM-5.3-Flash-exl3",
   "revision": "332ab457b709b7ba30dd9a448be5de03b80a7ac9",
   "manifest": "models/turboderp-GLM-5.3-Flash-exl3-3.05bpw.json"
  }
 ],
 "host": {
  "hostname": "omarchy-gpu",
  "cpu": "AMD Ryzen 9 5950X 16-Core Processor",
  "ram_total_gib": 125.7,
  "ram_layout": "DIMM_A1 32 GiB DDR4 3200 MT/s, DIMM_A2 32 GiB DDR4 3200 MT/s, DIMM_B1 32 GiB DDR4 3200 MT/s, DIMM_B2 32 GiB DDR4 3200 MT/s (AM4: 2 channels)",
  "kernel": "7.2.5-3-omarchy",
  "driver": "610.57.04",
  "cuda": "13.3",
  "cpus_online": "0-1,3-17,19-31",
  "cpus_offline": "2,18",
  "cpu_boost": false,
  "cpu_max_mhz": 3401,
  "machine_checks_this_boot": 6
 },
 "gpus": [
  {
   "index": 0,
   "uuid": "GPU-816b43e4-8d65-dfd7-2234-8517b3dfbf2d",
   "name": "NVIDIA GeForce RTX 3090",
   "bus_id": "00000000:04:00.0",
   "pcie_gen_max": 4,
   "pcie_width": 8,
   "display_active": false,
   "used_by_run": false
  },
  {
   "index": 1,
   "uuid": "GPU-c67ac872-3371-a88c-aa92-d969daf6405e",
   "name": "NVIDIA GeForce RTX 3090",
   "bus_id": "00000000:0B:00.0",
   "pcie_gen_max": 4,
   "pcie_width": 8,
   "display_active": false,
   "used_by_run": false
  },
  {
   "index": 2,
   "uuid": "GPU-76d3c6af-7f18-69d2-3145-899103de1722",
   "name": "NVIDIA GeForce RTX 3090",
   "bus_id": "00000000:0C:00.0",
   "pcie_gen_max": 4,
   "pcie_width": 8,
   "display_active": false,
   "used_by_run": true
  }
 ],
 "layout": {
  "name": "current",
  "cards": 1,
  "display": "off"
 },
 "context": {
  "configured": 131072,
  "occupied_max": {
   "value": 93437,
   "unit": "tokens"
  },
  "headroom_min": {
   "state": "unavailable",
   "reason": "no occupancy measured in this run"
  }
 },
 "cache_state": {
  "cold": true,
  "reset": [
   "container",
   "prefix_kv",
   "expert_cache",
   "ngram_row_cache",
   "page_cache"
  ]
 },
 "provenance": {
  "bench_commit": "d36a5ce7e63d3abd393f2511b2a55930e67bf549",
  "bench_dirty": true,
  "pins_sha256": "858564b6e07125ac48c2ef6c5178d0a53596f21eb5f0ebda43906472b01b4d78",
  "registry_commit": "d21258dd744e7c78be28177c90060c6af8e10b7e",
  "kit_commit": "ef883d269e50ecbea290f919f095e2f3ca633b42",
  "harness_commit": "04d809ceab9df28f9adaed044884180159172930"
 },
 "metrics": {
  "gates_passed": {
   "value": [
    "load",
    "chat",
    "reasoning",
    "tools",
    "context"
   ],
   "unit": "gates",
   "detail": {
    "load": true,
    "chat": true,
    "reasoning": true,
    "tools": true,
    "context": true,
    "speed": false
   }
  },
  "lab_decode_c1_tps": {
   "value": 7.4,
   "unit": "tok/s"
  },
  "lab_prefill_tps": {
   "value": 739,
   "unit": "tok/s",
   "detail": "context-gate prompt tokens / total request seconds"
  },
  "mtp_accepted_tps": {
   "state": "skipped",
   "reason": "speculative decoding off in this launch"
  },
  "mtp_acceptance_rate": {
   "state": "skipped",
   "reason": "speculative decoding off in this launch"
  }
 },
 "eligibility": "none",
 "artifacts": [
  "artifacts/runs/20261004T110201Z-lab-exllamav3-glm-5.3-flash-exl3-3.05bpw-exact-128k-1x-gpu2-lab"
 ],
 "notes": "served=glm-5.3-flash"
}
