{
 "schema": "local-ai-bench/result/v1",
 "run_id": "20261003T002815Z-fn-llamacpp-iq4xs-128k-1x-gpu2-load",
 "ticket": "T11",
 "kind": "load",
 "model": "qwen3.8-flash-next",
 "config_id": "fn-llamacpp-iq4xs-128k-1x-gpu2",
 "outcome": "crash",
 "failed_stage": "load",
 "exit_status": 1,
 "started_at": "2026-10-03T00:28:15.465370+00:00",
 "finished_at": "2026-10-03T00:28:37.560982+00:00",
 "launch": {
  "file": "launches/llama.cpp-qwen3.8-flash-next-iq4xs-128k-1x.json",
  "sha256": "697275c75292e4ef45a02a99b2a98c23b7a55630ed0430af0972a1dcf84b47d8",
  "image": "ghcr.io/ggml-org/llama.cpp:server-cuda12-b11312@sha256:6cdf9529493b9581c421fec5628dcd4b8dd7c09507fc7d27ad739cc9205d66f5",
  "argv": [
   "/app/llama-server",
   "--model",
   "/models/Qwen3.8-Flash-Next-IQ4_XS/Qwen3.8-Flash-Next-IQ4_XS-00001-of-00003.gguf",
   "--ctx-size",
   "131072",
   "--flash-attn",
   "on",
   "--cache-type-k",
   "q8_0",
   "--cache-type-v",
   "q8_0",
   "--fit",
   "on",
   "--parallel",
   "1",
   "--threads",
   "16",
   "--jinja",
   "--no-mmproj",
   "--metrics",
   "--host",
   "0.0.0.0",
   "--port",
   "8080",
   "-ot",
   "per_layer_token_embd.weight=CPU"
  ],
  "env": {
   "CUDA_DEVICE_ORDER": "PCI_BUS_ID"
  },
  "engine": "llama.cpp",
  "engine_source_commit": null,
  "mtp": null,
  "entrypoint": "/app/llama-server",
  "port": 8080,
  "shm": "16g",
  "flags": [],
  "weights_at": [
   "/models"
  ]
 },
 "weights": [
  {
   "repo": "bartowski/Qwen3.8-Flash-Next-GGUF",
   "revision": "928589fdb66c6ff07f22ac561e3fbce76553548f",
   "manifest": "models/bartowski-Qwen3.8-Flash-Next-IQ4_XS.json"
  }
 ],
 "host": {
  "hostname": "omarchy-gpu",
  "cpu": "AMD Ryzen 9 5950X 16-Core Processor",
  "ram_total_gib": 125.7,
  "ram_layout": "DIMM_A1 32 GiB DDR4 3200 MT/s, DIMM_A2 32 GiB DDR4 3200 MT/s, DIMM_B1 32 GiB DDR4 3200 MT/s, DIMM_B2 32 GiB DDR4 3200 MT/s (AM4: 2 channels)",
  "kernel": "7.2.5-3-omarchy",
  "driver": "610.57.04",
  "cuda": "13.3",
  "cpus_online": "0-31",
  "cpus_offline": "none",
  "machine_checks_this_boot": 0
 },
 "gpus": [
  {
   "index": 0,
   "uuid": "GPU-816b43e4-8d65-dfd7-2234-8517b3dfbf2d",
   "name": "NVIDIA GeForce RTX 3090",
   "bus_id": "00000000:04:00.0",
   "pcie_gen_max": 4,
   "pcie_width": 8,
   "display_active": false,
   "used_by_run": false
  },
  {
   "index": 1,
   "uuid": "GPU-c67ac872-3371-a88c-aa92-d969daf6405e",
   "name": "NVIDIA GeForce RTX 3090",
   "bus_id": "00000000:0B:00.0",
   "pcie_gen_max": 4,
   "pcie_width": 8,
   "display_active": true,
   "used_by_run": false
  },
  {
   "index": 2,
   "uuid": "GPU-76d3c6af-7f18-69d2-3145-899103de1722",
   "name": "NVIDIA GeForce RTX 3090",
   "bus_id": "00000000:0C:00.0",
   "pcie_gen_max": 4,
   "pcie_width": 8,
   "display_active": false,
   "used_by_run": true
  }
 ],
 "layout": {
  "name": "current",
  "cards": 1,
  "display": "on"
 },
 "context": {
  "configured": 131072,
  "occupied_max": {
   "state": "unavailable",
   "reason": "no occupancy measured in this run"
  },
  "headroom_min": {
   "state": "unavailable",
   "reason": "no occupancy measured in this run"
  }
 },
 "cache_state": {
  "cold": true,
  "reset": [
   "container",
   "prefix_kv",
   "expert_cache",
   "ngram_row_cache",
   "page_cache"
  ]
 },
 "provenance": {
  "bench_commit": "1fa796f9d509705f24dc1870357869cf7c293bc2",
  "bench_dirty": false,
  "pins_sha256": "858564b6e07125ac48c2ef6c5178d0a53596f21eb5f0ebda43906472b01b4d78",
  "registry_commit": "d21258dd744e7c78be28177c90060c6af8e10b7e",
  "kit_commit": "ef883d269e50ecbea290f919f095e2f3ca633b42",
  "harness_commit": "04d809ceab9df28f9adaed044884180159172930"
 },
 "metrics": {
  "load_seconds": {
   "state": "unavailable",
   "reason": "not produced (crash/load)"
  },
  "host_ram_drop_gb": {
   "state": "unavailable",
   "reason": "not produced (crash/load)"
  },
  "vram_ready_mib": {
   "state": "unavailable",
   "reason": "not produced (crash/load)"
  }
 },
 "eligibility": "none",
 "artifacts": [
  "artifacts/runs/20261003T002815Z-lab-fn-llamacpp-iq4xs-128k-1x-gpu2"
 ],
 "notes": "user, abort\n0.18.994.031 E ggml_backend_cuda_buffer_type_alloc_buffer: allocating 65358.17 MiB on device 0: cudaMalloc failed: out of memory\n0.18.994.037 E alloc_tensor_range: failed to allocate CUDA0 buffer of size 68533006336\n0.19.139.695 E llama_model_load: error loading model: unable to allocate CUDA0 buffer\n0.19.139.702 E llama_model_load_from_file_impl: failed to load model\n0.19.139.708 E cmn  common_init_: failed to load model '/models/Qwen3.8-Flash-Next-IQ4_XS/Qwen3.8-Flash-Next-IQ4_XS-00001-of-00003.gguf'\n0.19.139.713 E srv    load_model: failed to load model, '/models/Qwen3.8-Flash-Next-IQ4_XS/Qwen3.8-Flash-Next-IQ4_XS-00001-of-00003.gguf'\n0.19.139.715 I srv    operator(): operator(): cleaning up before exit...\n0.19.140.570 E srv  llama_server: exiting due to model loading error\n"
}
