{
 "schema": "local-ai-bench/result/v1",
 "run_id": "20261004T105602Z-sglang-qwen3.8-flash-next-exl3-4.05bpw-offload-200k-rtx-3090-24gb-gpu2-load",
 "ticket": "T15",
 "kind": "load",
 "model": "qwen3.8-flash-next",
 "config_id": "sglang-qwen3.8-flash-next-exl3-4.05bpw-offload-200k-rtx-3090-24gb-gpu2",
 "outcome": "crash",
 "failed_stage": "load",
 "exit_status": 0,
 "started_at": "2026-10-04T10:56:02.903526+00:00",
 "finished_at": "2026-10-04T10:56:58.930837+00:00",
 "launch": {
  "file": "launches/sglang-qwen3.8-flash-next-exl3-4.05bpw-offload-200k-rtx-3090-24gb.json",
  "sha256": "ce41d62d49f95e441684dbb7afcbfd2cd54c0cdf49476dda70cda1eca5bc7928",
  "image": "ghcr.io/0xsero/sglang-exl3-flashnext@sha256:ce47df25cee7ff235d8bfdf3e79e3828aa11bee3624db3ce121ebc79a8375dd3",
  "argv": [
   "/opt/entrypoint.sh",
   "python3",
   "-m",
   "sglang.launch_server",
   "--model-path",
   "/models/turboderp-Qwen3.8-Flash-Next-exl3-4.05bpw_h6_ng6",
   "--quantization",
   "exl3",
   "--trust-remote-code",
   "--host",
   "0.0.0.0",
   "--port",
   "30100",
   "--served-model-name",
   "flashnext",
   "--disable-shared-experts-fusion",
   "--kv-cache-dtype",
   "fp8_e4m3",
   "--context-length",
   "204800",
   "--mem-fraction-static",
   "0.88",
   "--chunked-prefill-size",
   "8192",
   "--max-running-requests",
   "4",
   "--cuda-graph-max-bs-decode",
   "8",
   "--cuda-graph-backend-prefill",
   "disabled",
   "--max-mamba-cache-size",
   "16",
   "--max-total-tokens",
   "210000",
   "--reasoning-parser",
   "qwen3",
   "--tool-call-parser",
   "qwen3_coder"
  ],
  "env": {
   "HF_HUB_OFFLINE": "1",
   "CUDA_DEVICE_ORDER": "PCI_BUS_ID",
   "SGLANG_EXL3_MODEL_PATH": "/models/turboderp-Qwen3.8-Flash-Next-exl3-4.05bpw_h6_ng6",
   "SGLANG_EXL3_MOE_OFFLOAD": "gpu_cache",
   "EXL3_MOE_CPU_THREADS": "24",
   "SGLANG_EXL3_EXPERT_CACHE_GB": "auto",
   "SGLANG_EXL3_EMBED_HOST": "1",
   "SGLANG_EXL3_OFFLOAD_STAGING_PARTS": "4",
   "SGLANG_EXL3_OFFLOAD_FUSED": "1",
   "SGLANG_EXL3_OFFLOAD_COMPACT_PARTS": "1",
   "SGLANG_EXL3_MOE_PREFILL_FP16_ACC": "1",
   "SGLANG_EXL3_NGRAM_TIER": "nvme",
   "SGLANG_EXL3_VISION": "1",
   "SGLANG_EXL3_MM_FAST_CPU": "1",
   "SGLANG_EXL3_VIT_SDPA": "1",
   "SGLANG_EXL3_VIT_MLP_CHUNK": "8192"
  },
  "engine": "sglang",
  "engine_source_commit": "https://github.com/0xSero/trellis-serve/tree/1e72b250cb529c6d81bf11b88740d9a0b22f62de",
  "mtp": null,
  "entrypoint": "/opt/entrypoint.sh",
  "port": 30100,
  "shm": "64g",
  "flags": [],
  "weights_at": [
   "/models/turboderp-Qwen3.8-Flash-Next-exl3-4.05bpw_h6_ng6"
  ]
 },
 "weights": [
  {
   "repo": "turboderp/Qwen3.8-Flash-Next-exl3",
   "revision": "55a732e0c4c3d4614bc42b68493bb930d9b02c0a",
   "manifest": "models/turboderp-Qwen3.8-Flash-Next-exl3-4.05bpw_h6_ng6.json"
  }
 ],
 "host": {
  "hostname": "omarchy-gpu",
  "cpu": "AMD Ryzen 9 5950X 16-Core Processor",
  "ram_total_gib": 125.7,
  "ram_layout": "DIMM_A1 32 GiB DDR4 3200 MT/s, DIMM_A2 32 GiB DDR4 3200 MT/s, DIMM_B1 32 GiB DDR4 3200 MT/s, DIMM_B2 32 GiB DDR4 3200 MT/s (AM4: 2 channels)",
  "kernel": "7.2.5-3-omarchy",
  "driver": "610.57.04",
  "cuda": "13.3",
  "cpus_online": "0-1,3-17,19-31",
  "cpus_offline": "2,18",
  "cpu_boost": false,
  "cpu_max_mhz": 3401,
  "machine_checks_this_boot": 6
 },
 "gpus": [
  {
   "index": 0,
   "uuid": "GPU-816b43e4-8d65-dfd7-2234-8517b3dfbf2d",
   "name": "NVIDIA GeForce RTX 3090",
   "bus_id": "00000000:04:00.0",
   "pcie_gen_max": 4,
   "pcie_width": 8,
   "display_active": false,
   "used_by_run": false
  },
  {
   "index": 1,
   "uuid": "GPU-c67ac872-3371-a88c-aa92-d969daf6405e",
   "name": "NVIDIA GeForce RTX 3090",
   "bus_id": "00000000:0B:00.0",
   "pcie_gen_max": 4,
   "pcie_width": 8,
   "display_active": false,
   "used_by_run": false
  },
  {
   "index": 2,
   "uuid": "GPU-76d3c6af-7f18-69d2-3145-899103de1722",
   "name": "NVIDIA GeForce RTX 3090",
   "bus_id": "00000000:0C:00.0",
   "pcie_gen_max": 4,
   "pcie_width": 8,
   "display_active": false,
   "used_by_run": true
  }
 ],
 "layout": {
  "name": "current",
  "cards": 1,
  "display": "off"
 },
 "context": {
  "configured": 204800,
  "occupied_max": {
   "state": "unavailable",
   "reason": "no occupancy measured in this run"
  },
  "headroom_min": {
   "state": "unavailable",
   "reason": "no occupancy measured in this run"
  }
 },
 "cache_state": {
  "cold": true,
  "reset": [
   "container",
   "prefix_kv",
   "expert_cache",
   "ngram_row_cache",
   "page_cache"
  ]
 },
 "provenance": {
  "bench_commit": "bd62e89c96f97db97945fd89be01017a19fea7c3",
  "bench_dirty": true,
  "pins_sha256": "858564b6e07125ac48c2ef6c5178d0a53596f21eb5f0ebda43906472b01b4d78",
  "registry_commit": "d21258dd744e7c78be28177c90060c6af8e10b7e",
  "kit_commit": "ef883d269e50ecbea290f919f095e2f3ca633b42",
  "harness_commit": "04d809ceab9df28f9adaed044884180159172930"
 },
 "metrics": {
  "load_seconds": {
   "state": "unavailable",
   "reason": "not produced (crash/load)"
  },
  "host_ram_drop_gb": {
   "state": "unavailable",
   "reason": "not produced (crash/load)"
  },
  "vram_ready_mib": {
   "state": "unavailable",
   "reason": "not produced (crash/load)"
  }
 },
 "eligibility": "none",
 "artifacts": [
  "artifacts/runs/20261004T105602Z-lab-sglang-qwen3.8-flash-next-exl3-4.05bpw-offload-200k-rtx-3090"
 ],
 "notes": "        ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^\n  File \"/opt/trellis-serve/cuda/src/sglang_exl3/offload/ngram_nvme.py\", line 116, in __init__\n    t = parse_table(model_path)\n        ^^^^^^^^^^^^^^^^^^^^^^^\n  File \"/opt/trellis-serve/cuda/src/sglang_exl3/offload/ngram_nvme.py\", line 94, in parse_table\n    raise ValueError(f\"{path}: shard_N.trellis tensors missing or not contiguous\")\nValueError: /models/turboderp-Qwen3.8-Flash-Next-exl3-4.05bpw_h6_ng6/ngram_embedding.safetensors: shard_N.trellis tensors missing or not contiguous\n\n[2026-10-04 10:56:54] Received sigquit from a child process. It usually means the child failed.\n[2026-10-04 10:56:54] kill_process_tree called: parent_pid=1, include_parent=True, pid=1\n[2026-10-04 10:56:54] kill_process_tree called: parent_pid=1, include_parent=False, pid=1\n"
}
