{
  "bench_run_identity": {
    "api_mode": "chat",
    "base_url": "http://127.0.0.1:18263",
    "created_at_utc": "2026-06-23T07:50:05.611088+00:00",
    "max_tokens": 512,
    "model": "gemma4-26b-a4b-q8",
    "prompt_chars": 306,
    "prompt_mode": "long",
    "prompt_preview": "Write a long deterministic decode benchmark response. Continue until you reach the token limit. Use numbered lines from 001 onward. Each line must contain the words benchmark, latency, memory, throughput, validation, and repeatability. Do n",
    "prompt_sha256": "16eb10ea9cc18dbf4826b5b701799018f8e4b0f37a198b86495179b19566b677",
    "prompt_tokens_requested": 512,
    "repeats": 8,
    "seed": 1,
    "usage_required": true
  },
  "bench_summary": {
    "completion_tokens": {
      "cv": 0.0,
      "max": 512,
      "mean": 512.0,
      "median": 512.0,
      "min": 512,
      "stdev": 0.0
    },
    "completion_tokens_total": 4096,
    "elapsed_s": {
      "cv": 0.0409631532402811,
      "max": 11.853832474007504,
      "mean": 11.102074027123308,
      "median": 10.889915428502718,
      "min": 10.697531885001808,
      "stdev": 0.4547759596579968
    },
    "prompt_tokens": {
      "cv": 0.0,
      "max": 75,
      "mean": 75.0,
      "median": 75.0,
      "min": 75,
      "stdev": 0.0
    },
    "requests": 8,
    "tok_s_after_ttft": {
      "cv": 0.04236141100197593,
      "max": 49.88620536626224,
      "mean": 47.92204795174685,
      "median": 48.79797988405781,
      "min": 44.65918713395411,
      "stdev": 2.0300455693403467
    },
    "tok_s_wall": {
      "cv": 0.03996658795005443,
      "max": 47.86150726204762,
      "mean": 46.18365106442844,
      "median": 47.01603872417265,
      "min": 43.19278183850567,
      "stdev": 1.8458029521211041
    },
    "ttft_s": {
      "cv": 0.05009522913241545,
      "max": 0.4491573779960163,
      "mean": 0.40083310099726077,
      "median": 0.3950690854835557,
      "min": 0.3888419840077404,
      "stdev": 0.020079826038314402
    }
  },
  "canary_pass_all": true,
  "canary_rows_completed": 384,
  "label": "gemma4-q8-gpu3-mtp-n3-aot-bmg-long-deep-20260623T0345",
  "launcher_identity": {
    "batch_size": "512",
    "cache_type_k": "f16",
    "cache_type_v": "f16",
    "ctx_size": "8192",
    "extra_llama_args": "--parallel 1 --cache-ram 0 --spec-type draft-mtp --spec-draft-model /mnt/fast-ai/llm-models/gemma4-26b-a4b-it-q8-gguf/mtp-gemma-4-26B-A4B-it.gguf --spec-draft-n-max 3 --spec-draft-device SYCL0 --spec-draft-ngl all --spec-draft-type-k f16 --spec-draft-type-v f16",
    "flash_attn": "off",
    "ggml_sycl_disable_dnn": null,
    "ggml_sycl_disable_graph": null,
    "ggml_sycl_disable_opt": "0",
    "gpu_index": "3",
    "oneapi_device_selector": null,
    "poll": "50",
    "port": "18263",
    "reasoning": "off",
    "threads": "16",
    "ubatch_size": "64"
  },
  "model_file_bytes": 27636230944,
  "model_path": "/mnt/fast-ai/llm-models/gemma4-26b-a4b-it-q8-gguf/gemma-4-26B-A4B-it-UD-Q8_K_XL.gguf",
  "run_dir": "/home/steve/qwen36-results-main/data/gemma4-q8-gpu3-mtp-n3-aot-bmg-long-deep-20260623T0345",
  "server_log": "/mnt/fast-ai/bench-results/gemma4-26b-a4b-q8/servers/gemma4-q8-gpu3-mtp-n3-aot-bmg-long-deep-20260623T0345.server.log"
}
