{
  "bench_run_identity": {
    "api_mode": "chat",
    "base_url": "http://127.0.0.1:18271",
    "created_at_utc": "2026-06-23T18:00:08.223785+00:00",
    "max_tokens": 512,
    "model": "gemma4-26b-a4b-q8",
    "prompt_chars": 3460,
    "prompt_mode": "filled-long",
    "prompt_preview": "You are running a deterministic Gemma B70 decode benchmark. Read the reference context, then produce a long numbered response until the token limit is reached. Do not summarize early.\n\nReference context:\nbenchmark latency memory throughput ",
    "prompt_sha256": "04f687a68d60254567d3252624e01a696b0c9843dbb02f5a789d75aada5e4da5",
    "prompt_tokens_requested": 512,
    "repeats": 8,
    "seed": 1,
    "usage_required": true
  },
  "bench_summary": {
    "completion_tokens": {
      "cv": 0.0,
      "max": 512,
      "mean": 512.0,
      "median": 512.0,
      "min": 512,
      "stdev": 0.0
    },
    "completion_tokens_total": 4096,
    "elapsed_s": {
      "cv": 1.0869368622572018,
      "max": 13.32976033401792,
      "mean": 3.6124016328758444,
      "median": 2.2206901739991736,
      "min": 2.2086933520040475,
      "stdev": 3.926452496050862
    },
    "prompt_tokens": {
      "cv": 0.0,
      "max": 588,
      "mean": 588.0,
      "median": 588.0,
      "min": 588,
      "stdev": 0.0
    },
    "requests": 8,
    "tok_s_after_ttft": {
      "cv": 0.34485852371195475,
      "max": 317.6715365353922,
      "mean": 280.6417012226978,
      "median": 315.7460091407923,
      "min": 41.30790808993625,
      "stdev": 96.78168277567104
    },
    "tok_s_wall": {
      "cv": 0.3289434874072691,
      "max": 231.81126503388947,
      "mean": 206.23605558808984,
      "median": 230.55990105860593,
      "min": 38.41029299629354,
      "stdev": 67.84000735426568
    },
    "ttft_s": {
      "cv": 0.18646403260250477,
      "max": 0.9350392339983955,
      "mean": 0.6398649531256524,
      "median": 0.5995441715058405,
      "min": 0.5935310759814456,
      "stdev": 0.11931179948082184
    }
  },
  "canary_pass_all": true,
  "canary_rows_completed": 384,
  "label": "gemma4-q8-gpu1-ngram-mod-20-32-64-ctx4096ub512-poll100-ctxcp0-filled-long-deep-20260623T1855",
  "launcher_identity": {
    "batch_size": "512",
    "cache_type_k": "f16",
    "cache_type_v": "f16",
    "ctx_size": "4096",
    "extra_llama_args": "--parallel 1 --cache-ram 0 --spec-type ngram-mod --ctx-checkpoints 0 --spec-ngram-mod-n-match 20 --spec-ngram-mod-n-min 32 --spec-ngram-mod-n-max 64",
    "flash_attn": "off",
    "ggml_sycl_disable_dnn": "<unset>",
    "ggml_sycl_disable_graph": "<unset>",
    "ggml_sycl_disable_opt": "0",
    "ggml_sycl_enable_vmm": "<unset>",
    "gpu_index": "1",
    "llama_cpp_commit": "c926ad098",
    "llama_mtp_draft_backend_topk": "<unset>",
    "llama_mtp_draft_fast_topk": "<unset>",
    "llama_mtp_draft_logit_gap_min": "<unset>",
    "llama_mtp_draft_profile": "<unset>",
    "llama_mtp_draft_top_k": "<unset>",
    "llama_server": "/home/steve/src/llama.cpp-latest-gemma/build-sycl-b70-aot-bmg-g31/bin/llama-server",
    "oneapi_device_selector": "level_zero:1",
    "poll": "100",
    "port": "18271",
    "reasoning": "off",
    "threads": "16",
    "ubatch_size": "512"
  },
  "model_file_bytes": 27636230944,
  "model_path": "/mnt/fast-ai/llm-models/gemma4-26b-a4b-it-q8-gguf/gemma-4-26B-A4B-it-UD-Q8_K_XL.gguf",
  "run_dir": "/home/steve/qwen36-results-main/data/gemma4-q8-gpu1-ngram-mod-20-32-64-ctx4096ub512-poll100-ctxcp0-filled-long-deep-20260623T1855",
  "server_log": "/mnt/fast-ai/bench-results/gemma4-26b-a4b-q8/servers/gemma4-q8-gpu1-ngram-mod-20-32-64-ctx4096ub512-poll100-ctxcp0-filled-long-deep-20260623T1855.server.log"
}
