{
  "bench_run_identity": {
    "api_mode": "chat",
    "base_url": "http://127.0.0.1:18272",
    "created_at_utc": "2026-06-23T21:00:45.763705+00:00",
    "max_tokens": 512,
    "model": "gemma4-26b-a4b-fp8tensor",
    "prompt_chars": 3460,
    "prompt_mode": "filled-long",
    "prompt_preview": "You are running a deterministic Gemma B70 decode benchmark. Read the reference context, then produce a long numbered response until the token limit is reached. Do not summarize early.\n\nReference context:\nbenchmark latency memory throughput ",
    "prompt_sha256": "04f687a68d60254567d3252624e01a696b0c9843dbb02f5a789d75aada5e4da5",
    "prompt_tokens_requested": 512,
    "repeats": 2,
    "seed": 1,
    "usage_required": true
  },
  "bench_summary": {
    "completion_tokens": {
      "cv": 0.0,
      "max": 512,
      "mean": 512.0,
      "median": 512.0,
      "min": 512,
      "stdev": 0.0
    },
    "completion_tokens_total": 1024,
    "elapsed_s": {
      "cv": 0.0031166791505030933,
      "max": 13.145889588980936,
      "mean": 13.116982056497363,
      "median": 13.116982056497363,
      "min": 13.088074524013791,
      "stdev": 0.04088142449300852
    },
    "prompt_tokens": {
      "cv": 0.0,
      "max": 588,
      "mean": 588.0,
      "median": 588.0,
      "min": 588,
      "stdev": 0.0
    },
    "requests": 2,
    "tok_s_after_ttft": {
      "cv": 0.00012994750546320313,
      "max": 40.31482652126549,
      "mean": 40.31112246273725,
      "median": 40.31112246273725,
      "min": 40.30741840420902,
      "stdev": 0.0052383298064544
    },
    "tok_s_wall": {
      "cv": 0.00311667915050317,
      "max": 39.119581651265094,
      "mean": 39.033558520637584,
      "median": 39.033558520637584,
      "min": 38.947535390010074,
      "stdev": 0.12165507801121653
    },
    "ttft_s": {
      "cv": 0.10229606519564757,
      "max": 0.44584734199452214,
      "mean": 0.41577273650909774,
      "median": 0.41577273650909774,
      "min": 0.38569813102367334,
      "stdev": 0.04253191496050746
    }
  },
  "canary_pass_all": true,
  "canary_rows_completed": 64,
  "fresh_response_validity": {
    "benchmark_repeats_mixed": true,
    "cached_tokens_all_zero": null,
    "cached_tokens_reported": false,
    "draft_history": "none",
    "first_request_cached_tokens": null,
    "first_request_completion_tokens": 512,
    "first_request_prompt_tokens": 588,
    "first_request_tok_s_after_ttft": 40.31482652126549,
    "first_request_tok_s_wall": 38.947535390010074,
    "first_request_ttft_s": 0.44584734199452214,
    "first_request_usage": {
      "completion_tokens": 512,
      "prompt_tokens": 588,
      "total_tokens": 1100
    },
    "prefix_caching": "disabled"
  },
  "label": "gemma4-vllm-fp8tensor-gpu2-compile12-piecewise-smoke-20260623T205416Z",
  "launcher_identity": {
    "compilation_config": "{\"use_inductor_graph_partition\":true,\"compile_sizes\":[1,2],\"cudagraph_mode\":\"PIECEWISE\"}",
    "dtype": "bfloat16",
    "generation_config": "vllm",
    "gpu_index": "2",
    "gpu_memory_utilization": "0.90",
    "kv_cache_dtype": "auto",
    "language_model_only": "true",
    "max_model_len": "8192",
    "max_num_batched_tokens": "1024",
    "max_num_seqs": "1",
    "model_alias": "gemma4-26b-a4b-fp8tensor",
    "oneapi_device_selector": "level_zero:*",
    "port": "18272",
    "prefix_caching": "disabled",
    "quantization": "fp8_per_tensor",
    "runtime": "vllm",
    "torchinductor_cache_dir": "/mnt/fast-ai/vllm-cache-exp/gemma4-26b-a4b-it-fp8tensor-gpu2/torchinductor",
    "vllm_bin": "/home/steve/.venvs/vllm-xpu/bin/vllm",
    "vllm_cache_root": "/mnt/fast-ai/vllm-cache-exp/gemma4-26b-a4b-it-fp8tensor-gpu2",
    "vllm_extra_args": "<unset>",
    "vllm_source_path": "/home/steve/src/vllm/vllm",
    "vllm_target_device": "xpu",
    "vllm_use_v1": "1",
    "vllm_version": "0.20.2rc1.dev13+g9557d9108.d20260620",
    "vllm_xpu_enable_xpu_graph": "1",
    "vllm_xpu_force_graph_with_comm": "1",
    "vllm_xpu_graph_noop_comm_capture": "1",
    "xpu_graph": "1",
    "ze_affinity_mask": "2"
  },
  "model_file_bytes": 51612009916,
  "model_path": "/mnt/fast-ai/llm-cache/hf/models--google--gemma-4-26B-A4B-it/snapshots/20da991ab4afab98e8f910c4a2e8f4fbefc404ad",
  "model_shard_count": 2,
  "run_dir": "/home/steve/qwen36-results-main/data/gemma4-vllm-fp8tensor-gpu2-compile12-piecewise-smoke-20260623T205416Z",
  "server_log": "/mnt/fast-ai/bench-results/gemma4-26b-a4b-vllm-int8pc/servers/gemma4-vllm-fp8tensor-gpu2-compile12-piecewise-smoke-20260623T205416Z.server.log"
}
