{
  "bench_run_identity": {
    "api_mode": "chat",
    "base_url": "http://127.0.0.1:18271",
    "created_at_utc": "2026-06-23T20:47:44.009700+00:00",
    "max_tokens": 512,
    "model": "gemma4-26b-a4b-int8pc",
    "prompt_chars": 3460,
    "prompt_mode": "filled-long",
    "prompt_preview": "You are running a deterministic Gemma B70 decode benchmark. Read the reference context, then produce a long numbered response until the token limit is reached. Do not summarize early.\n\nReference context:\nbenchmark latency memory throughput ",
    "prompt_sha256": "04f687a68d60254567d3252624e01a696b0c9843dbb02f5a789d75aada5e4da5",
    "prompt_tokens_requested": 512,
    "repeats": 4,
    "seed": 1,
    "usage_required": true
  },
  "bench_summary": {
    "completion_tokens": {
      "cv": 0.0,
      "max": 512,
      "mean": 512.0,
      "median": 512.0,
      "min": 512,
      "stdev": 0.0
    },
    "completion_tokens_total": 2048,
    "elapsed_s": {
      "cv": 0.00546767375309072,
      "max": 15.191912604990648,
      "mean": 15.078176575254474,
      "median": 15.052271265507443,
      "min": 15.016251165012363,
      "stdev": 0.0824425503049862
    },
    "prompt_tokens": {
      "cv": 0.0,
      "max": 588,
      "mean": 588.0,
      "median": 588.0,
      "min": 588,
      "stdev": 0.0
    },
    "requests": 4,
    "tok_s_after_ttft": {
      "cv": 0.006032851469171167,
      "max": 34.997263596401574,
      "mean": 34.8886759002239,
      "median": 34.99221769697348,
      "min": 34.57300461054707,
      "stdev": 0.21047819966210246
    },
    "tok_s_wall": {
      "cv": 0.005449627521667192,
      "max": 34.09639292614872,
      "mean": 33.957119391534775,
      "median": 34.01497118390253,
      "min": 33.7021422721853,
      "stdev": 0.18505365239264662
    },
    "ttft_s": {
      "cv": 0.08913155436732953,
      "max": 0.4563014959858265,
      "mean": 0.4025247277531889,
      "median": 0.38556352151499595,
      "min": 0.3826703719969373,
      "stdev": 0.035877654655927876
    }
  },
  "canary_pass_all": true,
  "canary_rows_completed": 128,
  "fresh_response_validity": {
    "benchmark_repeats_mixed": true,
    "cached_tokens_all_zero": null,
    "cached_tokens_reported": false,
    "draft_history": "none",
    "first_request_cached_tokens": null,
    "first_request_completion_tokens": 512,
    "first_request_prompt_tokens": 588,
    "first_request_tok_s_after_ttft": 34.997263596401574,
    "first_request_tok_s_wall": 33.938713547577926,
    "first_request_ttft_s": 0.4563014959858265,
    "first_request_usage": {
      "completion_tokens": 512,
      "prompt_tokens": 588,
      "total_tokens": 1100
    },
    "prefix_caching": "disabled"
  },
  "label": "gemma4-vllm-int8pc-gpu1-piecewise-selectorfix-smoke-20260623T204041Z",
  "launcher_identity": {
    "compilation_config": "{\"use_inductor_graph_partition\":true,\"compile_sizes\":[1],\"cudagraph_mode\":\"PIECEWISE\"}",
    "dtype": "bfloat16",
    "generation_config": "vllm",
    "gpu_index": "1",
    "gpu_memory_utilization": "0.90",
    "kv_cache_dtype": "auto",
    "language_model_only": "true",
    "max_model_len": "8192",
    "max_num_batched_tokens": "1024",
    "max_num_seqs": "1",
    "model_alias": "gemma4-26b-a4b-int8pc",
    "oneapi_device_selector": "level_zero:*",
    "port": "18271",
    "prefix_caching": "disabled",
    "quantization": "int8_per_channel_weight_only",
    "runtime": "vllm",
    "torchinductor_cache_dir": "/mnt/fast-ai/vllm-cache-exp/gemma4-26b-a4b-it-int8pc-gpu1/torchinductor",
    "vllm_bin": "/home/steve/.venvs/vllm-xpu/bin/vllm",
    "vllm_cache_root": "/mnt/fast-ai/vllm-cache-exp/gemma4-26b-a4b-it-int8pc-gpu1",
    "vllm_extra_args": "<unset>",
    "vllm_source_path": "/home/steve/src/vllm/vllm",
    "vllm_target_device": "xpu",
    "vllm_use_v1": "1",
    "vllm_version": "0.20.2rc1.dev13+g9557d9108.d20260620",
    "vllm_xpu_enable_xpu_graph": "1",
    "vllm_xpu_force_graph_with_comm": "1",
    "vllm_xpu_graph_noop_comm_capture": "1",
    "xpu_graph": "1",
    "ze_affinity_mask": "1"
  },
  "model_file_bytes": 51612009916,
  "model_path": "/mnt/fast-ai/llm-cache/hf/models--google--gemma-4-26B-A4B-it/snapshots/20da991ab4afab98e8f910c4a2e8f4fbefc404ad",
  "model_shard_count": 2,
  "run_dir": "/home/steve/qwen36-results-main/data/gemma4-vllm-int8pc-gpu1-piecewise-selectorfix-smoke-20260623T204041Z",
  "server_log": "/mnt/fast-ai/bench-results/gemma4-26b-a4b-vllm-int8pc/servers/gemma4-vllm-int8pc-gpu1-piecewise-selectorfix-smoke-20260623T204041Z.server.log"
}
