{
  "date": "2026-06-07",
  "host": "steve-b70s",
  "model": "Intel/gemma-4-12B-it-int4-AutoRound",
  "served_model_name": "gemma4-12b-it-int4-autoround",
  "profile": "gemma4-12b-it-int4-autoround-c8",
  "status": "production",
  "endpoint": "http://0.0.0.0:8000/v1",
  "auth": "none",
  "modalities": [
    "text",
    "image"
  ],
  "hardware": {
    "gpu": "Intel Arc Pro B70",
    "gpu_count": 4,
    "tensor_parallel_size": 4
  },
  "runtime": {
    "engine": "vLLM",
    "backend": "XPU",
    "dtype": "bfloat16",
    "quantization": "Intel AutoRound/INC INT4 W4A16",
    "max_model_len": 32768,
    "max_num_seqs": 8,
    "max_active_generations": 8,
    "max_num_batched_tokens": 4096,
    "prefix_caching": true,
    "frontdoor_queue_timeout_s": 0,
    "frontdoor_queue_policy": "fail-fast when all 8 generation slots are busy",
    "xpu_graph": true,
    "compilation_config": {
      "use_inductor_graph_partition": true,
      "compile_sizes": [
        1
      ],
      "cudagraph_mode": "PIECEWISE"
    }
  },
  "startup": {
    "torch_compile_s": 4.12,
    "graph_capture_s": 4.0,
    "engine_init_s": 19.05,
    "gpu_kv_cache_tokens": 1004909,
    "full_32768_token_concurrency_estimate": 30.67
  },
  "quality_gate": {
    "expected_outputs_passed": true,
    "baseline_hashes_matched": true,
    "checks": {
      "exact_ok": "OK",
      "copy_phrase": "satin cobalt orbit",
      "small_arithmetic": "7",
      "red_image": "Red"
    }
  },
  "benchmarks": {
    "c8_119_prompt_128_output_promoted_runs": [
      {
        "mean_ttft_s": 1.6576312522229273,
        "wall_output_tok_s": 613.7164231537495
      },
      {
        "mean_ttft_s": 1.3544009968900355,
        "wall_output_tok_s": 751.5790896495571
      },
      {
        "mean_ttft_s": 1.3651087650941918,
        "wall_output_tok_s": 745.4813946672974
      }
    ],
    "c8_119_prompt_128_output_mean": {
      "mean_ttft_s": 1.4590470047357182,
      "wall_output_tok_s": 703.5923024902013
    },
    "c8_30690_prompt_1_output_validation": {
      "mean_ttft_s": 22.27967816920136,
      "wall_output_tok_s": 0.20433003996603097
    },
    "c8_sustained_decode": {
      "prompt_tokens_each": 119,
      "concurrency": 8,
      "output_256_repeat_mean": {
        "mean_ttft_s": 2.529805063267122,
        "wall_output_tok_s": 796.1763558639424
      },
      "output_512_repeat_mean": {
        "mean_ttft_s": 2.5417523958312813,
        "wall_output_tok_s": 780.9684204492766
      },
      "output_1024_repeat_mean": {
        "mean_ttft_s": 2.518777515135298,
        "wall_output_tok_s": 731.1185687690644
      }
    },
    "c1_c2_c4_c8_512_output_scaling": {
      "prompt_tokens_each": 119,
      "output_tokens_each": 512,
      "c1": {
        "mean_ttft_s": 2.189566157059744,
        "wall_output_tok_s": 112.77097490423834
      },
      "c2": {
        "mean_ttft_s": 2.4145851635257713,
        "wall_output_tok_s": 205.46827380978502
      },
      "c4": {
        "mean_ttft_s": 2.500018831982743,
        "wall_output_tok_s": 398.9838032188172
      },
      "c8": {
        "mean_ttft_s": 2.526166310359258,
        "wall_output_tok_s": 784.6918451280836
      }
    },
    "c8_long_prompt_decode": {
      "output_tokens_each": 128,
      "concurrency": 8,
      "coldish_15357_prompt_tokens": {
        "mean_ttft_s": 21.45621856309299,
        "wall_output_tok_s": 47.12305574320813
      },
      "coldish_28774_prompt_tokens": {
        "mean_ttft_s": 22.43952533364063,
        "wall_output_tok_s": 45.1305820320481
      },
      "prefix_cache_repeat_28774_prompt_tokens": {
        "mean_ttft_s": 3.746319937883527,
        "wall_output_tok_s": 270.739868095527
      }
    },
    "production_soak": {
      "main_run_started_utc": "20260607T090844Z",
      "scheduled_cycles": 32,
      "clean_cycles": 31,
      "quality_anomaly_cycles": [
        17
      ],
      "clean_cycles_wall_output_tok_s": {
        "mean": 781.0416678960535,
        "min": 765.4444488463562,
        "max": 784.3659935750953
      },
      "clean_cycles_mean_ttft_s": {
        "mean": 2.5505383233093815,
        "min": 2.516572086882661,
        "max": 2.6483440785086714
      },
      "quality_anomaly": {
        "cycle": 17,
        "failed_check": "copy_phrase",
        "expected": "satin cobalt orbit",
        "observed": "slh cobalt orbit",
        "immediate_manual_reruns_passed": 3,
        "quality_stress_repeats_passed": 25
      },
      "harness_bug": "Original final-interval sleep logic spun after the last scheduled cycle; cycles 33+ in the first run are not normal soak samples. scripts/run-gemma4-production-soak.sh was fixed to sleep to the deadline and exit.",
      "fixed_harness_continuation": {
        "started_utc": "20260607T170030Z",
        "cycles": [
          {
            "cycle": 1,
            "utc": "20260607T170030Z",
            "quality_expected": true,
            "quality_baseline": true,
            "wall_output_tok_s": 779.2028275527277,
            "mean_ttft_s": 2.5587674672715366
          },
          {
            "cycle": 2,
            "utc": "20260607T171530Z",
            "quality_expected": true,
            "quality_baseline": true,
            "wall_output_tok_s": 784.7459001414585,
            "mean_ttft_s": 2.5455494657508098
          }
        ]
      },
      "frontdoor_streaming_fix": {
        "cause": "The LAN frontdoor proxied SSE responses with response.read(65536), which buffered text/event-stream events and made client-observed TTFT look much higher than backend TTFT.",
        "fix": "Forward text/event-stream responses line-by-line and flush each SSE line in scripts/openai-lan-frontdoor.py.",
        "backend_first_text_s": {
          "max_tokens_128": 0.0341,
          "max_tokens_512": 0.0342
        },
        "frontdoor_first_text_after_fix_s": {
          "max_tokens_128": 0.0343,
          "max_tokens_512": 0.0371
        },
        "public_endpoint_clean_bench_after_fix": {
          "c1_119_prompt_512_output": {
            "mean_ttft_s": 0.036198701011016965,
            "wall_output_tok_s": 112.86580140875809
          },
          "c8_119_prompt_512_output": {
            "mean_ttft_s": 0.09884867299115285,
            "wall_output_tok_s": 783.6173319534963
          }
        },
        "fail_fast_overload_check": {
          "simultaneous_generation_requests": 9,
          "admitted_requests": 8,
          "rejected_requests": 1,
          "rejected_status": 503,
          "rejected_elapsed_s": 0.221,
          "admitted_first_stream_s_range": [
            0.25,
            0.312
          ]
        }
      }
    }
  },
  "raw_result_paths": {
    "xpugraph_validation": "/mnt/fast-ai/bench-results/gemma4-12b-it-int4-autoround/xpugraph-validation-20260607T075901Z",
    "post_promotion_validation": "/mnt/fast-ai/bench-results/gemma4-12b-it-int4-autoround/prod-c8-xpugraph-promoted-20260607T080322Z",
    "sustained_decode_256": "/mnt/fast-ai/bench-results/gemma4-12b-it-int4-autoround/prod-c8-xpugraph-256o-repeat-20260607T084540Z",
    "sustained_decode_512": "/mnt/fast-ai/bench-results/gemma4-12b-it-int4-autoround/prod-c8-xpugraph-512o-repeat-20260607T084633Z",
    "sustained_decode_1024": "/mnt/fast-ai/bench-results/gemma4-12b-it-int4-autoround/prod-c8-xpugraph-1024o-repeat-20260607T084718Z",
    "sustained_decode_scaling_512": "/mnt/fast-ai/bench-results/gemma4-12b-it-int4-autoround/prod-c8-xpugraph-scaling-512o-20260607T084806Z",
    "long_prompt_decode": "/mnt/fast-ai/bench-results/gemma4-12b-it-int4-autoround/prod-c8-xpugraph-longprompt-decode-20260607T090436Z",
    "long_prompt_decode_prefix_cache_repeat": "/mnt/fast-ai/bench-results/gemma4-12b-it-int4-autoround/prod-c8-xpugraph-longprompt-30000p-128o-cache-repeat-20260607T090558Z.json",
    "production_soak_main": "/mnt/fast-ai/bench-results/gemma4-12b-it-int4-autoround/prod-c8-soak-20260607T090844Z",
    "production_soak_fixed_harness_continuation": "/mnt/fast-ai/bench-results/gemma4-12b-it-int4-autoround/prod-c8-soak-20260607T170030Z",
    "profile_trial_xpugraph": "/mnt/fast-ai/bench-results/gemma4-12b-it-int4-autoround/profile-trials/gemma4-12b-it-int4-autoround-c8-xpugraph-20260607T074722Z",
    "profile_trial_mbt8192": "/mnt/fast-ai/bench-results/gemma4-12b-it-int4-autoround/profile-trials/gemma4-12b-it-int4-autoround-c8-mbt8192-20260607T073623Z",
    "profile_trial_mbt2048": "/mnt/fast-ai/bench-results/gemma4-12b-it-int4-autoround/profile-trials/gemma4-12b-it-int4-autoround-c8-mbt2048-20260607T074213Z"
  },
  "localmaxxing": [
    {
      "id": "cmq3jm75g000tlj01bx4frdf0",
      "shape": "c8, 119 prompt tokens, 256 output tokens",
      "tokSOut": 796.1763558639424,
      "ttftMs": 2529.805063267122,
      "status": "APPROVED"
    },
    {
      "id": "cmq3jm7cx000wlj01wm75wqmk",
      "shape": "c8, 119 prompt tokens, 512 output tokens",
      "tokSOut": 780.9684204492766,
      "ttftMs": 2541.752395831281,
      "status": "APPROVED"
    }
  ],
  "rejected_branches": [
    {
      "profile": "gemma4-12b-it-int4-autoround-c8-mbt8192",
      "quality_matched": true,
      "reason": "Short decode fell to about 245.75 tok/s and GPU KV dropped to 730,379 tokens."
    },
    {
      "profile": "gemma4-12b-it-int4-autoround-c8-mbt2048",
      "quality_matched": true,
      "reason": "GPU KV rose to 1,201,507 tokens, but short decode fell to about 235.37 tok/s."
    },
    {
      "profile": "gemma4-12b-it-int4-autoround-c8-gmem097",
      "quality_matched": null,
      "reason": "Rejected at startup. Free memory on xpu:0 was about 30.61/31.89 GiB, below the 0.97 utilization request of about 30.93 GiB."
    },
    {
      "profile": "gemma4-12b-it-int4-autoround-c8-gmem096",
      "quality_matched": null,
      "reason": "Rejected at startup or engine initialization near the same memory boundary."
    },
    {
      "profile": "gemma4-12b-it-int4-autoround-c8-cs1-8",
      "quality_matched": true,
      "reason": "Rejected. First compile had six ocloc/IGC error-code-245 fallbacks and torch.compile took 315.80 s; cached repeat validation later hit UR_RESULT_ERROR_DEVICE_LOST during sampling."
    },
    {
      "profile": "gemma4-12b-it-int4-autoround-c8-xpugraph-mbt2048",
      "quality_matched": null,
      "reason": "Rejected. It raised GPU KV to 1,201,940 tokens and 36.68x theoretical 32K concurrency, but first compile took 214.55 s and it hit UR_RESULT_ERROR_DEVICE_LOST during the canary/sampling path."
    },
    {
      "profile": "gemma4-12b-it-int4-autoround-c8-nolog",
      "quality_matched": true,
      "reason": "Not promoted. Short decode was 714.81 tok/s, within the promoted graph profile's normal variance, and one-token TTFT worsened."
    }
  ]
}
