[
  {
    "label": "gemma4-26b-a4b-q8-b70-llamacpp-mtp-n7-q8target-q40draft-fresh-20260624T0812",
    "payload": {
      "batchSize": 1,
      "contextLength": 8192,
      "engineFlags": {
        "actualOutputTokens": 512,
        "actualPromptTokens": 588,
        "attentionBackend": "llama.cpp SYCL",
        "backendSampling": false,
        "benchmarkJson": "data/gemma4-q8-gpu0-mtp-n7-draftq40-full-20260624T081218Z/p512o512.json",
        "cachedTokensRow0": 0,
        "canaryRowsPassed": 384,
        "chunkedPrefill": false,
        "commandSnippet": "ONEAPI_DEVICE_SELECTOR=level_zero:0 GGML_SYCL_ENABLE_VMM=0 LLAMA_SERVER=/home/steve/src/llama.cpp-gemma-record-stack/build-sycl-b70-aot-bmg-g31/bin/llama-server MTP_DRAFT_MODEL=/mnt/fast-ai/llm-models/gemma4-26b-a4b-it-q8-gguf/MTP/gemma-4-26B-A4B-it-Q4_0-MTP.gguf MTP_DRAFT_FAST_ARGMAX=1 MTP_BACKEND_SAMPLING=0 MTP_N_MAX=7 MTP_N_MIN=2 MTP_P_MIN=0.12 MTP_DRAFT_DEVICE=SYCL0 MTP_DRAFT_THREADS=32 MTP_DRAFT_THREADS_BATCH=32 MTP_EXTRA_ARGS=\"--ctx-checkpoints 0\" CTX_SIZE=8192 BATCH_SIZE=512 UBATCH_SIZE=512 POLL=100 THREADS=16 CANARY_REPEATS=96 BENCH_REPEATS=8 PROMPT_TOKENS=512 MAX_TOKENS=512 BENCH_PROMPT_MODE=filled-long scripts/run-gemma4-26b-mtp-candidate.sh gemma4-q8-gpu0-mtp-n7-draftq40-full-20260624T081218Z",
        "concurrency": 1,
        "contBatching": false,
        "draftModelFile": "gemma-4-26B-A4B-it-Q4_0-MTP.gguf",
        "draftModelFileBytes": 321126560,
        "draftModelRepo": "local quantized from unsloth/gemma-4-26B-A4B-it-GGUF F16 MTP draft via llama-quantize Q4_0",
        "extraFlags": "AOT BMG-G31 SYCL build; VMM disabled; poll=100; ubatch=512; draft threads/batch=32; ctx-checkpoints=0; fast argmax draft selection; no backend sampling; reasoning off. Quantization file is Unsloth UD-Q8_K_XL GGUF; submitted quantization normalized to Q8_K_XL for API schema.",
        "firstRequestTokSOut": 95.26352416631231,
        "flashAttn": false,
        "freshResponseValidity": "headline tokSOut uses row 0 only: first measured request after canaries, cached_tokens=0. Repeated rows are support/stability only and are not used as headline. Speculative draft source is a Q4_0 Gemma MTP draft for the current request, not ngram/history or learned repeated continuation reuse.",
        "kvCacheDtype": "f16",
        "llamaCppCommit": "c926ad098",
        "maxRunningSeqs": 1,
        "pipelineParallel": 1,
        "prefixCaching": false,
        "promptMode": "filled-long",
        "promptTokensRequested": 512,
        "serverLog": "/mnt/fast-ai/bench-results/gemma4-26b-a4b-q8/servers/gemma4-q8-gpu0-mtp-n7-draftq40-full-20260624T081218Z.server.log",
        "specDecoding": true,
        "specDraftModel": "/mnt/fast-ai/llm-models/gemma4-26b-a4b-it-q8-gguf/MTP/gemma-4-26B-A4B-it-Q4_0-MTP.gguf",
        "specDraftNMin": 2,
        "specDraftPMin": 0.12,
        "specDraftQuantization": "Q4_0 draft only; target/verifier remains UD-Q8_K_XL",
        "specMethod": "draft-mtp",
        "specNumTokens": 7,
        "summaryJson": "data/gemma4-q8-gpu0-mtp-n7-draftq40-full-20260624T081218Z/summary.json",
        "temperature": 0,
        "tensorParallel": 1,
        "tokSOutMaxSupport": 95.51641267061656,
        "tokSOutMeanSupport": 95.38558173206405,
        "tokSOutMinSupport": 95.22637213832232
      },
      "engineName": "llama.cpp",
      "engineVersion": "c926ad098 local B70 SYCL/AOT patch stack",
      "hardware": {
        "cpu": "AMD Ryzen Threadripper PRO 5955WX 16-Cores",
        "gpuCount": 1,
        "gpuName": "Intel Arc Pro B70",
        "hwClass": "DISCRETE_GPU",
        "os": "Ubuntu 24.04.4 LTS",
        "ramGb": 128,
        "vramGb": 32
      },
      "hfId": "unsloth/gemma-4-26B-A4B-it-GGUF",
      "modelRevision": "gemma-4-26B-A4B-it-UD-Q8_K_XL.gguf",
      "notes": "Fresh-response-valid Gemma 4 26B A4B Q8-class GGUF run on one Intel Arc Pro B70. Headline is the first measured p512/o512 row with cached_tokens=0, per fresh-response rule; later repeated rows are support only and are not averaged into the headline. Validation: chat canary 384/384 pass, all benchmark rows cached_tokens=0. Target model remains Unsloth UD-Q8_K_XL; only the MTP draft model is Q4_0, and all accepted tokens are verified by the Q8 target. High n-gram/history-accelerated rows from this project are intentionally excluded from this claim. Run dir: data/gemma4-q8-gpu0-mtp-n7-draftq40-full-20260624T081218Z . Experiment note: experiments/gemma4-26b-a4b-q8-b70/sweeps/20260624T0235-mtp-fused-unroll-feasibility.md . GitHub repo: https://github.com/steveseguin/b70-optimization-lab",
      "outputTokens": 512,
      "promptTokens": 588,
      "quantization": "Q8_K_XL",
      "tokSOut": 95.26352416631231,
      "tokSTotal": 81.28549578539435,
      "ttftMs": 924.2217319842894
    }
  }
]
