[
  {
    "label": "deepseek-v4-flash-k160-b70-tp4-dspark7-sharded-target-argmax-realistic-80.820tok-20260718",
    "payload": {
      "hfId": "0xSero/DeepSeek-V4-Flash-180B",
      "modelRevision": "7c360e1cd4a5168099dbc54d16d929bf6df04990",
      "engineName": "vllm",
      "engineVersion": "0.1.dev1172+g4a6fd8747.xpu; local vLLM 264c7f2f7; XPU kernels 313156737; oneCCL 48fda4f0e",
      "backend": "xpu",
      "quantization": "fp8",
      "hardware": {
        "hwClass": "DISCRETE_GPU",
        "gpuName": "Intel Arc Pro B70",
        "gpuCount": 4,
        "vramGb": 32,
        "cpu": "AMD Ryzen Threadripper PRO 5955WX 16-Cores",
        "ramGb": 128,
        "os": "Ubuntu 24.04.4 LTS"
      },
      "contextLength": 256,
      "batchSize": 1,
      "promptTokens": 62,
      "outputTokens": 128,
      "tokSOut": 80.82005189243556,
      "tokSTotal": 67.76281785757796,
      "ttftMs": 342.77765700244345,
      "engineFlags": {
        "apiMode": "chat",
        "attentionBackend": "vLLM XPU",
        "benchmarkJson": "/mnt/fast-ai/bench-results/deepseek-v4-flash-xpu/dspark7-sharded-target-argmax-candidate-20260718T2100Z/strict-screen.json",
        "confirmationBenchmarkJson": "/mnt/fast-ai/bench-results/deepseek-v4-flash-xpu/dspark7-sharded-target-argmax-candidate-20260718T2100Z/strict-confirmation.json",
        "thirdBenchmarkJson": "/mnt/fast-ai/bench-results/deepseek-v4-flash-xpu/dspark7-sharded-target-argmax-candidate-20260718T2100Z/strict-third.json",
        "commandIdentityEnv": "/mnt/fast-ai/bench-results/deepseek-v4-flash-xpu/dspark7-sharded-target-argmax-candidate-20260718T2100Z/identity.txt",
        "freshResponseHeadlineValid": true,
        "freshResponseValidity": "Fixed realistic prompt suite; each prompt sent once as a cold response; cached_tokens=0 for every request; no prompt/KV cache reuse, context checkpoints, response reuse, n-gram/history acceleration, or warmed repeated prompts.",
        "headlineUse": "fresh-realistic-suite",
        "historyAccelerated": false,
        "responseReuse": false,
        "prefixCaching": false,
        "contextCheckpoints": 0,
        "localmaxxingSubmissionAllowedUnderCurrentPolicy": true,
        "primaryMetricName": "median_tok_s_1_100_after_ttft",
        "metricWindowGeneratedTokens": 100,
        "realisticSuiteGatePassed": true,
        "realisticSuiteCachedTokensAllZero": true,
        "realisticSuiteId": "rapid-model-snapshots-b70-realistic-v1",
        "realisticSuitePath": "repro/rapid-model-snapshots-b70/realistic-suite-v1.json",
        "realisticSuiteVersion": 1,
        "tokenTimingSource": "openai_stream_token_ids_chunk_timestamp",
        "temperature": 0,
        "tokSOutMedian": 80.82005189243556,
        "tokSOutP10": 71.66955635980304,
        "tokSOutMean": 80.0334096575548,
        "tokSOutStdev": 6.2791635620605195,
        "tokSFullAfterTtftMedian": 80.74548522063007,
        "tokSTotalWallMedian": 67.76281785757796,
        "ttftMsMedian": 342.77765700244345,
        "commandSnippet": "VLLM_XPU_GREEDY_SHARDED_TARGET_ARGMAX=1 VLLM_XPU_V4_ROUTER_NORM_MAX_M=8 VLLM_XPU_V4_COMPRESSOR_BATCHED_EXACT_MAX_M=8 VLLM_XPU_V4_BLOCK_FP8_W8A16_MAX_M=8 VLLM_XPU_MXFP4_SMALL_M_N=128 DSPARK_GRAPH_MODE=piecewise DSPARK_DRAFT_GRAPH_MODE=piecewise DSPARK_SPEC_TOKENS=7 VLLM_XPU_DSPARK_EXACT_QUERY_CAPTURE=1 VLLM_XPU_GREEDY_FUSED_REJECTION=1 VLLM_XPU_DSPARK_FIXED_M7_TARGET_INPUTS=1 VLLM_XPU_DSPARK_PERSISTENT_MARKOV=1 VLLM_XPU_DSPARK_REPLICATED_MARKOV_W1=1 experiments/deepseek-v4-flash-reap-xpu-b70/scripts/serve-k160-dspark-candidate.sh",
        "vllmCommit": "264c7f2f7df21ddeeab32ecca0353133344f1ac9",
        "xpuKernelCommit": "31315673737d95da0f79179c8f755260ef02c1d6",
        "oneCCLCommit": "48fda4f0e074db005596d6899d5227d3f0316c12",
        "compilationConfig": "{\"cudagraph_mode\":\"PIECEWISE\"}",
        "tensorParallelSize": 4,
        "expertParallel": true,
        "targetGraphMode": "PIECEWISE",
        "draftGraphMode": "PIECEWISE breakable exact-M7",
        "targetVerifierWidth": 8,
        "specDecoding": true,
        "specMethod": "dspark",
        "specNumTokens": 7,
        "targetModelVerifiedAcceptedTokens": true,
        "draftModel": "deepseek-ai/DeepSeek-V4-Flash",
        "draftModelRevision": "aa22cb07426656189b2573b8e77a9b7333b8ae0f",
        "persistentMarkov": true,
        "replicatedMarkovW1": true,
        "fixedM7TargetInputs": true,
        "greedyFusedRejection": true,
        "m8CompressorBmmExact": true,
        "m8SelectiveW8A16": true,
        "m8NativeRouterNormExact": true,
        "shardedTargetArgmax": true,
        "shardedTargetArgmaxGuardedGreedyOnly": true,
        "nativeTargetTokenRejection": true,
        "fullVocabularyLogitsAllgatherElided": true,
        "fullVocabularyFp32SamplerMaterializationElided": true,
        "strictSuiteMedians": [80.82005189243556, 76.90017809136465, 78.28722593298039],
        "suiteMedianOfMedians": 78.28722593298039,
        "exactCanarySuites": 4,
        "exactCanaryRequests": 24,
        "realisticRequests": 36,
        "singleActiveGeneration": true,
        "aggregateThroughput": false,
        "priorRecordTokS": 80.1635783863126,
        "priorRecordLocalMaxxingId": "cmrqp2uoa05ublg01lh6yluj8"
      },
      "notes": "Four-B70 single-session DeepSeek V4 Flash K160 DSpark7 record. The guarded greedy target path projects each local 32320-token vocabulary shard, gathers only tiny top-1 value/index pairs, and commits target tokens with a native SYCL rejection op instead of all-gathering and sampling full 129280-token FP32 logits. Three fresh strict suite medians are 80.820052, 76.900178, and 78.287226 tok/s; all 36 requests are cache-zero. Four ordered exact-canary suites pass 24/24. Unsupported sampler settings fail closed to the canonical path. The unchanged K160 target verifies accepted tokens. No aggregate throughput, cache/history reuse, model substitution, or benchmark routing."
    }
  }
]
