{"id":"cmq8yhxvo001ipb0149aoa79o","modelId":"cmq8yhxva001fpb01m0buuxx5","modelRevision":"cced56592e8c8935f8220836b4baa04dfd389118","hardwareId":"cmormmlvb0009ky04i6pvj96b","engineId":"cmq8yhxvh001gpb01rcufqsrm","userId":"cmoext1fr0000jl04skh3hjqf","promptTokens":512,"outputTokens":512,"contextLength":32768,"batchSize":1,"prefillTokens":null,"ttftMs":76.45406149094924,"tokSOut":99.42835812273452,"tokSPrefill":null,"tokSTotal":196.3252731420561,"peakVramGb":null,"notes":"Quality-gated Qwen3.6 35B Quark W8A8 INT8 run on 4x Intel Arc Pro B70. Metric is p512/n512 streaming /v1/completions, temperature 0, four repeats after warmup. tokSOut is corrected steady-state output throughput after first streamed text chunk; e2e output mean was 98.16 tok/s. Frontdoor quality rerun8 passed exact canaries, JSON schema/semantics, copy phrase, 8K long-context needle, repeat stability, and baseline hash parity. Peak VRAM was not submitted because this run used --skip-vram.","adminNotes":null,"status":"APPROVED","lastEditedAt":null,"createdAt":"2026-06-11T03:47:04.260Z","updatedAt":"2026-06-11T03:47:04.260Z","model":{"id":"cmq8yhxva001fpb01m0buuxx5","hfId":"nameistoken/Qwen3.6-35B-A3B-Quark-W8A8-INT8","displayName":"Qwen3.6-35B-A3B-Quark-W8A8-INT8","organization":"nameistoken","family":"Qwen","params":35,"activeParams":3,"isMoE":true,"architecture":"qwen3_5_moe","license":"apache-2.0","description":"Model synced from HuggingFace Hub","hfUrl":"https://huggingface.co/nameistoken/Qwen3.6-35B-A3B-Quark-W8A8-INT8","pipelineTag":"image-text-to-text","tags":"[\"transformers\",\"safetensors\",\"qwen3_5_moe\",\"image-text-to-text\",\"moe\",\"mixture-of-experts\",\"quantized\",\"int8\",\"w8a8\",\"quark\",\"vllm\",\"conversational\",\"text-generation-inference\",\"en\",\"base_model:Qwen/Qwen3.6-35B-A3B\",\"base_model:quantized:Qwen/Qwen3.6-35B-A3B\",\"license:apache-2.0\",\"endpoints_compatible\",\"8-bit\",\"region:us\"]","modalities":"[]","lastSyncedAt":"2026-06-11T03:47:04.244Z","createdAt":"2026-06-11T03:47:04.247Z","updatedAt":"2026-06-11T03:47:04.247Z","baseModelId":"cmo6t4g0t0001jv044k932p1y"},"hardware":{"id":"cmormmlvb0009ky04i6pvj96b","hwClass":"DISCRETE_GPU","gpuName":"Intel Arc Pro B70","gpuCount":4,"vramGb":32,"chipVendor":null,"chipFamily":null,"chipVariant":null,"unifiedMemoryGb":null,"npuTops":null,"cpu":null,"ramGb":15,"os":"Ubuntu 24.04.4 LTS","powerWatts":null,"isHeterogeneousGpu":false,"createdAt":"2026-05-04T20:02:59.255Z","updatedAt":"2026-05-04T20:02:59.255Z","gpuSlots":[]},"engine":{"id":"cmq8yhxvh001gpb01rcufqsrm","engineName":"vllm","engineVersion":"0.20.2rc1.dev2+gc51df4300.d20260523","quantization":"Quark W8A8 INT8","backend":null,"createdAt":"2026-06-11T03:47:04.254Z","updatedAt":"2026-06-11T03:47:04.254Z"},"engineFlags":{"id":"cmq8yhxvo001jpb016qbt4h7t","benchmarkRunId":"cmq8yhxvo001ipb0149aoa79o","commandSnippet":"VLLM_USE_V1=1 VLLM_TARGET_DEVICE=xpu XPU_GRAPH=1 VLLM_XPU_ENABLE_XPU_GRAPH=1 VLLM_XPU_FORCE_GRAPH_WITH_COMM=1 VLLM_XPU_GRAPH_NOOP_COMM_CAPTURE=1 VLLM_XPU_USE_CUSTOM_OP_COLLECTIVES=1 VLLM_XPU_COMPILE_ALLREDUCE_CUSTOM_OP=1 VLLM_XPU_CUSTOM_ALLREDUCE_GRAPH_CLONE_INPUT=1 VLLM_XPU_CUSTOM_ALLREDUCE_CLONE_INPUT=1 VLLM_XPU_QUARK_W8A8_MOE=1 VLLM_XPU_GDN_REUSE_QKVZ_BA_QUANT=clone ONEAPI_DEVICE_SELECTOR=level_zero:0,1,2,3 ZE_AFFINITY_MASK=0,1,2,3 CCL_ATL_TRANSPORT=ofi CCL_TOPO_P2P_ACCESS=1 FI_TCP_IFACE=eth1 CCL_KVS_IFACE=eth1 vllm serve /mnt/fast-ai/llm-cache/hf/models--nameistoken--Qwen3.6-35B-A3B-Quark-W8A8-INT8/snapshots/cced56592e8c8935f8220836b4baa04dfd389118 --host 127.0.0.1 --port 18080 --trust-remote-code --served-model-name qwen36-35b-a3b-fp8 --dtype auto --quantization quark --tensor-parallel-size 4 --pipeline-parallel-size 1 --distributed-executor-backend mp --max-model-len 32768 --max-num-batched-tokens 8192 --max-num-seqs 48 --gpu-memory-utilization 0.95 --kv-cache-dtype auto --no-enable-prefix-caching --language-model-only --compilation-config {\"cudagraph_mode\":\"PIECEWISE\"} --generation-config vllm","tensorParallel":4,"pipelineParallel":1,"gpuLayers":null,"splitMode":null,"kvCacheDtype":"auto","gpuMemUtil":0.95,"kvCacheSizeMb":null,"prefixCaching":false,"attentionBackend":"flash_attn","flashAttn":true,"chunkedPrefill":true,"prefillChunkSize":8192,"contBatching":true,"cpuOffloadGb":null,"cpuLayers":null,"ropeScaling":null,"ropeScale":null,"yarnExtFactor":null,"engineQuant":"quark","sglangQuant":null,"maxRunningSeqs":48,"schedulerDelayFactor":null,"numParallel":null,"concurrency":1,"specDecoding":false,"specMethod":null,"specModel":null,"specDraftModel":null,"specNumTokens":null,"specNgramSize":null,"specDraftTp":null,"mtpEnabled":false,"mtpDraftLayers":null,"temperature":0,"topP":null,"topK":null,"minP":null,"repeatPenalty":null,"mirostat":null,"extraFlags":"XPU PIECEWISE graph capture; graph-safe vLLM custom all-reduce; native XPU Quark W8A8 INT8 dense and MoE paths; GDN qkvz/ba quant reuse clone guard.","createdAt":"2026-06-11T03:47:04.260Z"},"user":{"id":"cmoext1fr0000jl04skh3hjqf","verified":false}}