{
 "format": "b70-model-package-catalog-v1",
 "source": "packages/*/package.json",
 "packages": [
  {
   "manifest": "packages/gemma4-26b-a4b-q8-b70/package.json",
   "format": "b70-model-package-v1",
   "id": "gemma4-26b-a4b-q8-b70-125tps-20260701",
   "name": "Gemma 4 26B A4B UD-Q8_K_XL on one Intel Arc Pro B70",
   "status": "candidate",
   "audience": "expert",
   "guide": "repro/gemma4-26b-a4b-q8-b70-125tps-20260701/README.md",
   "clean_host_tested": false,
   "library": {
    "model_family": "Gemma 4",
    "publisher": "Google",
    "variant": "26B A4B",
    "summary": "Google's Gemma 4 26B mixture-of-experts on one Arc Pro B70 in 8-bit form, with its built-in draft head speeding up generation. A strong all-rounder that also reads images.",
    "quantization": "UD-Q8_K_XL + Q4_0 MTP",
    "runtime_label": "llama.cpp SYCL",
    "operating_systems": [
     "Linux"
    ],
    "delivery": [
     "native"
    ],
    "modalities": [
     "text"
    ],
    "use_cases": [
     "general",
     "coding",
     "long context"
    ],
    "tags": [
     "one card",
     "speculative decoding",
     "source reconstruction"
    ],
    "published_at": "2026-08-22",
    "featured_metric": {
     "value": 122.16035656735696,
     "unit": "tok/s",
     "label": "decode (MTP-assisted)",
     "scope": "Class-balanced median of per-input-class medians using conventional 99-interval rates on the fixed cold suite with the Q4_0 MTP draft assisting. The all-prompt median is 123.727369 tok/s; the historical 100-event compatibility figure is 124.977141 tok/s.",
     "evidence": "data/gemma4-q8-gpu0-finalpostnorm-reproexact-full512-20260701T084728Z/summary.json"
    }
   },
   "contributors": [
    {
     "id": "b70-optimization-lab",
     "name": "neural.download lab",
     "kind": "lab",
     "profile": "https://github.com/steveseguin/b70-optimization-lab",
     "contribution": "Gemma 4 B70 bring-up, llama.cpp/SYCL and MoE optimization, target-verified MTP work, source reconstruction, and package validation.",
     "status": "integrated",
     "validated_effect": "The lane's roughly 15.55 tok/s early one-card Q8 starting point and 124.977 tok/s high use different configurations, so they document lab history and are not presented as one like-for-like boost.",
     "evidence": "results/gemma4-26b-a4b-q8-b70/README.md"
    }
   ],
   "performance_profiles": [
    {
     "id": "decode-vs-context",
     "label": "Decode over active context",
     "metric": "decode",
     "unit": "tok/s",
     "x_label": "Actual prompt / active context tokens",
     "scope": "Mean after-TTFT decode rate across four independent one-B70 lanes; unique cache-zero prompts and exact JSON gates.",
     "evidence": "data/gemma4-26b-a4b-q8-b70-context-performance-profile-20260702.json",
     "points": [
      {
       "context_tokens": 741,
       "value": 161.89843663684636,
       "samples": 4
      },
      {
       "context_tokens": 2806,
       "value": 144.8662058096366,
       "samples": 4
      },
      {
       "context_tokens": 5643,
       "value": 141.11391660033576,
       "samples": 4
      },
      {
       "context_tokens": 10976,
       "value": 135.93492558486764,
       "samples": 4
      },
      {
       "context_tokens": 16213,
       "value": 127.63665710590647,
       "samples": 4
      },
      {
       "context_tokens": 22730,
       "value": 120.48241930243164,
       "samples": 4
      },
      {
       "context_tokens": 30400,
       "value": 114.00353788462567,
       "samples": 4
      },
      {
       "context_tokens": 32571,
       "value": 114.8486529751413,
       "samples": 4
      }
     ]
    },
    {
     "id": "prefill-vs-context",
     "label": "Prefill over prompt length",
     "metric": "prefill",
     "unit": "tok/s",
     "x_label": "Actual prompt tokens",
     "scope": "Approximate prompt tokens divided by TTFT, averaged across four independent one-B70 lanes.",
     "evidence": "data/gemma4-26b-a4b-q8-b70-context-performance-profile-20260702.json",
     "points": [
      {
       "context_tokens": 741,
       "value": 847.9619132833765,
       "samples": 4
      },
      {
       "context_tokens": 2806,
       "value": 1401.0150752344389,
       "samples": 4
      },
      {
       "context_tokens": 5643,
       "value": 1469.4385849395155,
       "samples": 4
      },
      {
       "context_tokens": 10976,
       "value": 1330.3881630879202,
       "samples": 4
      },
      {
       "context_tokens": 16213,
       "value": 1256.9527613956354,
       "samples": 4
      },
      {
       "context_tokens": 22730,
       "value": 1128.9763236685953,
       "samples": 4
      },
      {
       "context_tokens": 30400,
       "value": 1028.1957534609096,
       "samples": 4
      },
      {
       "context_tokens": 32571,
       "value": 1001.687171893108,
       "samples": 4
      }
     ]
    },
    {
     "id": "ttft-vs-context",
     "label": "Time to first token",
     "metric": "ttft",
     "unit": "ms",
     "x_label": "Actual prompt tokens",
     "scope": "Mean TTFT across four independent one-B70 lanes under the same cache-zero service sweep.",
     "evidence": "data/gemma4-26b-a4b-q8-b70-context-performance-profile-20260702.json",
     "points": [
      {
       "context_tokens": 741,
       "value": 873.9445414976217,
       "samples": 4
      },
      {
       "context_tokens": 2806,
       "value": 2002.938948746305,
       "samples": 4
      },
      {
       "context_tokens": 5643,
       "value": 3840.5072987661697,
       "samples": 4
      },
      {
       "context_tokens": 10976,
       "value": 8250.644568965072,
       "samples": 4
      },
      {
       "context_tokens": 16213,
       "value": 12899.11724528065,
       "samples": 4
      },
      {
       "context_tokens": 22730,
       "value": 20134.24033729825,
       "samples": 4
      },
      {
       "context_tokens": 30400,
       "value": 29567.641104455106,
       "samples": 4
      },
      {
       "context_tokens": 32571,
       "value": 32517.333841737127,
       "samples": 4
      }
     ]
    }
   ],
   "hardware": {
    "accelerator": "Intel Arc Pro B70 32 GiB",
    "cards": 1,
    "target_model_bytes": 27636230944
   },
   "model": {
    "repository": "unsloth/gemma-4-26B-A4B-it-GGUF",
    "revision": "3bb10d594514ef4edb7f3a65d41a7e4eb8c5767a",
    "manifest": "repro/gemma4-26b-a4b-q8-b70-125tps-20260701/model-manifest.json"
   },
   "runtime": {
    "kind": "native",
    "project": "ggml-org/llama.cpp",
    "revision": "c926ad09857517978575d6a74d225b463f7417a0",
    "build": "repro/gemma4-26b-a4b-q8-b70-125tps-20260701/restore-and-build.sh"
   },
   "project_patches": {
    "required": true,
    "items": [
     "patches/gemma4-26b-a4b-q8-b70/llama-cpp-c926ad098-gemma4-q8-record-source-20260701.diff.gz.b64"
    ],
    "decoded_sha256": "2dab9dce3d6a41cba8edad559eb754088c6f5ca1de6531f408c069e45b7f727a"
   },
   "commands": {
    "preflight": "LLAMA_SERVER=/path/to/build/bin/llama-server MODEL=/models/gemma-4-26B-A4B-it-UD-Q8_K_XL.gguf MTP_DRAFT_MODEL=/models/MTP/gemma-4-26B-A4B-it-Q4_0-MTP.gguf DRAFT_SHA256=<local-digest> repro/gemma4-26b-a4b-q8-b70-125tps-20260701/preflight.sh",
    "launch": "Run the foreground full-gate command documented in repro/gemma4-26b-a4b-q8-b70-125tps-20260701/README.md",
    "health": "The full-gate wrapper waits for the local llama-server health endpoint before testing",
    "benchmark": "repro/gemma4-26b-a4b-q8-b70-125tps-20260701/run.sh performs the 512-row canary and 12-prompt cold benchmark",
    "stop": "The wrapper stops its server; otherwise Ctrl-C and verify pgrep -x llama-server returns no process"
   },
   "dependencies": [
    "repro/gemma4-26b-a4b-q8-b70-125tps-20260701/README.md",
    "repro/gemma4-26b-a4b-q8-b70/long-context-suite-v1.json",
    "repro/gemma4-26b-a4b-q8-b70/realistic-suite-v1.json",
    "repro/gemma4-26b-a4b-q8-b70/run-vdr2-selecteddown-record.sh",
    "scripts/bench-openai-long-context-suite.py",
    "scripts/bench-openai-realistic-suite.py",
    "scripts/bench-openai-single-decode.py",
    "scripts/gemma4-text-canary.py",
    "scripts/run-gemma4-26b-first-baseline.sh",
    "scripts/run-gemma4-26b-llamacpp-replica.sh",
    "repro/gemma4-26b-a4b-q8-b70-125tps-20260701/model-manifest.json",
    "repro/gemma4-26b-a4b-q8-b70-125tps-20260701/restore-and-build.sh",
    "repro/gemma4-26b-a4b-q8-b70-125tps-20260701/prepare-draft.sh",
    "repro/gemma4-26b-a4b-q8-b70-125tps-20260701/preflight.sh",
    "repro/gemma4-26b-a4b-q8-b70-125tps-20260701/run.sh",
    "patches/gemma4-26b-a4b-q8-b70/README.md",
    "patches/gemma4-26b-a4b-q8-b70/record-source-verification-20260822.json",
    "results/gemma4-26b-a4b-q8-b70/README.md",
    "data/gemma4-26b-a4b-q8-b70-context-performance-profile-20260702.json",
    "data/gemma4-long-context-service-gate-20260702Tservice-ladder-current-rep4.json",
    "data/gemma4-q8-gpu0-finalpostnorm-reproexact-full512-20260701T084728Z/summary.json",
    "patches/gemma4-26b-a4b-q8-b70/llama-cpp-c926ad098-gemma4-q8-record-source-20260701.diff.gz.b64"
   ],
   "evidence": [
    "patches/gemma4-26b-a4b-q8-b70/record-source-verification-20260822.json",
    "data/gemma4-q8-gpu0-finalpostnorm-reproexact-full512-20260701T084728Z/summary.json",
    "data/gemma4-q8-gpu0-125repro-compat2026.1-20260907T040139Z/summary.json",
    "data/gemma4-q8-gpu0-125repro-container2026.0-20260907T040615Z/summary.json",
    "data/gemma4-q8-gpu0-125repro-container2026.0-run2-20260907T041051Z/summary.json"
   ],
   "missing": [
    "historical llama-server binary SHA-256",
    "historical locally quantized Q4_0 MTP draft SHA-256 and byte size (the reconstructed draft is repeat-stable at 1f6706e4\u2026 across three builds)",
    "beginner recovery flow exercised end-to-end on a fresh host (documented; the pinned oneAPI 2026.0 container path and two clean-rebuild gate replays were verified on the lab host on 2026-09-07)"
   ]
  },
  {
   "manifest": "packages/laguna-s-2.1-int4-b70-125tps/package.json",
   "format": "b70-model-package-v1",
   "id": "laguna-s-2.1-int4-b70-125tps-20260731",
   "name": "Laguna S 2.1 INT4 with DFlash on four Intel Arc Pro B70 cards",
   "status": "candidate",
   "audience": "expert",
   "guide": "repro/laguna-s-2.1-int4-b70-125tps-20260731/README.md",
   "clean_host_tested": false,
   "library": {
    "model_family": "Laguna S 2.1",
    "publisher": "Poolside",
    "variant": "INT4 target plus INT4 DFlash draft",
    "summary": "Poolside's Laguna S 2.1 coding assistant on four Arc Pro B70 cards in 4-bit form, with a draft model speeding up generation. A replay of the lab's record run, quality-checked.",
    "quantization": "INT4 / BF16 KV",
    "runtime_label": "vLLM XPU",
    "operating_systems": [
     "Linux"
    ],
    "delivery": [
     "native"
    ],
    "modalities": [
     "text"
    ],
    "use_cases": [
     "general",
     "coding",
     "speculative decoding research"
    ],
    "tags": [
     "four cards",
     "TP4",
     "EP4",
     "DFlash",
     "lab replay",
     "originating host"
    ],
    "published_at": "2026-08-22",
    "featured_metric": {
     "value": 125.4619731637751,
     "unit": "tok/s",
     "label": "conventional decode median",
     "scope": "Conventional 99-interval median across the sealed 13-prompt, one-start, cache-zero record gate.",
     "evidence": "data/laguna-shared-elementwise-m12-record-20260731.json"
    }
   },
   "contributors": [
    {
     "id": "b70-optimization-lab",
     "name": "neural.download lab",
     "kind": "lab",
     "profile": "https://github.com/steveseguin/b70-optimization-lab",
     "contribution": "Laguna B70 bring-up, exact M12 verifier and DFlash path, shared-elementwise optimization, runtime provenance, fail-closed quality gates, and record packaging.",
     "status": "integrated",
     "validated_effect": "The shared-elementwise M12 record measured 125.461973 tok/s, 0.6575% above the preceding 124.642413 tok/s record, with 13/13 exact prompts and cache-zero evidence.",
     "evidence": "data/laguna-shared-elementwise-m12-record-20260731.json"
    }
   ],
   "hardware": {
    "accelerator": "Intel Arc Pro B70 32 GiB",
    "cards": 4
   },
   "model": {
    "repository": "poolside/Laguna-S-2.1-INT4",
    "revision": "4bbfc285f2f8b3b6b526274c133b7b17aae6c8cb",
    "manifest": "repro/laguna-s-2.1-int4-b70-102tps-20260726/manifests/model-release-files.sha256"
   },
   "runtime": {
    "kind": "native",
    "repository": "vllm-project/vllm",
    "revision": "1a7f61feffbc61b21b73f812d231c7426386ccdc",
    "build": "repro/laguna-s-2.1-int4-b70-125tps-20260731/README.md"
   },
   "project_patches": {
    "required": true,
    "items": [
     "patches/laguna-s-2.1-xpu-b70/vllm-laguna-shared-elementwise-m12-1a7f61fef-20260731.bundle",
     "patches/laguna-s-2.1-xpu-b70/vllm-xpu-kernels-laguna-shared-elementwise-m12-99886d783-20260731.bundle"
    ],
    "reason": "The record is tied to the lab's exact vLLM and XPU-kernel experiment commits and native artifact lock."
   },
   "commands": {
    "preflight": "repro/laguna-s-2.1-int4-b70-102tps-20260726/verify-record.sh && git bundle verify patches/laguna-s-2.1-xpu-b70/vllm-laguna-shared-elementwise-m12-1a7f61fef-20260731.bundle && git bundle verify patches/laguna-s-2.1-xpu-b70/vllm-xpu-kernels-laguna-shared-elementwise-m12-99886d783-20260731.bundle",
    "launch": "repro/laguna-s-2.1-int4-b70-125tps-20260731/run-record-gate.sh",
    "health": "curl -fsS http://127.0.0.1:18080/health",
    "benchmark": "The formal gate sends exactly one cold 13-prompt suite and writes the timestamped run directory.",
    "stop": "The formal gate performs clean teardown; verify no vLLM worker or listener on port 18080 remains."
   },
   "dependencies": [
    "repro/laguna-s-2.1-int4-b70-125tps-20260731/README.md",
    "data/laguna-s-2.1-width12-dflash-fp8-record-20260726.json",
    "data/localmaxxing-laguna-s-2.1-int4-b70-width12-dflash-fp8-102.971tok-20260726.queue.json",
    "data/localmaxxing-responses/laguna-s-2.1-int4-b70-width12-dflash-fp8-102.971tok-20260726.response.json",
    "experiments/laguna-s-2.1-xpu-b70/realistic-suite-v1.json",
    "experiments/laguna-s-2.1-xpu-b70/tools/capture_laguna_m8_idle_snapshot.py",
    "experiments/laguna-s-2.1-xpu-b70/tools/compare_exact_runs.py",
    "experiments/laguna-s-2.1-xpu-b70/tools/laguna_nvme_paths.sh",
    "experiments/laguna-s-2.1-xpu-b70/tools/run_laguna_dflash_segmented_smoke.py",
    "experiments/laguna-s-2.1-xpu-b70/tools/run_laguna_mwide_measurement_leg.sh",
    "experiments/laguna-s-2.1-xpu-b70/tools/run_laguna_replemb_measurement_leg.sh",
    "experiments/laguna-s-2.1-xpu-b70/tools/runtime-lock-shared-elementwise-m12.json",
    "experiments/laguna-s-2.1-xpu-b70/tools/serve_laguna_mwide_graph_nvme.sh",
    "repro/gemma4-26b-a4b-q8-b70/realistic-suite-v1.json",
    "repro/laguna-s-2.1-int4-b70-102tps-20260726/evidence/record-run.sha256",
    "repro/laguna-s-2.1-int4-b70-102tps-20260726/manifests/runtime-lock.json",
    "repro/laguna-s-2.1-int4-b70-102tps-20260726/teacher-text-sha256-v1.json",
    "repro/laguna-s-2.1-int4-b70-102tps-20260726/teacher-token-oracle-v1.json",
    "scripts/bench-openai-realistic-suite.py",
    "scripts/qualify_realistic_window_metrics.py",
    "repro/laguna-s-2.1-int4-b70-125tps-20260731/run-record-gate.sh",
    "repro/laguna-s-2.1-int4-b70-102tps-20260726/README.md",
    "repro/laguna-s-2.1-int4-b70-102tps-20260726/restore-models.sh",
    "repro/laguna-s-2.1-int4-b70-102tps-20260726/restore-sources.sh",
    "repro/laguna-s-2.1-int4-b70-102tps-20260726/manifests/model-release-files.sha256",
    "patches/laguna-s-2.1-xpu-b70/README.md",
    "results/laguna-s-2.1-int4-b70/README.md",
    "data/laguna-shared-elementwise-m12-record-20260731.json",
    "patches/laguna-s-2.1-xpu-b70/vllm-laguna-shared-elementwise-m12-1a7f61fef-20260731.bundle",
    "patches/laguna-s-2.1-xpu-b70/vllm-xpu-kernels-laguna-shared-elementwise-m12-99886d783-20260731.bundle",
    "repro/laguna-s-2.1-int4-b70-102tps-20260726/verify-record.sh",
    "repro/laguna-s-2.1-int4-b70-125tps-20260731/teacher-q1-canonical-bench.json"
   ],
   "missing": [
    "portable runtime rebuild",
    "tested platform installation",
    "record-specific model acquisition helper",
    "non-originating-host replay",
    "beginner recovery flow",
    "decode, prefill, and TTFT context sweep"
   ]
  },
  {
   "manifest": "packages/lfm25-26b-q8-b70/package.json",
   "format": "b70-model-package-v1",
   "id": "lfm25-26b-q8-b70",
   "name": "LFM2.5 2.6B Q8_0 on one Intel Arc Pro B70",
   "status": "candidate",
   "audience": "beginner",
   "guide": "repro/lfm25-26b-q8-b70/README.md",
   "clean_host_tested": false,
   "library": {
    "model_family": "LFM2.5",
    "publisher": "Liquid AI",
    "variant": "2.6B",
    "summary": "Liquid AI's compact LFM2.5 2.6B on one Arc Pro B70 in 8-bit form with stock, unpatched llama.cpp - the simplest single-command recipe on the site.",
    "quantization": "Q8_0",
    "runtime_label": "llama.cpp SYCL",
    "operating_systems": [
     "Linux"
    ],
    "delivery": [
     "native"
    ],
    "modalities": [
     "text"
    ],
    "use_cases": [
     "general",
     "chat",
     "starter"
    ],
    "tags": [
     "one card",
     "target only",
     "cache zero",
     "stock upstream",
     "novice"
    ],
    "published_at": "2026-08-27",
    "featured_metric": {
     "value": 132.13745667864433,
     "unit": "tok/s",
     "label": "strict class-balanced decode",
     "scope": "Median of two fresh-server class-balanced medians over the complete 12-prompt/six-class, 512-cap, cache-zero native HTTP suite; one B70, TP1, MTP0, reasoning off, F16 KV, 8K configured context. Both objective-canary batteries passed and complete token arrays matched 12/12 across servers.",
     "evidence": "data/2026-08-27-lfm25-q8-tp1-strict-headline-result.json"
    },
    "benchmark_status": "Strict single-user headline qualified on 2026-08-27. The older 132.351606/132.467576 observations remain historical only because their raw artifacts were not retained."
   },
   "contributors": [
    {
     "id": "b70-optimization-lab",
     "name": "neural.download lab",
     "kind": "lab",
     "profile": "https://github.com/steveseguin/b70-optimization-lab",
     "contribution": "Intake verification, bring-up, baselines, depth sweep, operating point, canary battery, package.",
     "status": "integrated",
     "validated_effect": "Stock-upstream characterization; no lab patches in this package. The strict paired single-user result is 132.137457 tok/s.",
     "evidence": "data/2026-08-27-lfm25-q8-tp1-strict-headline-result.json"
    }
   ],
   "performance_profiles": [
    {
     "id": "decode-vs-context-depth",
     "label": "Raw decode over existing context depth",
     "metric": "decode",
     "unit": "tok/s",
     "x_label": "Existing context depth before tg128",
     "scope": "llama-bench raw engine rates (pp2048/tg128, fa on, 5 reps); see depth-sweep.svg and sweep JSON in the guide directory. The directly measured zero-depth point remains in the linked raw evidence; no missing depth is interpolated.",
     "evidence": "repro/lfm25-26b-q8-b70/lfm25-26b-q8.sweep.json",
     "points": [
      {
       "context_tokens": 2048,
       "value": 131.259551,
       "samples": 5
      },
      {
       "context_tokens": 4096,
       "value": 127.211224,
       "samples": 5
      },
      {
       "context_tokens": 8192,
       "value": 120.195857,
       "samples": 5
      },
      {
       "context_tokens": 16384,
       "value": 107.980065,
       "samples": 5
      },
      {
       "context_tokens": 24576,
       "value": 98.138413,
       "samples": 5
      },
      {
       "context_tokens": 32768,
       "value": 89.93812,
       "samples": 5
      }
     ]
    },
    {
     "id": "prefill-vs-context-depth",
     "label": "Raw pp2048 over existing context depth",
     "metric": "prefill",
     "unit": "tok/s",
     "x_label": "Existing context depth before pp2048",
     "scope": "llama-bench raw engine rates (pp2048/tg128, fa on, 5 reps); see depth-sweep.svg and sweep JSON in the guide directory. The directly measured zero-depth point remains in the linked raw evidence; no missing depth is interpolated.",
     "evidence": "repro/lfm25-26b-q8-b70/lfm25-26b-q8.sweep.json",
     "points": [
      {
       "context_tokens": 2048,
       "value": 4699.373204,
       "samples": 5
      },
      {
       "context_tokens": 4096,
       "value": 4584.945629,
       "samples": 5
      },
      {
       "context_tokens": 8192,
       "value": 4399.719712,
       "samples": 5
      },
      {
       "context_tokens": 16384,
       "value": 3752.42491,
       "samples": 5
      },
      {
       "context_tokens": 24576,
       "value": 3698.15006,
       "samples": 5
      },
      {
       "context_tokens": 32768,
       "value": 2824.910192,
       "samples": 5
      }
     ]
    }
   ],
   "known_limitations": [
    "Emits untagged reasoning prose before final answers regardless of the reasoning flag; answers correct but verbose.",
    "The quality gate establishes objective correctness and exact fresh-server repeatability for this identity; it is not a cross-model capability comparison.",
    "Clean-host beginner flow not yet executed."
   ],
   "hardware": {
    "accelerator": "Intel Arc Pro B70 32 GiB",
    "cards": 1,
    "host_ram_min_bytes": 34359738368,
    "host_ram_plus_swap_min_bytes": 34359738368,
    "model_weight_bytes": 2875312096
   },
   "model": {
    "repository": "LiquidAI/LFM2.5-2.6B-GGUF",
    "revision": "f4a289c8a200a5ca71005ba7abc2dad33058a450",
    "manifest": "repro/lfm25-26b-q8-b70/model-manifest.json"
   },
   "runtime": {
    "kind": "native",
    "project": "ggml-org/llama.cpp",
    "revision": "9fee29e9435f865ec0b811a783a6471a136d9317",
    "build": "cmake -G Ninja -B build-sycl-aot-bmg-g31 -DCMAKE_BUILD_TYPE=Release -DCMAKE_CXX_COMPILER=icpx -DCMAKE_C_COMPILER=icx -DGGML_SYCL=ON -DGGML_SYCL_F16=ON -DGGML_SYCL_DEVICE_ARCH=bmg_g31 -DGGML_SYCL_MAX_PARALLEL_LINK_JOBS=32 -DLLAMA_CURL=OFF && cmake --build build-sycl-aot-bmg-g31 --target llama-server llama-bench -j 24"
   },
   "project_patches": {
    "required": false,
    "items": []
   },
   "commands": {
    "preflight": "python3 scripts/verify-neural-download-model.py repro/lfm25-26b-q8-b70/model-manifest.json \"$MODEL_DIR\" && source /opt/intel/oneapi/setvars.sh --force && export ONEAPI_DEVICE_SELECTOR=level_zero:0",
    "launch": "$BUILD_DIR/bin/llama-server --model $MODEL_DIR/LFM2.5-2.6B-Q8_0.gguf --alias lfm25 --reasoning off --ctx-size 8192 --cache-type-k f16 --cache-type-v f16 --device SYCL0 --gpu-layers 99 --split-mode none --flash-attn auto --parallel 1 --cache-ram 0 --ctx-checkpoints 0 --no-cache-prompt --slot-prompt-similarity 0 --fit off --metrics --no-webui --host 127.0.0.1 --port 18100",
    "health": "curl -fsS http://127.0.0.1:18100/health",
    "benchmark": "python3 scripts/bench-openai-realistic-suite.py --base-url http://127.0.0.1:18100 --model lfm25 --api-mode native-raw --suite repro/qwen36-27b-autoround-int4-b70/realistic-suite-v1.json --max-tokens 512 --metric-tokens 100 --seed 42 --timeout 900 --return-token-ids --require-natural-eos --request-extra-json '{\"cache_prompt\":false,\"seed\":42,\"temperature\":0,\"top_p\":1}' --out performance.json && python3 scripts/neural-download-canaries.py --base-url http://127.0.0.1:18100 --model lfm25 --out canaries.json",
    "stop": "pkill -x llama-server"
   },
   "dependencies": [
    "repro/lfm25-26b-q8-b70/README.md",
    "repro/gemma4-26b-a4b-q8-b70/realistic-suite-v1.json",
    "repro/lfm25-26b-q8-b70/model-manifest.json",
    "repro/lfm25-26b-q8-b70/lfm25-26b-q8.sweep.json",
    "repro/lfm25-26b-q8-b70/lfm25-26b-q8.meta.json",
    "repro/lfm25-26b-q8-b70/depth-sweep.svg",
    "data/2026-08-27-neural-download-stock-headline-closure-prereg.json",
    "data/2026-08-27-lfm25-q8-tp1-strict-headline-result.json",
    "data/2026-08-27-lfm25-q8-tp1-strict-headline-comparison.json",
    "data/neural-download-stock-headline-lfm25-20260827-r1a/qualification.json",
    "data/neural-download-stock-headline-lfm25-20260827-r1b/qualification.json",
    "scripts/run-neural-download-stock-headline-attempt.sh",
    "data/neural-download-audit-host-depth-sweeps-20260822/README.md",
    "scripts/verify-neural-download-model.py",
    "docs/neural-download-packet-standard.md",
    "scripts/bench-openai-realistic-suite.py",
    "repro/qwen36-27b-autoround-int4-b70/realistic-suite-v1.json",
    "scripts/neural-download-canaries.py"
   ],
   "missing": [
    "tested clean-host platform installation",
    "output-qualified HTTP concurrency",
    "beginner recovery flow"
   ]
  },
  {
   "manifest": "packages/minimax-m27-int4-autoround-b70/package.json",
   "format": "b70-model-package-v1",
   "id": "minimax-m27-b70-89tps-20260520",
   "name": "MiniMax M2.7 AutoRound INT4 on four Intel Arc Pro B70 cards",
   "status": "candidate",
   "audience": "expert",
   "guide": "repro/minimax-m27-b70-89tps-20260520/README.md",
   "clean_host_tested": false,
   "library": {
    "model_family": "MiniMax M2.7",
    "publisher": "MiniMax AI",
    "variant": "AutoRound W4A16 INT4",
    "summary": "MiniMax M2.7, a 229B mixture-of-experts for long conversations, on four Arc Pro B70 cards in 4-bit form with vLLM. Quality-checked token by token.",
    "quantization": "AutoRound W4A16 INT4",
    "runtime_label": "vLLM XPU + llm-scaler",
    "operating_systems": [
     "Linux"
    ],
    "delivery": [
     "native"
    ],
    "modalities": [
     "text"
    ],
    "use_cases": [
     "general",
     "coding",
     "long generation"
    ],
    "tags": [
     "four cards",
     "TP4",
     "AutoRound",
     "2K benchmark",
     "quality gated"
    ],
    "published_at": "2026-08-22",
    "featured_metric": {
     "value": 89.31419538094708,
     "unit": "tok/s",
     "label": "mean output throughput",
     "scope": "Mean output throughput across four promoted warm p512/n1536, batch-one, 2K-context runs after the strict quality gate.",
     "evidence": "repro/minimax-m27-b70-89tps-20260520/results/promoted-result-20260519.json"
    }
   },
   "contributors": [
    {
     "id": "b70-optimization-lab",
     "name": "neural.download lab",
     "kind": "lab",
     "profile": "https://github.com/steveseguin/b70-optimization-lab",
     "contribution": "B70/XPU integration, MiniMax MoE work-sharing and custom-op optimizations, graph/runtime fixes, exact-token quality gates, benchmarking, and packaging.",
     "status": "integrated",
     "validated_effect": "The promoted four-run mean was 89.314195 output tok/s, 0.4343% above the preceding promoted result, with exact-token and semantic gates passing.",
     "evidence": "repro/minimax-m27-b70-89tps-20260520/results/promoted-result-20260519.json"
    },
    {
     "id": "lasimeri",
     "name": "Lasimeri",
     "kind": "external",
     "profile": "https://huggingface.co/Lasimeri",
     "contribution": "Published the MiniMax M2.7 AutoRound W4A16 INT4 checkpoint used by this lane.",
     "status": "acknowledged",
     "validated_effect": "The checkpoint is the model dependency used in the validated lab result; no separate quantization-vs-base speed or quality uplift is assigned here.",
     "evidence": "repro/minimax-m27-b70-89tps-20260520/manifests/model-pin.json"
    }
   ],
   "hardware": {
    "accelerator": "Intel Arc Pro B70 32 GiB",
    "cards": 4,
    "host_ram_min_bytes": 17179869184,
    "host_ram_plus_swap_min_bytes": 85899345920
   },
   "model": {
    "repository": "Lasimeri/MiniMax-M2.7-int4-AutoRound",
    "revision": "1afac074ecf7c3c4504c68b83d127506f8a7e5a4",
    "manifest": "repro/minimax-m27-b70-89tps-20260520/manifests/model-pin.json"
   },
   "runtime": {
    "kind": "native",
    "repository": "vllm-project/vllm",
    "revision": "c51df43005726a09c6eb7348e8c1b00501c70a8e",
    "secondary_repository": "intel/llm-scaler@4bfc0070090cc54afdb2d46b8e57882359141568",
    "build": "repro/minimax-m27-b70-89tps-20260520/scripts/02-build-stack.sh"
   },
   "project_patches": {
    "required": true,
    "items": [
     "repro/minimax-m27-b70-89tps-20260520/patches/vllm-active-promoted-minimax-89tps-20260520.patch.gz.b64",
     "repro/minimax-m27-b70-89tps-20260520/patches/llm-scaler-active-promoted-minimax-89tps-20260520.patch.gz.b64"
    ],
    "reason": "The result depends on the retained broad vLLM and llm-scaler active-source snapshots."
   },
   "commands": {
    "preflight": "bash repro/minimax-m27-b70-89tps-20260520/scripts/03-verify-runtime.sh",
    "launch": "bash repro/minimax-m27-b70-89tps-20260520/scripts/04-run-quality-gate.sh",
    "health": "bash repro/minimax-m27-b70-89tps-20260520/scripts/03-verify-runtime.sh",
    "benchmark": "bash repro/minimax-m27-b70-89tps-20260520/scripts/05-run-benchmark.sh",
    "stop": "After either runner exits, verify no vLLM worker remains with pgrep -af 'vllm|multiprocessing.spawn'."
   },
   "dependencies": [
    "repro/minimax-m27-b70-89tps-20260520/README.md",
    "scripts/bench-vllm-minimax-autoround-xpu.sh",
    "scripts/inspect-minimax-aot-boundary-context.py",
    "scripts/inspect-vllm-runtime.py",
    "scripts/run-minimax-strict-quality-gated-candidate.sh",
    "scripts/run-vllm-minimax-quality-check.py",
    "repro/minimax-m27-b70-89tps-20260520/scripts/00-install-system-deps.sh",
    "repro/minimax-m27-b70-89tps-20260520/scripts/01-download-model.sh",
    "repro/minimax-m27-b70-89tps-20260520/scripts/02-build-stack.sh",
    "repro/minimax-m27-b70-89tps-20260520/scripts/03-verify-runtime.sh",
    "repro/minimax-m27-b70-89tps-20260520/scripts/04-run-quality-gate.sh",
    "repro/minimax-m27-b70-89tps-20260520/scripts/05-run-benchmark.sh",
    "repro/minimax-m27-b70-89tps-20260520/manifests/model-pin.json",
    "repro/minimax-m27-b70-89tps-20260520/results/promoted-result-20260519.json",
    "data/localmaxxing-minimax-m27-prod-c1-systemd-near32k-20260526.payload.json",
    "data/localmaxxing-responses/minimax-m27-prod-c1-systemd-near32k-20260526.response.json",
    "docs/minimax-production-c1-service.md",
    "results/minimax-m27-int4-autoround-b70/README.md",
    "repro/minimax-m27-b70-89tps-20260520/patches/vllm-active-promoted-minimax-89tps-20260520.patch.gz.b64",
    "repro/minimax-m27-b70-89tps-20260520/patches/llm-scaler-active-promoted-minimax-89tps-20260520.patch.gz.b64"
   ],
   "missing": [
    "current clean-host replay",
    "historical run-time full model payload manifest",
    "beginner recovery and platform compatibility boundary",
    "persistent OpenAI-compatible service wrapper for this exact 89 tok/s lane",
    "decode, prefill, and TTFT context sweep"
   ]
  },
  {
   "manifest": "packages/muse-glimmer-30b-q8-woq-b70/package.json",
   "format": "b70-model-package-v1",
   "id": "muse-glimmer-30b-q8-woq-b70-100tps-20260813",
   "name": "Muse-Glimmer 30B Q8/WOQ with DFlash on four Intel Arc Pro B70 cards",
   "status": "candidate",
   "audience": "expert",
   "guide": "repro/muse-glimmer-30b-q8-woq-b70-100tps-20260813/README.md",
   "clean_host_tested": false,
   "library": {
    "model_family": "Muse-Glimmer",
    "publisher": "Meta Models",
    "variant": "30B UD-Q8_K_XL target plus BF16 DFlash assistant",
    "summary": "Meta's Muse-Glimmer 30B, a model that reads images as well as text, on four Arc Pro B70 cards in 8-bit form with a draft model speeding up generation.",
    "quantization": "UD-Q8_K_XL / BF16 draft",
    "runtime_label": "llama.cpp SYCL",
    "operating_systems": [
     "Linux"
    ],
    "delivery": [
     "native"
    ],
    "modalities": [
     "text"
    ],
    "use_cases": [
     "general",
     "coding",
     "structured output"
    ],
    "tags": [
     "four cards",
     "TP4",
     "DFlash",
     "oneDNN WOQ",
     "target verified"
    ],
    "published_at": "2026-08-22",
    "featured_metric": {
     "value": 100.3685,
     "unit": "tok/s",
     "label": "pooled canonical mean",
     "scope": "Pooled arithmetic mean across two fresh canonical full-256 three-prompt runs; individual prompts and full natural completions can be slower.",
     "evidence": "repro/muse-glimmer-30b-q8-woq-b70-100tps-20260813/manifests/expected-result.json"
    }
   },
   "contributors": [
    {
     "id": "b70-optimization-lab",
     "name": "neural.download lab",
     "kind": "lab",
     "profile": "https://github.com/steveseguin/b70-optimization-lab",
     "contribution": "B70/SYCL bring-up, fixed-width oneDNN Q8 WOQ, distributed argmax/local-winner reuse, DFlash integration, quality gates, and reproduction packaging.",
     "status": "integrated",
     "validated_effect": "Two independent canonical runs measured 100.088 and 100.649 tok/s; the cold 15-prompt first-100 median was 161.900 tok/s with the documented target-verification boundary.",
     "evidence": "results/muse-glimmer-30b-q8-woq-b70/README.md"
    }
   ],
   "hardware": {
    "accelerator": "Intel Arc Pro B70 32 GiB",
    "cards": 4,
    "model_weight_bytes": 37425857088
   },
   "model": {
    "repository": "unsloth/Muse-Glimmer-30B-GGUF",
    "revision": "faa5b025c584459c13febfa5c59883516710ae39",
    "manifest": "repro/muse-glimmer-30b-q8-woq-b70-100tps-20260813/manifests/models.json"
   },
   "runtime": {
    "kind": "native",
    "repository": "ggml-org/llama.cpp",
    "revision": "030ebb558a5820b444a8f836ed5cdd46c9b4bd7a",
    "build": "repro/muse-glimmer-30b-q8-woq-b70-100tps-20260813/scripts/build.sh"
   },
   "project_patches": {
    "required": true,
    "items": [
     "patches/muse-glimmer-30b-b70/llama.cpp-030ebb558-to-q8-woq-century-20260813.patch"
    ],
    "reason": "The record path depends on the lab's complete base-to-record llama.cpp/SYCL patch."
   },
   "commands": {
    "preflight": "python3 repro/muse-glimmer-30b-q8-woq-b70-100tps-20260813/scripts/verify-evidence.py && (cd repro/muse-glimmer-30b-q8-woq-b70-100tps-20260813 && sha256sum -c SHA256SUMS)",
    "launch": "MUSE_TARGET_MODEL=/path/to/Muse-Glimmer-30B-UD-Q8_K_XL.gguf MUSE_DRAFT_MODEL=/path/to/dflash-bf16.gguf LLAMA_CPP_ROOT=/path/to/llama.cpp-muse repro/muse-glimmer-30b-q8-woq-b70-100tps-20260813/scripts/run-realistic-suite.sh /new/results/muse-realistic",
    "health": "curl -fsS http://127.0.0.1:19494/health",
    "benchmark": "Inspect /new/results/muse-realistic/qualification.txt and bootstrap.json after the combined fail-closed runner exits.",
    "stop": "The runner stops its server automatically; on interruption, verify with pgrep -af llama-server before another run."
   },
   "dependencies": [
    "repro/muse-glimmer-30b-q8-woq-b70-100tps-20260813/README.md",
    "repro/gemma4-26b-a4b-q8-b70/realistic-suite-v1.json",
    "scripts/bench-openai-realistic-suite.py",
    "repro/muse-glimmer-30b-q8-woq-b70-100tps-20260813/scripts/download-models.sh",
    "repro/muse-glimmer-30b-q8-woq-b70-100tps-20260813/scripts/restore-source.sh",
    "repro/muse-glimmer-30b-q8-woq-b70-100tps-20260813/scripts/build.sh",
    "repro/muse-glimmer-30b-q8-woq-b70-100tps-20260813/scripts/run-realistic-suite.sh",
    "repro/muse-glimmer-30b-q8-woq-b70-100tps-20260813/manifests/models.json",
    "repro/muse-glimmer-30b-q8-woq-b70-100tps-20260813/manifests/source.json",
    "repro/muse-glimmer-30b-q8-woq-b70-100tps-20260813/manifests/expected-result.json",
    "patches/muse-glimmer-30b-b70/README.md",
    "results/muse-glimmer-30b-q8-woq-b70/README.md",
    "patches/muse-glimmer-30b-b70/llama.cpp-030ebb558-to-q8-woq-century-20260813.patch",
    "repro/muse-glimmer-30b-q8-woq-b70-100tps-20260813/scripts/verify-evidence.py",
    "repro/muse-glimmer-30b-q8-woq-b70-100tps-20260813"
   ],
   "missing": [
    "tested platform installer",
    "original download-time complete draft input identity",
    "independent clean-host replay",
    "beginner recovery flow",
    "decode, prefill, and TTFT context sweep"
   ]
  },
  {
   "manifest": "packages/nemotron-35-lightning-30b-a3b-b70/package.json",
   "format": "b70-model-package-v1",
   "id": "nemotron-35-lightning-30b-a3b-b70",
   "name": "Nemotron 3.5 Lightning 30B-A3B UD-Q4_K_M on one Intel Arc Pro B70",
   "status": "candidate",
   "audience": "intermediate",
   "guide": "repro/nemotron-35-lightning-30b-a3b-b70/README.md",
   "clean_host_tested": false,
   "library": {
    "model_family": "Nemotron 3.5",
    "publisher": "NVIDIA (GGUF by unsloth)",
    "variant": "Lightning 30B-A3B",
    "summary": "NVIDIA's Nemotron 3.5 Lightning, a 30B hybrid Mamba mixture-of-experts, on one Arc Pro B70 in 4-bit form with stock llama.cpp. Its speed barely drops as the conversation grows.",
    "quantization": "UD-Q4_K_M",
    "runtime_label": "llama.cpp SYCL",
    "operating_systems": [
     "Linux"
    ],
    "delivery": [
     "native"
    ],
    "modalities": [
     "text"
    ],
    "use_cases": [
     "general",
     "long context",
     "chat"
    ],
    "tags": [
     "one card",
     "moe",
     "hybrid mamba",
     "target only",
     "cache zero",
     "stock upstream"
    ],
    "published_at": "2026-08-22",
    "featured_metric": null,
    "benchmark_status": "The 72.169452/72.035976 varied-suite observations and reasoning-off canary summary are preserved in the guide, but their raw operating-point and canary JSON files are not closed in this repository. Import and hash-bind those files, then replay the quality/determinism gate before assigning a strict headline."
   },
   "contributors": [
    {
     "id": "b70-optimization-lab",
     "name": "neural.download lab",
     "kind": "lab",
     "profile": "https://github.com/steveseguin/b70-optimization-lab",
     "contribution": "Intake verification, bring-up, baselines, depth sweep, operating point, canary battery, package.",
     "status": "integrated",
     "validated_effect": "Stock-upstream characterization; no lab patches in this package.",
     "evidence": "experiments/qwen38-27b-b70/data/2026-08-22-neural-download-firstwave-baselines.json"
    }
   ],
   "performance_profiles": [
    {
     "id": "decode-vs-context-depth",
     "label": "Raw decode over existing context depth",
     "metric": "decode",
     "unit": "tok/s",
     "x_label": "Existing context depth before tg128",
     "scope": "llama-bench raw engine rates (pp2048/tg128, fa on, 5 reps); only -11% decode from 0 to 32K depth. The directly measured zero-depth point remains in the linked raw evidence; no missing depth is interpolated.",
     "evidence": "repro/nemotron-35-lightning-30b-a3b-b70/nemotron-35-lightning.sweep.json",
     "points": [
      {
       "context_tokens": 2048,
       "value": 72.420657,
       "samples": 5
      },
      {
       "context_tokens": 4096,
       "value": 71.846756,
       "samples": 5
      },
      {
       "context_tokens": 8192,
       "value": 70.6952,
       "samples": 5
      },
      {
       "context_tokens": 16384,
       "value": 68.557408,
       "samples": 5
      },
      {
       "context_tokens": 24576,
       "value": 66.552475,
       "samples": 5
      },
      {
       "context_tokens": 32768,
       "value": 64.622975,
       "samples": 5
      }
     ]
    },
    {
     "id": "prefill-vs-context-depth",
     "label": "Raw pp2048 over existing context depth",
     "metric": "prefill",
     "unit": "tok/s",
     "x_label": "Existing context depth before pp2048",
     "scope": "llama-bench raw engine rates (pp2048/tg128, fa on, 5 reps); only -11% decode from 0 to 32K depth. The directly measured zero-depth point remains in the linked raw evidence; no missing depth is interpolated.",
     "evidence": "repro/nemotron-35-lightning-30b-a3b-b70/nemotron-35-lightning.sweep.json",
     "points": [
      {
       "context_tokens": 2048,
       "value": 1166.922303,
       "samples": 5
      },
      {
       "context_tokens": 4096,
       "value": 1153.135384,
       "samples": 5
      },
      {
       "context_tokens": 8192,
       "value": 1145.645326,
       "samples": 5
      },
      {
       "context_tokens": 16384,
       "value": 1106.136083,
       "samples": 5
      },
      {
       "context_tokens": 24576,
       "value": 1102.582316,
       "samples": 5
      },
      {
       "context_tokens": 32768,
       "value": 1018.926677,
       "samples": 5
      }
     ]
    }
   ],
   "known_limitations": [
    "With reasoning ON, repeated identical requests are not hash-stable (thinking-channel sampling); deterministic with reasoning off - the recipe default.",
    "MoE rate reflects ~3B active parameters.",
    "1M native context not exercised beyond 32K here.",
    "Two-card layer split measured 69.45/69.49 tok/s vs 72.17/72.04 on one card (-3.7%): use one card for single-stream decode."
   ],
   "hardware": {
    "accelerator": "Intel Arc Pro B70 32 GiB",
    "cards": 1,
    "host_ram_min_bytes": 34359738368,
    "host_ram_plus_swap_min_bytes": 34359738368,
    "model_weight_bytes": 25272338336
   },
   "model": {
    "repository": "unsloth/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-GGUF",
    "revision": "f2d3fe3694501008786e81e5f20360cbf715496a",
    "manifest": "repro/nemotron-35-lightning-30b-a3b-b70/model-manifest.json"
   },
   "runtime": {
    "kind": "native",
    "project": "ggml-org/llama.cpp",
    "revision": "9fee29e9435f865ec0b811a783a6471a136d9317",
    "build": "cmake -G Ninja -B build-sycl-aot-bmg-g31 -DCMAKE_BUILD_TYPE=Release -DCMAKE_CXX_COMPILER=icpx -DCMAKE_C_COMPILER=icx -DGGML_SYCL=ON -DGGML_SYCL_F16=ON -DGGML_SYCL_DEVICE_ARCH=bmg_g31 -DGGML_SYCL_MAX_PARALLEL_LINK_JOBS=32 -DLLAMA_CURL=OFF && cmake --build build-sycl-aot-bmg-g31 --target llama-server llama-bench -j 24"
   },
   "project_patches": {
    "required": false,
    "items": []
   },
   "commands": {
    "preflight": "python3 scripts/verify-neural-download-model.py repro/nemotron-35-lightning-30b-a3b-b70/model-manifest.json \"$MODEL_DIR\" && source /opt/intel/oneapi/setvars.sh --force && export ONEAPI_DEVICE_SELECTOR=level_zero:0",
    "launch": "BUILD_DIR/bin/llama-server --model $MODEL_DIR/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-UD-Q4_K_M.gguf --alias nemotron --reasoning off --ctx-size 8192 --cache-type-k f16 --cache-type-v f16 --device SYCL0 --gpu-layers 99 --flash-attn auto --parallel 1 --cache-ram 0 --ctx-checkpoints 0 --fit off --metrics --no-webui --host 127.0.0.1 --port 18100",
    "health": "curl -fsS http://127.0.0.1:18100/health",
    "benchmark": "python3 scripts/bench-openai-realistic-suite.py --base-url http://127.0.0.1:18100 --model nemotron --api-mode completions --suite repro/qwen36-27b-autoround-int4-b70/realistic-suite-v1.json --max-tokens 512 --metric-tokens 100 --seed 1 --request-extra-json '{\"cache_prompt\":false,\"seed\":42,\"temperature\":0}' --out bench.json",
    "stop": "pkill -x llama-server"
   },
   "dependencies": [
    "repro/nemotron-35-lightning-30b-a3b-b70/README.md",
    "repro/gemma4-26b-a4b-q8-b70/realistic-suite-v1.json",
    "repro/nemotron-35-lightning-30b-a3b-b70/model-manifest.json",
    "repro/nemotron-35-lightning-30b-a3b-b70/nemotron-35-lightning.sweep.json",
    "repro/nemotron-35-lightning-30b-a3b-b70/nemotron-35-lightning.meta.json",
    "repro/nemotron-35-lightning-30b-a3b-b70/depth-sweep.svg",
    "scripts/verify-neural-download-model.py",
    "docs/neural-download-packet-standard.md",
    "experiments/qwen38-27b-b70/data/2026-08-22-neural-download-firstwave-baselines.json",
    "scripts/bench-openai-realistic-suite.py",
    "repro/qwen36-27b-autoround-int4-b70/realistic-suite-v1.json"
   ],
   "missing": [
    "tested clean-host platform installation",
    "context beyond 32K of the 1M native window"
   ]
  },
  {
   "manifest": "packages/ornith-15-35b-a3b-q4km-b70/package.json",
   "format": "b70-model-package-v1",
   "id": "ornith-15-35b-a3b-q4km-b70",
   "name": "Ornith 1.5 35B-A3B Q4_K_M on one Intel Arc Pro B70",
   "status": "candidate",
   "audience": "intermediate",
   "guide": "repro/ornith-15-35b-a3b-q4km-b70/README.md",
   "clean_host_tested": false,
   "library": {
    "model_family": "Ornith 1.5",
    "publisher": "Ornith AI",
    "variant": "35B-A3B",
    "summary": "Ornith AI's 35B mixture-of-experts (about 3B active per word) on one Arc Pro B70 in 4-bit form - the fastest single-card result this lab has measured, using the lab's tuned kernel stack.",
    "quantization": "Q4_K_M",
    "runtime_label": "llama.cpp SYCL",
    "operating_systems": [
     "Linux"
    ],
    "delivery": [
     "native"
    ],
    "modalities": [
     "text"
    ],
    "use_cases": [
     "general",
     "coding",
     "chat"
    ],
    "tags": [
     "one card",
     "moe",
     "target only",
     "cache zero",
     "lab optimized"
    ],
    "published_at": "2026-08-22",
    "featured_metric": null,
    "benchmark_status": "The 131.460231 tok/s two-server observation and matched patch A/B evidence remain valid scoped measurements, but the runtime produced 0/12 identical complete natural-response hashes across fresh stock servers. Keep the mechanism and context evidence; require a stable, registered cross-server oracle before assigning a strict package headline."
   },
   "contributors": [
    {
     "id": "b70-optimization-lab",
     "name": "neural.download lab",
     "kind": "lab",
     "profile": "https://github.com/steveseguin/b70-optimization-lab",
     "contribution": "Intake verification, bring-up, baselines, depth sweeps, ordered MoE add-reduction, recurrent convolution-SiLU, residual-RMSNorm, recurrent concat/state, direct gather/state, recurrent alpha-gate, routed-expert gate/up, MoE shared-branch residual/RMSNorm, GDN RMSNorm/SiLU-gate, in-place GDN state I/O, full-attention Q/K normalization-IMRoPE, shared-expert gate/residual/RMS patches, Ornith-specific Level Zero runtime screening, operating point, canary battery, and package.",
     "status": "integrated",
     "validated_effect": "+4.85% matched fresh-server decode from ordered expert reduction, +2.10% incremental from recurrent convolution-SiLU, +1.37% incremental from residual-RMSNorm, +2.74% incremental from recurrent concat/state, +1.12% incremental from direct recurrent gather, +2.04% incremental from recurrent alpha-gate, +2.33% incremental from routed gate/up, +1.41% incremental from MoE shared-branch residual/RMSNorm, +0.78% incremental from GDN RMSNorm/SiLU-gate, +6.80% incremental from in-place GDN state I/O, +1.87% incremental from full-attention Q/K normalization-IMRoPE, +1.09% incremental serving from disabling Level Zero copy offload, then +1.38% matched conventional mean-of-run-medians from the shared-gate/residual/RMS fusion (129.676 to 131.460 tok/s; 12/12 prompt-paired wins). The legacy compatibility means 130.986 to 132.788 remain retained separately.",
     "evidence": "experiments/ornith-15-b70/data/2026-08-23-ornith35b-shared-gate-residual-rms-summary.json"
    }
   ],
   "performance_profiles": [
    {
     "id": "decode-vs-context-depth",
     "label": "Current twelve-feature raw decode over existing context depth",
     "metric": "decode",
     "unit": "tok/s",
     "x_label": "Existing context depth before tg128",
     "scope": "llama-bench raw engine rates using the exact twelve-feature source patch and accepted copy-offload setting, graph off, pp2048/tg128, flash attention on, F16 KV, and 5 repetitions at every displayed depth. No point is scaled and no missing depth is interpolated.",
     "evidence": "repro/ornith-15-35b-a3b-q4km-b70/ornith-15-35b-a3b-q4km-twelve-feature.sweep.json",
     "points": [
      {
       "context_tokens": 0,
       "value": 141.917514,
       "samples": 5
      },
      {
       "context_tokens": 2048,
       "value": 136.848608,
       "samples": 5
      },
      {
       "context_tokens": 4096,
       "value": 133.329784,
       "samples": 5
      },
      {
       "context_tokens": 8192,
       "value": 126.829116,
       "samples": 5
      },
      {
       "context_tokens": 16384,
       "value": 116.267169,
       "samples": 5
      },
      {
       "context_tokens": 24576,
       "value": 106.967195,
       "samples": 5
      },
      {
       "context_tokens": 32768,
       "value": 99.614237,
       "samples": 5
      }
     ]
    },
    {
     "id": "prefill-vs-context-depth",
     "label": "Current twelve-feature raw pp2048 over existing context depth",
     "metric": "prefill",
     "unit": "tok/s",
     "x_label": "Existing context depth before pp2048",
     "scope": "llama-bench raw engine rates using the exact twelve-feature source patch and accepted copy-offload setting, graph off, pp2048/tg128, flash attention on, F16 KV, and 5 repetitions at every displayed depth. No point is scaled and no missing depth is interpolated.",
     "evidence": "repro/ornith-15-35b-a3b-q4km-b70/ornith-15-35b-a3b-q4km-twelve-feature.sweep.json",
     "points": [
      {
       "context_tokens": 0,
       "value": 1422.407346,
       "samples": 5
      },
      {
       "context_tokens": 2048,
       "value": 1343.876428,
       "samples": 5
      },
      {
       "context_tokens": 4096,
       "value": 1338.212192,
       "samples": 5
      },
      {
       "context_tokens": 8192,
       "value": 1301.167019,
       "samples": 5
      },
      {
       "context_tokens": 16384,
       "value": 1238.569995,
       "samples": 5
      },
      {
       "context_tokens": 24576,
       "value": 1212.927246,
       "samples": 5
      },
      {
       "context_tokens": 32768,
       "value": 1115.599876,
       "samples": 5
      }
     ]
    },
    {
     "id": "aggregate-decode-vs-concurrent-sequences",
     "label": "Accepted-stack aggregate decode over concurrent sequences",
     "metric": "aggregate_decode",
     "unit": "tok/s",
     "x_metric": "concurrent_sequences",
     "x_label": "Concurrent engine sequences",
     "scope": "Direct llama-batched-bench raw-engine continuous batching with independent pp1024 prompts, interleaved tg256 decode, c65536, flash attention on, F16 KV, one B70, and the exact accepted twelve-feature stack. This excludes HTTP, JSON, queueing, and server-scheduler overhead. Per-user values are arithmetic aggregate/users. No point is scaled, interpolated, or extrapolated.",
     "evidence": "experiments/ornith-15-b70/data/2026-08-23-ornith35b-multirow-aggregate-summary.json",
     "points": [
      {
       "concurrent_sequences": 1,
       "value": 98.024651,
       "per_user_value": 98.024651,
       "samples": 1
      },
      {
       "concurrent_sequences": 2,
       "value": 102.671738,
       "per_user_value": 51.335869,
       "samples": 1
      },
      {
       "concurrent_sequences": 4,
       "value": 118.882942,
       "per_user_value": 29.720736,
       "samples": 1
      },
      {
       "concurrent_sequences": 8,
       "value": 146.283051,
       "per_user_value": 18.285381,
       "samples": 1
      },
      {
       "concurrent_sequences": 16,
       "value": 162.503937,
       "per_user_value": 10.156496,
       "samples": 1
      },
      {
       "concurrent_sequences": 32,
       "value": 216.513077,
       "per_user_value": 6.766034,
       "samples": 1
      }
     ]
    }
   ],
   "known_limitations": [
    "MoE rate reflects ~3B active parameters; not comparable to dense 35B expectations.",
    "Canary battery is objective self-consistency only; no cross-model oracle.",
    "Stock two-card layer split measured 102.01/102.20 tok/s versus 104.84/104.81 on one card (-2.6%): use one card for single-stream decode.",
    "Fresh stock servers matched 0/12 complete realistic-suite hashes across processes on this runtime. A realistic same-process repeat also produced four hashes across eight requests; use same-frozen-binary door-off/on exactness and activation counts for patch validation, while the short exact-answer 8x canary remains a narrower check.",
    "The aggregate concurrency profile is a raw-engine continuous-batching measurement, not an HTTP serving users/sec result. The optional multi-row research patch improves the four-sequence aggregate rate by a confirmed 2.21% but is not enabled by the single-user package launch.",
    "Stage the GGUF on local or sufficiently fast direct-attached storage; optimized first-token setup rereads roughly 20.0 GB of expert tensors, making a 100 Mb/s NFS mmap unsuitable for cold benchmarking."
   ],
   "hardware": {
    "accelerator": "Intel Arc Pro B70 32 GiB",
    "cards": 1,
    "host_ram_min_bytes": 34359738368,
    "host_ram_plus_swap_min_bytes": 34359738368,
    "model_weight_bytes": 21713462848
   },
   "model": {
    "repository": "ornith-ai/Ornith-1.5-35B-A3B-GGUF",
    "revision": "fbbaed45c2f0e200276ffa51701a24d45dc7f57e",
    "manifest": "repro/ornith-15-35b-a3b-q4km-b70/model-manifest.json"
   },
   "runtime": {
    "kind": "native",
    "project": "ggml-org/llama.cpp",
    "revision": "9fee29e9435f865ec0b811a783a6471a136d9317",
    "build": "cmake -G Ninja -B build-sycl-aot-bmg-g31 -DCMAKE_BUILD_TYPE=Release -DCMAKE_CXX_COMPILER=icpx -DCMAKE_C_COMPILER=icx -DBUILD_SHARED_LIBS=ON -DGGML_NATIVE=ON -DGGML_SYCL=ON -DGGML_SYCL_F16=ON -DGGML_SYCL_GRAPH=ON -DGGML_SYCL_DNN=ON -DGGML_SYCL_HOST_MEM_FALLBACK=ON -DGGML_SYCL_DEVICE_ARCH=bmg_g31 -DGGML_SYCL_MAX_PARALLEL_LINK_JOBS=8 -DLLAMA_CURL=OFF && cmake --build build-sycl-aot-bmg-g31 --target llama-server llama-bench -j2"
   },
   "project_patches": {
    "required": true,
    "items": [
     "patches/ornith-15-35b-a3b-q4km-b70/llama-cpp-ornith15-twelve-feature-stack-shared-gate-residual-rms-20260823.patch"
    ],
    "decoded_sha256": "7b9204f8f44608fc5b1858a15498b3cf9bf52b4f02c27c0f91a1807af5b5d15d",
    "reason": "The validated one-card serving stack depends on the lab's strict ordered eight-expert FP32 reduction, Ornith-shape recurrent convolution-SiLU fusion, graph-visible residual/RMSNorm fusion, Qwen-derived recurrent concat/state fusion, direct gathered-state materialization, recurrent alpha-gate fusion, tuned routed-expert gate/up/SWIGLU fusion, MoE shared-branch residual/RMSNorm extension, exact GDN RMSNorm/SiLU-gate fusion, in-place GDN persistent-state I/O, full-attention Q/K RMSNorm-IMRoPE with direct K-cache output, and shared-expert sigmoid/broadcast integration into the residual/RMS launch."
   },
   "commands": {
    "preflight": "python3 scripts/verify-neural-download-model.py repro/ornith-15-35b-a3b-q4km-b70/model-manifest.json \"$MODEL_DIR\" && echo '7b9204f8f44608fc5b1858a15498b3cf9bf52b4f02c27c0f91a1807af5b5d15d  patches/ornith-15-35b-a3b-q4km-b70/llama-cpp-ornith15-twelve-feature-stack-shared-gate-residual-rms-20260823.patch' | sha256sum -c - && source /opt/intel/oneapi/setvars.sh --force && export ONEAPI_DEVICE_SELECTOR=level_zero:0 && export UR_L0_V2_FORCE_DISABLE_COPY_OFFLOAD=1",
    "launch": "UR_L0_V2_FORCE_DISABLE_COPY_OFFLOAD=1 GGML_SYCL_ENABLE_GRAPH=0 GGML_SYCL_FUSED_MOE_ADD_REDUCE=1 GGML_SYCL_FUSED_ORNITH_CONV_SILU=1 GGML_SYCL_FUSED_RESIDUAL_RMS_NORM=1 GGML_SYCL_FUSED_ORNITH_CONCAT_STATE=1 GGML_SYCL_FUSED_ORNITH_CONCAT_STATE_DIRECT=1 GGML_SYCL_FUSED_ORNITH_ALPHA_GATE=1 GGML_SYCL_FUSED_ORNITH_MOE_GATE_UP=1 GGML_SYCL_FUSED_ORNITH_MOE_SHARED_RESIDUAL_RMS=1 GGML_SYCL_FUSED_ORNITH_GDN_RMS_GATE=1 GGML_SYCL_FUSED_ORNITH_GDN_STATE_IO=1 GGML_SYCL_FUSED_ORNITH_QK_NORM_ROPE=1 GGML_SYCL_FUSED_ORNITH_MOE_GATE_RESIDUAL_RMS=1 BUILD_DIR/bin/llama-server --model $MODEL_DIR/Ornith-1.5-35B-Q4_K_M.gguf --alias ornith35b --reasoning off --ctx-size 8192 --cache-type-k f16 --cache-type-v f16 --device SYCL0 --gpu-layers 99 --flash-attn auto --parallel 1 --cache-ram 0 --ctx-checkpoints 0 --fit off --metrics --no-webui --host 127.0.0.1 --port 18100",
    "health": "curl -fsS http://127.0.0.1:18100/health",
    "benchmark": "python3 scripts/bench-openai-realistic-suite.py --base-url http://127.0.0.1:18100 --model ornith35b --api-mode completions --suite repro/qwen36-27b-autoround-int4-b70/realistic-suite-v1.json --max-tokens 512 --metric-tokens 100 --seed 1 --request-extra-json '{\"cache_prompt\":false,\"seed\":42,\"temperature\":0}' --out bench.json",
    "stop": "pkill -x llama-server"
   },
   "dependencies": [
    "repro/ornith-15-35b-a3b-q4km-b70/README.md",
    "repro/gemma4-26b-a4b-q8-b70/realistic-suite-v1.json",
    "repro/ornith-15-35b-a3b-q4km-b70/model-manifest.json",
    "repro/ornith-15-35b-a3b-q4km-b70/ornith-15-35b-a3b-q4km.sweep.json",
    "repro/ornith-15-35b-a3b-q4km-b70/ornith-15-35b-a3b-q4km.meta.json",
    "repro/ornith-15-35b-a3b-q4km-b70/depth-sweep.svg",
    "repro/ornith-15-35b-a3b-q4km-b70/ornith-15-35b-a3b-q4km-ten-feature.sweep.json",
    "repro/ornith-15-35b-a3b-q4km-b70/ornith-15-35b-a3b-q4km-ten-feature.meta.json",
    "repro/ornith-15-35b-a3b-q4km-b70/optimized-depth-sweep-ten-feature.svg",
    "repro/ornith-15-35b-a3b-q4km-b70/ornith-15-35b-a3b-q4km-eleven-feature.sweep.json",
    "repro/ornith-15-35b-a3b-q4km-b70/ornith-15-35b-a3b-q4km-eleven-feature.meta.json",
    "repro/ornith-15-35b-a3b-q4km-b70/optimized-depth-sweep-eleven-feature.svg",
    "repro/ornith-15-35b-a3b-q4km-b70/ornith-15-35b-a3b-q4km-twelve-feature.sweep.json",
    "repro/ornith-15-35b-a3b-q4km-b70/ornith-15-35b-a3b-q4km-twelve-feature.meta.json",
    "repro/ornith-15-35b-a3b-q4km-b70/optimized-depth-sweep-twelve-feature.svg",
    "scripts/verify-neural-download-model.py",
    "patches/ornith-15-35b-a3b-q4km-b70/README.md",
    "patches/ornith-15-35b-a3b-q4km-b70/llama-cpp-ornith15-ten-feature-stack-gdn-state-io-20260823.patch",
    "patches/ornith-15-35b-a3b-q4km-b70/llama-cpp-ornith15-eleven-feature-stack-qk-norm-rope-20260823.patch",
    "patches/ornith-15-35b-a3b-q4km-b70/llama-cpp-ornith15-twelve-feature-stack-shared-gate-residual-rms-20260823.patch",
    "experiments/ornith-15-b70/notes/2026-08-22-ornith35b-moe-add-reduce-positive.md",
    "experiments/ornith-15-b70/data/2026-08-22-ornith35b-moe-add-reduce-summary.json",
    "experiments/ornith-15-b70/data/2026-08-22-ornith35b-moe-add-reduce-canaries.json",
    "experiments/ornith-15-b70/notes/2026-08-22-ornith35b-conv-silu-positive.md",
    "experiments/ornith-15-b70/data/2026-08-22-ornith35b-conv-silu-summary.json",
    "experiments/ornith-15-b70/data/2026-08-22-ornith35b-conv-silu-canaries.json",
    "experiments/ornith-15-b70/notes/2026-08-22-ornith35b-residual-rms-positive.md",
    "experiments/ornith-15-b70/data/2026-08-22-ornith35b-residual-rms-summary.json",
    "experiments/ornith-15-b70/data/2026-08-22-ornith35b-residual-rms-canaries.json",
    "experiments/ornith-15-b70/notes/2026-08-22-ornith35b-concat-state-positive.md",
    "experiments/ornith-15-b70/data/2026-08-22-ornith35b-concat-state-summary.json",
    "experiments/ornith-15-b70/data/2026-08-22-ornith35b-concat-state-canaries.json",
    "experiments/ornith-15-b70/notes/2026-08-22-ornith35b-concat-state-direct-positive.md",
    "experiments/ornith-15-b70/data/2026-08-22-ornith35b-concat-state-direct-summary.json",
    "experiments/ornith-15-b70/data/2026-08-22-ornith35b-concat-state-direct-canaries.json",
    "experiments/ornith-15-b70/data/2026-08-22-ornith35b-concat-state-direct-exactness.json",
    "experiments/ornith-15-b70/notes/2026-08-22-ornith35b-alpha-gate-positive.md",
    "experiments/ornith-15-b70/data/2026-08-22-ornith35b-alpha-gate-summary.json",
    "experiments/ornith-15-b70/data/2026-08-22-ornith35b-alpha-gate-canaries.json",
    "experiments/ornith-15-b70/data/2026-08-22-ornith35b-alpha-gate-exactness.json",
    "experiments/ornith-15-b70/notes/2026-08-23-ornith35b-moe-gate-up-positive.md",
    "experiments/ornith-15-b70/data/2026-08-23-ornith35b-moe-gate-up-summary.json",
    "experiments/ornith-15-b70/data/2026-08-23-ornith35b-moe-gate-up-canaries.json",
    "experiments/ornith-15-b70/data/2026-08-23-ornith35b-moe-gate-up-exactness.json",
    "experiments/ornith-15-b70/notes/2026-08-23-ornith35b-moe-shared-residual-rms-positive.md",
    "experiments/ornith-15-b70/data/2026-08-23-ornith35b-moe-shared-residual-rms-summary.json",
    "experiments/ornith-15-b70/data/2026-08-23-ornith35b-moe-shared-residual-rms-canaries.json",
    "experiments/ornith-15-b70/data/2026-08-23-ornith35b-moe-shared-residual-rms-exactness.json",
    "experiments/ornith-15-b70/notes/2026-08-23-ornith35b-gdn-rms-silu-gate-positive.md",
    "experiments/ornith-15-b70/data/2026-08-23-ornith35b-gdn-rms-gate-summary.json",
    "experiments/ornith-15-b70/data/2026-08-23-ornith35b-gdn-rms-gate-canaries.json",
    "experiments/ornith-15-b70/data/2026-08-23-ornith35b-gdn-rms-gate-exactness.json",
    "experiments/ornith-15-b70/notes/2026-08-23-ornith35b-gdn-state-io-positive.md",
    "experiments/ornith-15-b70/data/2026-08-23-ornith35b-gdn-state-io-summary.json",
    "experiments/ornith-15-b70/data/2026-08-23-ornith35b-gdn-state-io-exactness.json",
    "experiments/ornith-15-b70/notes/2026-08-23-ornith35b-ten-feature-depth-sweep.md",
    "experiments/ornith-15-b70/notes/2026-08-23-ornith35b-qk-norm-rope-positive.md",
    "experiments/ornith-15-b70/data/2026-08-23-ornith35b-qk-norm-rope-summary.json",
    "experiments/ornith-15-b70/data/2026-08-23-ornith35b-qk-norm-rope-exactness.json",
    "experiments/ornith-15-b70/notes/2026-08-23-ornith35b-copy-offload-positive.md",
    "experiments/ornith-15-b70/data/2026-08-23-ornith35b-copy-offload-summary.json",
    "experiments/ornith-15-b70/notes/2026-08-23-ornith35b-shared-gate-residual-rms-positive.md",
    "experiments/ornith-15-b70/data/2026-08-23-ornith35b-shared-gate-residual-rms-summary.json",
    "experiments/ornith-15-b70/data/2026-08-23-ornith35b-shared-gate-residual-rms-exactness.json",
    "experiments/ornith-15-b70/notes/2026-08-23-ornith35b-multirow-aggregate-positive.md",
    "experiments/ornith-15-b70/data/2026-08-23-ornith35b-multirow-aggregate-summary.json",
    "experiments/ornith-15-b70/patches/llamacpp-ornith15-multirow-aggregate-fusions-candidate-20260823.patch",
    "experiments/ornith-15-b70/notes/2026-08-23-ornith35b-twelve-feature-depth-sweep.md",
    "experiments/ornith-15-b70/notes/2026-08-23-ornith35b-eleven-feature-depth-sweep.md",
    "experiments/ornith-15-b70/notes/2026-08-23-ornith35b-optimized-depth-sweep.md",
    "docs/neural-download-packet-standard.md",
    "scripts/bench-openai-realistic-suite.py",
    "repro/qwen36-27b-autoround-int4-b70/realistic-suite-v1.json"
   ],
   "missing": [
    "tested clean-host platform installation",
    "beginner recovery flow"
   ]
  },
  {
   "manifest": "packages/ornith-15-9b-q8-b70/package.json",
   "format": "b70-model-package-v1",
   "id": "ornith-15-9b-q8-b70",
   "name": "Ornith 1.5 9B Q8_0 on one Intel Arc Pro B70",
   "status": "candidate",
   "audience": "beginner",
   "guide": "repro/ornith-15-9b-q8-b70/README.md",
   "clean_host_tested": false,
   "library": {
    "model_family": "Ornith 1.5",
    "publisher": "Ornith AI",
    "variant": "9B",
    "summary": "Ornith AI's dense 9B on one Arc Pro B70 in 8-bit form with stock llama.cpp. A small, official, beginner-friendly deployment.",
    "quantization": "Q8_0",
    "runtime_label": "llama.cpp SYCL",
    "operating_systems": [
     "Linux"
    ],
    "delivery": [
     "native"
    ],
    "modalities": [
     "text"
    ],
    "use_cases": [
     "general",
     "chat",
     "starter"
    ],
    "tags": [
     "one card",
     "target only",
     "cache zero",
     "stock upstream",
     "beginner"
    ],
    "published_at": "2026-08-27",
    "featured_metric": null,
    "benchmark_status": "Strict headline withheld after output-gate failure. Two complete fresh-server, cache-zero, 12-prompt/six-class attempts measured 49.593582 and 49.515869 tok/s and passed both objective-canary batteries, but complete token arrays matched only 8/12 across servers."
   },
   "contributors": [
    {
     "id": "b70-optimization-lab",
     "name": "neural.download lab",
     "kind": "lab",
     "profile": "https://github.com/steveseguin/b70-optimization-lab",
     "contribution": "Intake verification, bring-up, baselines, depth sweep, operating point, canary battery, package.",
     "status": "integrated",
     "validated_effect": "Stock-upstream characterization; no lab patches in this package. The strict replay retained 49.593582 and 49.515869 tok/s as scoped diagnostics but withheld a headline after 8/12 fresh-server token-array equality.",
     "evidence": "data/2026-08-27-ornith15-9b-q8-tp1-strict-headline-comparison.json"
    }
   ],
   "performance_profiles": [
    {
     "id": "decode-vs-context-depth",
     "label": "Raw decode over existing context depth",
     "metric": "decode",
     "unit": "tok/s",
     "x_label": "Existing context depth before tg128",
     "scope": "llama-bench raw engine rates (pp2048/tg128, fa on, 5 reps). The directly measured zero-depth point remains in the linked raw evidence; no missing depth is interpolated.",
     "evidence": "repro/ornith-15-9b-q8-b70/ornith-15-9b-q8.sweep.json",
     "points": [
      {
       "context_tokens": 2048,
       "value": 49.338288,
       "samples": 5
      },
      {
       "context_tokens": 4096,
       "value": 48.554943,
       "samples": 5
      },
      {
       "context_tokens": 8192,
       "value": 47.044737,
       "samples": 5
      },
      {
       "context_tokens": 16384,
       "value": 44.349773,
       "samples": 5
      },
      {
       "context_tokens": 24576,
       "value": 41.957791,
       "samples": 5
      },
      {
       "context_tokens": 32768,
       "value": 39.83848,
       "samples": 5
      }
     ]
    },
    {
     "id": "prefill-vs-context-depth",
     "label": "Raw pp2048 over existing context depth",
     "metric": "prefill",
     "unit": "tok/s",
     "x_label": "Existing context depth before pp2048",
     "scope": "llama-bench raw engine rates (pp2048/tg128, fa on, 5 reps). The directly measured zero-depth point remains in the linked raw evidence; no missing depth is interpolated.",
     "evidence": "repro/ornith-15-9b-q8-b70/ornith-15-9b-q8.sweep.json",
     "points": [
      {
       "context_tokens": 2048,
       "value": 1623.026366,
       "samples": 5
      },
      {
       "context_tokens": 4096,
       "value": 1601.167744,
       "samples": 5
      },
      {
       "context_tokens": 8192,
       "value": 1568.596728,
       "samples": 5
      },
      {
       "context_tokens": 16384,
       "value": 1489.240859,
       "samples": 5
      },
      {
       "context_tokens": 24576,
       "value": 1448.815985,
       "samples": 5
      },
      {
       "context_tokens": 32768,
       "value": 1313.738568,
       "samples": 5
      }
     ]
    }
   ],
   "known_limitations": [
    "The 2026-08-27 strict pair passed both objective-canary batteries but matched only 8/12 complete natural-response token arrays across fresh servers; no general single-user headline is published.",
    "Clean-host beginner flow not yet executed.",
    "The measured decode rate requires local or sufficiently fast direct-attached model storage; a 100 Mb/s NFS-backed mmap reduced the matched audit-host diagnostic from 50.149 to 25.642 tok/s."
   ],
   "hardware": {
    "accelerator": "Intel Arc Pro B70 32 GiB",
    "cards": 1,
    "host_ram_min_bytes": 34359738368,
    "host_ram_plus_swap_min_bytes": 34359738368,
    "model_weight_bytes": 9527501248
   },
   "model": {
    "repository": "ornith-ai/Ornith-1.5-9B-GGUF",
    "revision": "85bf2b98cdcbad4291cb4f46943526cc089f75a0",
    "manifest": "repro/ornith-15-9b-q8-b70/model-manifest.json"
   },
   "runtime": {
    "kind": "native",
    "project": "ggml-org/llama.cpp",
    "revision": "9fee29e9435f865ec0b811a783a6471a136d9317",
    "build": "cmake -G Ninja -B build-sycl-aot-bmg-g31 -DCMAKE_BUILD_TYPE=Release -DCMAKE_CXX_COMPILER=icpx -DCMAKE_C_COMPILER=icx -DGGML_SYCL=ON -DGGML_SYCL_F16=ON -DGGML_SYCL_DEVICE_ARCH=bmg_g31 -DGGML_SYCL_MAX_PARALLEL_LINK_JOBS=32 -DLLAMA_CURL=OFF && cmake --build build-sycl-aot-bmg-g31 --target llama-server llama-bench -j 24"
   },
   "project_patches": {
    "required": false,
    "items": []
   },
   "commands": {
    "preflight": "python3 scripts/verify-neural-download-model.py repro/ornith-15-9b-q8-b70/model-manifest.json \"$MODEL_DIR\" && source /opt/intel/oneapi/setvars.sh --force && export ONEAPI_DEVICE_SELECTOR=level_zero:0",
    "launch": "$BUILD_DIR/bin/llama-server --model $MODEL_DIR/Ornith-1.5-9B-Q8_0.gguf --alias ornith9b --reasoning off --ctx-size 8192 --cache-type-k f16 --cache-type-v f16 --device SYCL0 --gpu-layers 99 --split-mode none --flash-attn auto --parallel 1 --cache-ram 0 --ctx-checkpoints 0 --no-cache-prompt --slot-prompt-similarity 0 --fit off --metrics --no-webui --host 127.0.0.1 --port 18100",
    "health": "curl -fsS http://127.0.0.1:18100/health",
    "benchmark": "python3 scripts/bench-openai-realistic-suite.py --base-url http://127.0.0.1:18100 --model ornith9b --api-mode native-raw --suite repro/qwen36-27b-autoround-int4-b70/realistic-suite-v1.json --max-tokens 512 --metric-tokens 100 --seed 42 --timeout 900 --return-token-ids --require-natural-eos --request-extra-json '{\"cache_prompt\":false,\"seed\":42,\"temperature\":0,\"top_p\":1}' --out performance.json && python3 scripts/neural-download-canaries.py --base-url http://127.0.0.1:18100 --model ornith9b --out canaries.json",
    "stop": "pkill -x llama-server"
   },
   "dependencies": [
    "repro/ornith-15-9b-q8-b70/README.md",
    "repro/gemma4-26b-a4b-q8-b70/realistic-suite-v1.json",
    "repro/ornith-15-9b-q8-b70/model-manifest.json",
    "repro/ornith-15-9b-q8-b70/ornith-15-9b-q8.sweep.json",
    "repro/ornith-15-9b-q8-b70/ornith-15-9b-q8.meta.json",
    "repro/ornith-15-9b-q8-b70/depth-sweep.svg",
    "data/2026-08-27-neural-download-stock-headline-closure-prereg.json",
    "data/2026-08-27-ornith15-9b-q8-tp1-strict-headline-comparison.json",
    "data/neural-download-stock-headline-ornith9-20260827-r1a/qualification.json",
    "data/neural-download-stock-headline-ornith9-20260827-r1b/qualification.json",
    "scripts/run-neural-download-stock-headline-attempt.sh",
    "data/neural-download-audit-host-depth-sweeps-20260822/README.md",
    "scripts/verify-neural-download-model.py",
    "docs/neural-download-packet-standard.md",
    "scripts/bench-openai-realistic-suite.py",
    "repro/qwen36-27b-autoround-int4-b70/realistic-suite-v1.json",
    "scripts/neural-download-canaries.py"
   ],
   "missing": [
    "tested clean-host platform installation",
    "additional operating points beyond the 8K standard",
    "beginner recovery flow"
   ]
  },
  {
   "manifest": "packages/qwen35-4b-w4a16-b70/package.json",
   "format": "b70-model-package-v1",
   "id": "qwen35-4b-w4a16-b70",
   "name": "Qwen3.5 4B W4A16 with its own MTP head on one or two Intel Arc Pro B70",
   "status": "candidate",
   "audience": "intermediate",
   "guide": "repro/qwen35-4b-w4a16-b70/README.md",
   "clean_host_tested": false,
   "library": {
    "model_family": "Qwen3.5",
    "publisher": "RedHatAI (W4A16 of Alibaba / Qwen)",
    "variant": "4B W4A16 (compressed-tensors INT4 weights, FP16 activations) with the publisher MTP head",
    "summary": "RedHatAI's W4A16 Qwen3.5-4B served by vLLM XPU on one B70 through the lab's R276 image, with the publisher's MTP head as a lossless speculative draft. The FP8-dynamic build of the same model cannot pass the base identity gate on this stack (three fresh-server pairs scored 11/12, 9/12, 11/12 with three tie-prone prompts); on the row-invariant W4A16 INT4 kernel the same gate passes 12/12, which is why this is the packaged route. The kernel is not the only reduction on that path: the RMSNorm this route runs is itself row-count dependent (measured 2026-09-08), so the route is exact in the regimes measured rather than exact by construction.",
    "quantization": "W4A16 (compressed-tensors INT4 weights, FP16 activations)",
    "runtime_label": "vLLM XPU",
    "operating_systems": [
     "Linux"
    ],
    "delivery": [
     "container"
    ],
    "modalities": [
     "text"
    ],
    "use_cases": [
     "general",
     "coding"
    ],
    "tags": [
     "qwen3.5",
     "4b",
     "int4",
     "w4a16",
     "compressed-tensors",
     "vllm",
     "xpu",
     "b70",
     "one card",
     "mtp",
     "draft int4 head",
     "batch-invariant",
     "xpu graph"
    ],
    "published_at": "2026-09-07",
    "featured_metric": {
     "value": 177.287,
     "unit": "tok/s",
     "label": "MTP depth 3 with the draft-only INT4 lm_head, one card, strict completions suite (center of two fresh servers, 177.406 / 177.168)",
     "scope": "Center of two fresh-server class-balanced medians over the strict fixed 12-prompt six-class suite through the completions API, 512-token cap, separate empty compile caches, 12/12 complete token arrays exact versus a same-configuration MTP0 oracle (102.625 / 102.376), canaries on every server, cache zero; one B70, R276 image, qwen3_5_mtp depth 3, draft-only INT4 lm_head, full decode-only XPU graph capture.",
     "evidence": "experiments/qwen35-4b-b70/data/2026-09-07-qwen35-4b-w4a16-matrix-result.json"
    },
    "benchmark_status": "Candidate (2026-09-07, campaign v1, one B70, R276 image): two fresh depth-3 servers 177.406 / 177.168 tok/s class-balanced median over tokens 1-100 on the strict 12-prompt completions suite, 12/12 vs each other and 12/12 vs the MTP0 oracle; two MTP0 servers 102.625 / 102.376 matched 12/12; canaries on every server; cache zero. The FP8-dynamic route of the same model fails that base gate (11/12, 9/12, 11/12 across three pairs). LocalMaxxing cmtrj2tp3000hps01n3fadg9d at 177.287. Identity ladders (128 tokens, two passes): without speculation exact through 32 users in both passes (1593.9 tok/s at c32; c64 1725.1 at 63/64 warm), with depth 3 exact through 16 users (1089.9 tok/s)."
   },
   "contributors": [
    {
     "id": "b70-optimization-lab",
     "name": "neural.download lab",
     "kind": "lab",
     "profile": "https://github.com/steveseguin/b70-optimization-lab",
     "contribution": "Fixed-K two-tier W4A16 kernel strategy (built for the Qwen3.8 INT4 lane, reused unchanged), lane composition on the R276 image and strict launchers, the FP8-versus-INT4 identity comparison, packaging.",
     "status": "integrated",
     "validated_effect": "On the row-invariant W4A16 kernel this model is repeat-exact (G1 12/12) where its FP8 build is not (11/12, 9/12, 11/12 across three fresh-server pairs), and MTP depth 3 with the draft-only INT4 head reaches 177.287 tok/s against 102.625 without speculation, 12/12 exact against the oracle.",
     "evidence": "experiments/qwen35-4b-b70/data/2026-09-07-qwen35-4b-w4a16-matrix-result.json"
    },
    {
     "id": "redhatai-qwen35-4b-w4a16",
     "name": "RedHatAI",
     "kind": "external",
     "profile": "https://huggingface.co/RedHatAI/Qwen3.5-4B-quantized.w4a16",
     "contribution": "W4A16 quantization of Qwen3.5-4B and the MTP head checkpoint served here unchanged.",
     "status": "credited",
     "validated_effect": "The published tensors serve as-is through vLLM's compressed-tensors path, which selects the lab's wNa16 INT4 kernel with no config relabel.",
     "evidence": "repro/qwen35-4b-w4a16-b70/manifests/model-direct-redhatai-qwen35-4b-w4a16-7a613872.json"
    }
   ],
   "hardware": {
    "accelerator": "Intel Arc Pro B70 32 GiB",
    "cards": 1,
    "host_ram_min_bytes": 16106127360,
    "host_ram_plus_swap_min_bytes": 21474836480,
    "model_weight_bytes": 5542033615,
    "note": "One card is the default profile; the two-card (TP2) profile is measured in the same recipe at 138.1 tok/s without speculation and a 240.6/227.5 depth-3 pair."
   },
   "model": {
    "repository": "RedHatAI/Qwen3.5-4B-quantized.w4a16",
    "revision": "7a613872f394578b0b52b683ff4ac47516b4bcaf",
    "manifest": "repro/qwen35-4b-w4a16-b70/manifests/model-direct-redhatai-qwen35-4b-w4a16-7a613872.json",
    "note": "Served exactly as published; the launcher verifies every LFS file against the manifest before the container starts."
   },
   "runtime": {
    "kind": "container",
    "image": "ghcr.io/steveseguin/vllm-openai-xpu-qwen38-int4@sha256:40d46730c9a24f9396cc67c0e5578dd80d11dfae7a4d23a55f97620140a0b3e6",
    "image_id_validated": "sha256:40d46730c9a24f9396cc67c0e5578dd80d11dfae7a4d23a55f97620140a0b3e6",
    "local_tag": "neural-download/vllm-openai-xpu:qwen38-int4-fp16-linear-classpad-cheapest-r293",
    "engine": "vLLM 0.27.2rc1.dev77+gac7509e2b (XPU) with the lab kernel library (vllm-xpu-kernels 1e90ffa6 + patches, oneDNN 0e2a5bfe + patches)",
    "launcher": "repro/qwen35-9b-fp8-b70/scripts/run-qwen35-9b-fp8-server.sh",
    "base": "R276 (sync-free grouped GDN branch) + R290-R293: the class-consistent FP16 linear mode (VLLM_XPU_FP16_LINEAR_CLASSPAD; CLASSPAD=0 runs the R276 code path unchanged)"
   },
   "project_patches": {
    "required": true,
    "items": [
     "experiments/qwen38-27b-b70/docker/r213b-w4a16-determinism-pad-op.py",
     "experiments/qwen38-27b-b70/docker/r224-fp16-linear-rowchunk.py",
     "experiments/qwen38-27b-b70/docker/r228-gdn-spec-group.py",
     "experiments/qwen38-27b-b70/docker/r256-draft-int4-head-fallback.py",
     "experiments/qwen38-27b-b70/patches/onednn-qwen38-w4a16-fixed-k-two-tier-r221-20260905.patch",
     "experiments/qwen38-27b-b70/patches/onednn-qwen38-w4a16-strategy-override-dump-r220-20260905.patch",
     "experiments/qwen38-27b-b70/patches/onednn-qwen38-w8a16-c-default-align-r137b-20260902.patch",
     "experiments/qwen38-27b-b70/patches/onednn-qwen38-w8a16-fixed-k-align16-r137a-20260902.patch"
    ],
    "note": "The image is the Qwen3.8 INT4 lane's R276 image; its overlays (determinism pad op inert here, FP16 row-chunk linears, grouped GDN speculative rows, draft-head fallback unused, sync-free GDN grouping) ship inside the container. No 9B-specific patch."
   },
   "commands": {
    "preflight": "docker pull ghcr.io/steveseguin/vllm-openai-xpu-qwen38-int4@sha256:40d46730c9a24f9396cc67c0e5578dd80d11dfae7a4d23a55f97620140a0b3e6 && docker tag ghcr.io/steveseguin/vllm-openai-xpu-qwen38-int4@sha256:40d46730c9a24f9396cc67c0e5578dd80d11dfae7a4d23a55f97620140a0b3e6 neural-download/vllm-openai-xpu:qwen38-int4-fp16-linear-classpad-cheapest-r293",
    "launch": "MODEL_DIR=/models/Qwen3.5-4B-quantized.w4a16 VLLM_CACHE_DIR=/tmp/qwen35-4b-cache MTP_DEPTH=3 CLASSPAD=0 repro/qwen35-4b-w4a16-b70/scripts/run-qwen35-4b-w4a16-server.sh",
    "health": "curl -fsS http://127.0.0.1:18131/health",
    "benchmark": "OUT_DIR=<new dir> BASE_URL=http://127.0.0.1:18131 MODEL_NAME=qwen35-4b-w4a16-mtp3 repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/bench-w8a16-mtp1-strict.sh",
    "stop": "docker stop qwen35-4b-w4a16-mtp3"
   },
   "dependencies": [
    "experiments/qwen35-4b-b70/data/2026-09-07-qwen35-4b-fp8-repeat-exactness.json",
    "experiments/qwen35-4b-b70/data/2026-09-07-qwen35-4b-w4a16-matrix-result.json",
    "experiments/qwen35-4b-b70/data/2026-09-08-qwen35-4b-w4a16-depth-sweep.json",
    "experiments/qwen35-4b-b70/data/qwen35-4b-w4a16-tp2-mtp3-graph1-dhint4-20260907-t1-strict-result.json",
    "experiments/qwen35-4b-b70/notes/2026-09-07-qwen35-4b-fp8-not-repeat-exact.md",
    "experiments/qwen35-9b-b70/data/2026-09-07-qwen35-9b-fp8-matrix-result.json",
    "experiments/qwen35-9b-b70/data/2026-09-07-qwen35-9b-w4a16-matrix-result.json",
    "experiments/qwen35-9b-b70/notes/2026-09-07-qwen35-9b-fp8-one-card-quick-lane.md",
    "experiments/qwen35-9b-b70/scripts/run-20260907-qwen35-9b-campaign-v2.sh",
    "experiments/qwen35-9b-b70/scripts/run-20260907-qwen35-campaign.sh",
    "experiments/qwen38-27b-b70/data/2026-08-25-qwen38-q4km-tp2-http-smallctx-suite.json",
    "experiments/qwen38-27b-b70/docker/r213b-w4a16-determinism-pad-op.py",
    "experiments/qwen38-27b-b70/docker/r224-fp16-linear-rowchunk.py",
    "experiments/qwen38-27b-b70/docker/r228-gdn-spec-group.py",
    "experiments/qwen38-27b-b70/docker/r256-draft-int4-head-fallback.py",
    "experiments/qwen38-27b-b70/patches/onednn-qwen38-w4a16-fixed-k-two-tier-r221-20260905.patch",
    "experiments/qwen38-27b-b70/patches/onednn-qwen38-w4a16-strategy-override-dump-r220-20260905.patch",
    "experiments/qwen38-27b-b70/patches/onednn-qwen38-w8a16-c-default-align-r137b-20260902.patch",
    "experiments/qwen38-27b-b70/patches/onednn-qwen38-w8a16-fixed-k-align16-r137a-20260902.patch",
    "packages/qwen35-4b-w4a16-b70/compose.yaml",
    "packages/qwen35-4b-w4a16-b70/env/one-gpu.env",
    "packages/qwen35-4b-w4a16-b70/env/two-gpu.env",
    "packages/qwen35-4b-w4a16-b70/scripts/download-model.sh",
    "packages/qwen35-4b-w4a16-b70/scripts/preflight.sh",
    "packages/qwen35-4b-w4a16-b70/scripts/render-compose.sh",
    "packages/qwen35-4b-w4a16-b70/scripts/smoke-test.sh",
    "packages/qwen35-4b-w4a16-b70/scripts/verify.sh",
    "repro/gemma4-26b-a4b-q8-b70/realistic-suite-v1.json",
    "repro/qwen35-4b-w4a16-b70/README.md",
    "repro/qwen35-4b-w4a16-b70/manifests/model-direct-redhatai-qwen35-4b-w4a16-7a613872.json",
    "repro/qwen35-4b-w4a16-b70/scripts/run-qwen35-4b-w4a16-server.sh",
    "repro/qwen35-9b-fp8-b70/README.md",
    "repro/qwen35-9b-fp8-b70/manifests/model-direct-redhatai-qwen35-9b-fp8-dynamic-790f0576.json",
    "repro/qwen35-9b-fp8-b70/scripts/run-qwen35-9b-fp8-server.sh",
    "repro/qwen35-9b-w4a16-b70/README.md",
    "repro/qwen35-9b-w4a16-b70/manifests/model-direct-redhatai-qwen35-9b-w4a16-a398088c.json",
    "repro/qwen35-9b-w4a16-b70/scripts/run-qwen35-9b-w4a16-server.sh",
    "repro/qwen36-27b-autoround-int4-b70/realistic-suite-v1.json",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/bench-w8a16-mtp1-strict.sh",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/run-w8a16-mtp0-strict-server.sh",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/run-w8a16-mtp1-strict-server.sh",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/verify-model-direct.sh",
    "scripts/bench-openai-concurrency-oracle.py",
    "scripts/bench-openai-realistic-suite.py",
    "scripts/compare-strict-attempt-outputs.py",
    "scripts/neural-download-canaries.py",
    "tools/check-container-packet.py",
    "tools/container-packet/download-model.sh",
    "tools/container-packet/preflight.sh",
    "tools/container-packet/smoke-test.sh",
    "tools/container-packet/verify.sh",
    "tools/render-container-compose.py",
    "tools/render-container-packet.sh",
    "experiments/qwen38-27b-b70/docker/Dockerfile.fp16-linear-classpad-r290",
    "experiments/qwen38-27b-b70/docker/r290-fp16-linear-classpad.py",
    "experiments/qwen38-27b-b70/docker/Dockerfile.fp16-linear-classpad-verified-r291",
    "experiments/qwen38-27b-b70/docker/r291-fp16-linear-classpad-verified.py",
    "experiments/qwen38-27b-b70/docker/Dockerfile.fp16-linear-classpad-nozero-r292",
    "experiments/qwen38-27b-b70/docker/r292-fp16-linear-classpad-nozero.py",
    "experiments/qwen38-27b-b70/docker/Dockerfile.fp16-linear-classpad-cheapest-r293",
    "experiments/qwen38-27b-b70/docker/r293-fp16-linear-classpad-cheapest.py",
    "experiments/qwen35-4b-b70/probes/fp16-linear-mclass-census.py",
    "experiments/qwen35-4b-b70/probes/fp16-linear-mclass-census-extended.py",
    "experiments/qwen35-4b-b70/probes/fp16-linear-classmap.py",
    "experiments/qwen35-4b-b70/probes/census-small-shapes.py",
    "experiments/qwen35-4b-b70/probes/validate-r290-classpad.py",
    "experiments/qwen35-4b-b70/probes/validate-classpad-tp1-shapes.py",
    "experiments/qwen35-4b-b70/probes/validate-r292-stale-pad.py",
    "experiments/qwen35-4b-b70/notes/2026-09-09-the-fp16-linear-chunk-is-a-throughput-tax.md",
    "experiments/qwen35-4b-b70/notes/2026-09-11-r293-class-consistent-fp16-linear-on-the-server.md",
    "repro/qwen38-27b-autoround-int4-b70/scripts/publish-r293-image-ghcr.sh",
    "experiments/qwen35-4b-b70/scripts/run-20260909-qwen35-4b-rowchunk-chain.sh",
    "experiments/qwen35-4b-b70/scripts/run-20260909-qwen35-4b-classpad-chain.sh",
    "experiments/qwen35-4b-b70/scripts/run-20260911-qwen35-4b-classpad-verified-chain.sh",
    "experiments/qwen35-4b-b70/scripts/run-20260911-qwen35-4b-classpad-r293-chain.sh",
    "experiments/qwen35-4b-b70/scripts/run-20260911-qwen35-4b-classpad-r293-tp2-stagger.sh",
    "experiments/qwen35-4b-b70/data/2026-09-11-qwen35-4b-r290-classpad-on-the-server.json",
    "experiments/qwen35-4b-b70/data/2026-09-09-qwen35-4b-fp16-linear-chunk-tax.json",
    "experiments/qwen35-4b-b70/data/2026-09-09-qwen35-4b-concurrency-throughput-matrix.json",
    "experiments/qwen35-4b-b70/data/2026-09-09-qwen35-4b-staggered-admission-is-exact.json"
   ],
   "evidence": [
    "experiments/qwen35-4b-b70/data/2026-09-07-qwen35-4b-fp8-repeat-exactness.json",
    "experiments/qwen35-4b-b70/data/2026-09-07-qwen35-4b-w4a16-matrix-result.json",
    "experiments/qwen35-4b-b70/data/2026-09-08-qwen35-4b-w4a16-depth-sweep.json",
    "experiments/qwen35-4b-b70/data/qwen35-4b-w4a16-tp2-mtp3-graph1-dhint4-20260907-t1-strict-result.json"
   ],
   "missing": [
    "clean-host replay"
   ],
   "performance_profiles": [
    {
     "id": "http-w4a16-mtp0-graph-tp1-identity-qualified-aggregate-vs-concurrent-users",
     "label": "Qwen3.5-4B W4A16 one card, no speculation, XPU graph capture, identity-qualified aggregate decode vs concurrent users (c1-c32; warm pass)",
     "metric": "aggregate_decode",
     "unit": "tok/s",
     "x_metric": "concurrent_sequences",
     "x_label": "Concurrent HTTP users",
     "scope": "Campaign v1 (2026-09-07): one B70, R276 image, strict launcher env, FULL_DECODE_ONLY capture sizes 1-64, no speculation, max-model-len 256, max-num-seqs 64, max-num-batched-tokens 512, 128 returned raw token IDs per response on the small-context suite, two passes on one server, warm pass shown. c1-c32 output-identity-qualified in both passes. Measured but withheld: c64 1725.1 (63/64 warm, 64/64 first pass).",
     "evidence": "experiments/qwen35-4b-b70/data/2026-09-07-qwen35-4b-w4a16-matrix-result.json",
     "points": [
      {
       "concurrent_sequences": 1,
       "value": 101.9,
       "samples": 1,
       "per_user_value": 101.9
      },
      {
       "concurrent_sequences": 2,
       "value": 193.5,
       "samples": 1,
       "per_user_value": 96.75
      },
      {
       "concurrent_sequences": 4,
       "value": 362.3,
       "samples": 1,
       "per_user_value": 90.575
      },
      {
       "concurrent_sequences": 8,
       "value": 655.5,
       "samples": 1,
       "per_user_value": 81.9375
      },
      {
       "concurrent_sequences": 16,
       "value": 1059.3,
       "samples": 1,
       "per_user_value": 66.2062
      },
      {
       "concurrent_sequences": 32,
       "value": 1593.9,
       "samples": 1,
       "per_user_value": 49.8094
      }
     ]
    },
    {
     "id": "http-w4a16-mtp3-drafthead-graph-tp1-identity-qualified-aggregate-vs-concurrent-users",
     "label": "Qwen3.5-4B W4A16 one card, MTP depth 3 with the draft INT4 head, identity-qualified aggregate decode vs concurrent users (c1-c16; warm pass)",
     "metric": "aggregate_decode",
     "unit": "tok/s",
     "x_metric": "concurrent_sequences",
     "x_label": "Concurrent HTTP users",
     "scope": "As above with qwen3_5_mtp depth 3 and the draft-only INT4 lm_head. c1-c16 output-identity-qualified in both passes. Measured but withheld: c32 1147.4 (30/32) and c64 1201.0 (55/64).",
     "evidence": "experiments/qwen35-4b-b70/data/2026-09-07-qwen35-4b-w4a16-matrix-result.json",
     "points": [
      {
       "concurrent_sequences": 1,
       "value": 159.2,
       "samples": 1,
       "per_user_value": 159.2
      },
      {
       "concurrent_sequences": 2,
       "value": 300.8,
       "samples": 1,
       "per_user_value": 150.4
      },
      {
       "concurrent_sequences": 4,
       "value": 537.8,
       "samples": 1,
       "per_user_value": 134.45
      },
      {
       "concurrent_sequences": 8,
       "value": 898.0,
       "samples": 1,
       "per_user_value": 112.25
      },
      {
       "concurrent_sequences": 16,
       "value": 1089.9,
       "samples": 1,
       "per_user_value": 68.1188
      }
     ]
    },
    {
     "id": "http-w4a16-mtp3-drafthead-decode-vs-active-context",
     "label": "Qwen3.5-4B W4A16 one card, MTP depth 3 with the draft INT4 head: decode over active context (2K-32K)",
     "metric": "decode",
     "unit": "tok/s",
     "x_label": "Actual prompt / active context tokens",
     "scope": "Campaign v2 (2026-09-07): one B70, one-slot vLLM HTTP completions at exactly 2K, 4K, 8K, 16K, 24K and 32K active context of unrepeated real content (three requests per depth, median shown), 128 output tokens, max-model-len 33024, cache zero, canaries before and after; every answer matched the same-configuration MTP0 oracle (18/18). No value is interpolated.",
     "evidence": "experiments/qwen35-4b-b70/data/2026-09-07-qwen35-4b-w4a16-matrix-result.json",
     "points": [
      {
       "context_tokens": 2048,
       "value": 176.989,
       "samples": 3
      },
      {
       "context_tokens": 4096,
       "value": 188.258,
       "samples": 3
      },
      {
       "context_tokens": 8192,
       "value": 188.765,
       "samples": 3
      },
      {
       "context_tokens": 16384,
       "value": 191.583,
       "samples": 3
      },
      {
       "context_tokens": 24576,
       "value": 161.462,
       "samples": 3
      },
      {
       "context_tokens": 32768,
       "value": 149.914,
       "samples": 3
      }
     ]
    },
    {
     "id": "http-w4a16-mtp0-decode-vs-active-context",
     "label": "Qwen3.5-4B W4A16 one card, no speculation: decode over active context (2K-32K)",
     "metric": "decode",
     "unit": "tok/s",
     "x_label": "Actual prompt / active context tokens",
     "scope": "Campaign v2 (2026-09-07): one B70, one-slot vLLM HTTP completions at exactly 2K, 4K, 8K, 16K, 24K and 32K active context of unrepeated real content (three requests per depth, median shown), 128 output tokens, max-model-len 33024, cache zero, canaries before and after; the oracle arm for the depth-3 profile. No value is interpolated.",
     "evidence": "experiments/qwen35-4b-b70/data/2026-09-07-qwen35-4b-w4a16-matrix-result.json",
     "points": [
      {
       "context_tokens": 2048,
       "value": 99.603,
       "samples": 3
      },
      {
       "context_tokens": 4096,
       "value": 98.131,
       "samples": 3
      },
      {
       "context_tokens": 8192,
       "value": 95.73,
       "samples": 3
      },
      {
       "context_tokens": 16384,
       "value": 91.57,
       "samples": 3
      },
      {
       "context_tokens": 24576,
       "value": 87.775,
       "samples": 3
      },
      {
       "context_tokens": 32768,
       "value": 84.438,
       "samples": 3
      }
     ]
    },
    {
     "id": "http-w4a16-r293-classpad-mtp0-graph-tp1-identity-qualified-aggregate-vs-concurrent-users",
     "label": "Qwen3.5-4B W4A16 one card, no speculation, CLASSPAD=1 (R293): identity-qualified aggregate decode vs concurrent users",
     "metric": "aggregate_decode",
     "unit": "tok/s",
     "x_metric": "concurrent_sequences",
     "x_label": "Concurrent HTTP users",
     "scope": "Chain 15 r1 (2026-09-11): one B70, R293 image with VLLM_XPU_FP16_LINEAR_CLASSPAD=1 (verified in the container and by the op's own census line), strict launcher env, FULL_DECODE_ONLY capture sizes 1-64, no speculation, max-model-len 256, max-num-seqs 64, max-num-batched-tokens 512, 128 returned raw token IDs per response on the small-context suite, four passes on one server, pass 2 shown. Points are rungs output-identity-qualified in the shown pass. Measured but withheld: c24 1354.9 (95/96), c64 2163.6 (255/256). With max-num-seqs 128 and max-num-batched-tokens 1024 (r6) c64/c96/c128 are exact at 2165.2 / 2400.1 / 2520.0; with the 5 ms admission stagger on the tie-site suite (r5) c64 is 1280/1280 over twenty passes at 2103.6, harness-certified.",
     "evidence": "experiments/qwen35-4b-b70/data/2026-09-11-qwen35-4b-r290-classpad-on-the-server.json",
     "points": [
      {
       "concurrent_sequences": 1,
       "value": 95.8,
       "samples": 1,
       "per_user_value": 95.8
      },
      {
       "concurrent_sequences": 2,
       "value": 183.1,
       "samples": 1,
       "per_user_value": 91.55
      },
      {
       "concurrent_sequences": 4,
       "value": 347.3,
       "samples": 1,
       "per_user_value": 86.825
      },
      {
       "concurrent_sequences": 8,
       "value": 637.5,
       "samples": 1,
       "per_user_value": 79.6875
      },
      {
       "concurrent_sequences": 12,
       "value": 827.0,
       "samples": 1,
       "per_user_value": 68.9167
      },
      {
       "concurrent_sequences": 16,
       "value": 1052.3,
       "samples": 1,
       "per_user_value": 65.7687
      },
      {
       "concurrent_sequences": 20,
       "value": 1198.5,
       "samples": 1,
       "per_user_value": 59.925
      },
      {
       "concurrent_sequences": 32,
       "value": 1624.5,
       "samples": 1,
       "per_user_value": 50.7656
      }
     ]
    },
    {
     "id": "http-w4a16-r293-classpad-mtp0-graph-tp2-identity-qualified-aggregate-vs-concurrent-users",
     "label": "Qwen3.5-4B W4A16 two cards, no speculation, CLASSPAD=1 (R293): identity-qualified aggregate decode vs concurrent users",
     "metric": "aggregate_decode",
     "unit": "tok/s",
     "x_metric": "concurrent_sequences",
     "x_label": "Concurrent HTTP users",
     "scope": "Chain 15 r2 (2026-09-11): two B70s (TP2), otherwise as the one-card R293 profile. Points are rungs output-identity-qualified in the shown pass (c1-c16). Measured but withheld: c20 1695.8 (78/80), c24 1947.3 (94/96), c32 2350.1 (126/128), c64 3316.7 (255/256). With max-num-seqs 128 and max-num-batched-tokens 1024 (r7): c64 3323.9 (255/256), c96 3810.2 (383/384), c128 4015.3 (512/512). With the 5 ms admission stagger on the tie-site suite (r10) c64 is 1280/1280 over twenty passes at 3163.9, harness-certified.",
     "evidence": "experiments/qwen35-4b-b70/data/2026-09-11-qwen35-4b-r290-classpad-on-the-server.json",
     "points": [
      {
       "concurrent_sequences": 1,
       "value": 126.4,
       "samples": 1,
       "per_user_value": 126.4
      },
      {
       "concurrent_sequences": 2,
       "value": 243.0,
       "samples": 1,
       "per_user_value": 121.5
      },
      {
       "concurrent_sequences": 4,
       "value": 462.9,
       "samples": 1,
       "per_user_value": 115.725
      },
      {
       "concurrent_sequences": 8,
       "value": 855.9,
       "samples": 1,
       "per_user_value": 106.9875
      },
      {
       "concurrent_sequences": 12,
       "value": 1138.3,
       "samples": 1,
       "per_user_value": 94.8583
      },
      {
       "concurrent_sequences": 16,
       "value": 1466.8,
       "samples": 1,
       "per_user_value": 91.675
      }
     ]
    },
    {
     "id": "http-w4a16-r293-classpad-mtp3-drafthead-graph-tp1-identity-qualified-aggregate-vs-concurrent-users",
     "label": "Qwen3.5-4B W4A16 one card, MTP depth 3 with the draft INT4 head, CLASSPAD=1 (R293): identity-qualified aggregate decode vs concurrent users",
     "metric": "aggregate_decode",
     "unit": "tok/s",
     "x_metric": "concurrent_sequences",
     "x_label": "Concurrent HTTP users",
     "scope": "Chain 15 r1 (2026-09-11), depth-3 lane of the same server pair. Points are rungs exact in the shown pass (c1-c16). Measured but withheld: c20 1355.3 (79/80), c24 1489.9 (92/96), c32 1607.9 (120/128), c64 1830.7 (222/256). Depth 2 (r8) leads depth 3 above c16: 1766.6 at c32 and 2030.6 at c64.",
     "evidence": "experiments/qwen35-4b-b70/data/2026-09-11-qwen35-4b-r290-classpad-on-the-server.json",
     "points": [
      {
       "concurrent_sequences": 1,
       "value": 151.4,
       "samples": 1,
       "per_user_value": 151.4
      },
      {
       "concurrent_sequences": 2,
       "value": 288.6,
       "samples": 1,
       "per_user_value": 144.3
      },
      {
       "concurrent_sequences": 4,
       "value": 522.7,
       "samples": 1,
       "per_user_value": 130.675
      },
      {
       "concurrent_sequences": 8,
       "value": 891.8,
       "samples": 1,
       "per_user_value": 111.475
      },
      {
       "concurrent_sequences": 12,
       "value": 1126.1,
       "samples": 1,
       "per_user_value": 93.8417
      },
      {
       "concurrent_sequences": 16,
       "value": 1352.7,
       "samples": 1,
       "per_user_value": 84.5438
      }
     ]
    }
   ],
   "container_packet": {
    "level": 2,
    "compose": "packages/qwen35-4b-w4a16-b70/compose.yaml",
    "generated": true,
    "generator": "packages/qwen35-4b-w4a16-b70/scripts/render-compose.sh",
    "generated_from": "repro/qwen35-4b-w4a16-b70/scripts/run-qwen35-4b-w4a16-server.sh",
    "ci_check": "tools/check-container-packet.py",
    "profiles": {
     "one-gpu": {
      "cards": 1,
      "tensor_parallel_size": 1
     },
     "two-gpu": {
      "cards": 2,
      "tensor_parallel_size": 2
     }
    },
    "env_files": [
     "packages/qwen35-4b-w4a16-b70/env/one-gpu.env",
     "packages/qwen35-4b-w4a16-b70/env/two-gpu.env"
    ],
    "scripts": {
     "preflight": "packages/qwen35-4b-w4a16-b70/scripts/preflight.sh",
     "download": "packages/qwen35-4b-w4a16-b70/scripts/download-model.sh",
     "verify": "packages/qwen35-4b-w4a16-b70/scripts/verify.sh",
     "smoke_test": "packages/qwen35-4b-w4a16-b70/scripts/smoke-test.sh"
    },
    "shared_implementation": "tools/container-packet/",
    "note": "compose.yaml is rendered from the argv repro/qwen35-4b-w4a16-b70/scripts/run-qwen35-4b-w4a16-server.sh hands to docker run, so it carries the measured container's 73 environment variables and full serve command verbatim rather than a hand-copied approximation. The model is mounted read-only, the image is pinned by digest, and the port is bound to loopback; CI fails the build if any of that stops holding."
   }
  },
  {
   "manifest": "packages/qwen35-9b-fp8-b70/package.json",
   "format": "b70-model-package-v1",
   "id": "qwen35-9b-fp8-b70",
   "name": "Qwen3.5 9B FP8-dynamic with its own MTP head on one Intel Arc Pro B70",
   "status": "candidate",
   "audience": "intermediate",
   "guide": "repro/qwen35-9b-fp8-b70/README.md",
   "clean_host_tested": false,
   "library": {
    "model_family": "Qwen3.5",
    "publisher": "RedHatAI (FP8-dynamic of Alibaba / Qwen)",
    "variant": "9B FP8-dynamic (compressed-tensors, per-channel FP8 weights, dynamic activations) with the publisher's MTP head",
    "summary": "RedHatAI's FP8-dynamic Qwen3.5-9B served by vLLM XPU on one B70 through the lab's published R276 image and the Qwen3.8 FP8 recipe's strict launchers, with the model's own MTP head as a lossless speculative draft (qwen3_5_mtp) and full decode-only XPU graph capture. On this hardware the W4A16 build of the same model (packages/qwen35-9b-w4a16-b70) is the better route: faster at every depth and byte-exact at every concurrency through 64 users, where this FP8 route flips a near-tie token from 16 users up.",
    "quantization": "FP8-dynamic (compressed-tensors, per-channel FP8 weights, dynamic activations)",
    "runtime_label": "vLLM XPU",
    "operating_systems": [
     "Linux"
    ],
    "delivery": [
     "container"
    ],
    "modalities": [
     "text"
    ],
    "use_cases": [
     "general",
     "coding"
    ],
    "tags": [
     "qwen3.5",
     "9b",
     "fp8",
     "compressed-tensors",
     "vllm",
     "xpu",
     "b70",
     "one card",
     "mtp",
     "draft int4 head",
     "xpu graph"
    ],
    "published_at": "2026-09-07",
    "featured_metric": {
     "value": 98.139,
     "unit": "tok/s",
     "label": "MTP depth 3 with the draft-only INT4 lm_head, one card, strict completions suite (center of two fresh servers, 98.251 / 98.027)",
     "scope": "Center of two fresh-server class-balanced medians (98.251 / 98.027) over the strict fixed 12-prompt six-class suite through the completions API, 512-token cap, separate empty compile caches, 12/12 complete token arrays exact versus a same-configuration MTP0 oracle (50.165 / 50.173), canaries on every server, cache zero; one B70, R276 image, qwen3_5_mtp depth 3, draft-only INT4 lm_head (VLLM_XPU_DRAFT_LM_HEAD_INT4=1), full decode-only XPU graph capture.",
     "evidence": "experiments/qwen35-9b-b70/data/2026-09-07-qwen35-9b-fp8-matrix-result.json"
    },
    "benchmark_status": "Candidate (2026-09-07, campaigns c1-c3, one B70, R276 image): two fresh depth-3 servers with the draft-only INT4 lm_head 98.251 / 98.027 tok/s class-balanced median over tokens 1-100 on the strict 12-prompt completions suite (512-token cap, cache zero, canaries on every server), 12/12 vs each other and 12/12 vs the MTP0 oracle; with the FP8 draft head 76.917 / 76.879 (c1); two MTP0 servers 50.18 / 50.15 matched 12/12. Depth 4 (c3) is withheld: repeat-exact but 8/12 vs the oracle and no faster. Identity ladders (128 tokens, two passes, c2): depth 3 exact through c8 (565 tok/s), c16 15/16 (738), c32 30/32 (884), c64 57/64 (831); MTP0 exact through c8 (355), c64 59/64 at 1254 tok/s aggregate. Chat-mode workload with visible thinking measures 131.5 tok/s on the same prompts (workload note, not the headline). Graph off (c4): 96.779 / 96.748, lossless (capture worth ~1.5% on one card). Two cards, TP2 (c5): MTP0 79.456 / 79.400 and depth 3 with the draft INT4 head 147.712 / 147.890, G1/G2/G3 12/12, LocalMaxxing cmtqzhlvn00c0pa0109xxfb2f. 2K-32K real-content ladders: one card depth 3 105.9 (2K) to 86.8 (32K) tok/s and two cards 171.3 to 140.6, every answer identical to the MTP0 oracle (18/18 each); depths 5 and 6 (c9/c10) withheld (91.3 / 88.6, 8/12). Matrix complete 2026-09-07."
   },
   "contributors": [
    {
     "id": "b70-optimization-lab",
     "name": "neural.download lab",
     "kind": "lab",
     "profile": "https://github.com/steveseguin/b70-optimization-lab",
     "contribution": "Lane composition on the existing deterministic vLLM XPU stack (R276 image, lab GDN kernels, strict launchers), speculative depth and capture sweeps, identity gates and ladders, packaging.",
     "status": "integrated",
     "validated_effect": "MTP depth 3 with XPU graph capture and the draft-only INT4 lm_head raised one-card decode from 50.2 (no speculation) to 98.1 tok/s on the strict completions suite, 12/12 exact against the MTP0 oracle on two fresh servers (campaign c2, 2026-09-07); the FP8 draft head alone gave 76.9 (c1).",
     "evidence": "experiments/qwen35-9b-b70/data/2026-09-07-qwen35-9b-fp8-matrix-result.json"
    },
    {
     "id": "redhatai-qwen35-9b-fp8",
     "name": "RedHatAI",
     "kind": "external",
     "profile": "https://huggingface.co/RedHatAI/Qwen3.5-9B-FP8-dynamic",
     "contribution": "FP8-dynamic quantization of Qwen3.5-9B and the MTP head checkpoint served here unchanged.",
     "status": "credited",
     "validated_effect": "The published tensors serve as-is through vLLM's compressed-tensors path; the MTP head yields 3-token drafts that the FP8 target verifies losslessly.",
     "evidence": "repro/qwen35-9b-fp8-b70/manifests/model-direct-redhatai-qwen35-9b-fp8-dynamic-790f0576.json"
    }
   ],
   "performance_profiles": [
    {
     "id": "http-fp8-mtp3-drafthead-graph-tp1-identity-qualified-aggregate-vs-concurrent-users",
     "label": "Qwen3.5-9B FP8 one card, MTP depth 3 with the draft INT4 head and XPU graph capture, identity-qualified aggregate decode vs concurrent users (c1-c8; warm pass)",
     "metric": "aggregate_decode",
     "unit": "tok/s",
     "x_metric": "concurrent_sequences",
     "x_label": "Concurrent HTTP users",
     "scope": "Campaign c2 (2026-09-07): one B70, R276 image, strict launcher env, FULL_DECODE_ONLY capture sizes 1-64, qwen3_5_mtp depth 3, VLLM_XPU_DRAFT_LM_HEAD_INT4=1, max-model-len 256, max-num-seqs 64, max-num-batched-tokens 512, 128 returned raw token IDs per response on the small-context suite, two passes on one server, warm pass shown. c1-c8 output-identity-qualified in both passes. Measured but withheld: c16 738.4 (15/16), c32 883.6 (30/32), c64 831.2 (57/64).",
     "evidence": "experiments/qwen35-9b-b70/data/2026-09-07-qwen35-9b-fp8-matrix-result.json",
     "points": [
      {
       "concurrent_sequences": 1,
       "value": 113.0,
       "samples": 1,
       "per_user_value": 113.0
      },
      {
       "concurrent_sequences": 2,
       "value": 211.1,
       "samples": 1,
       "per_user_value": 105.55
      },
      {
       "concurrent_sequences": 4,
       "value": 310.7,
       "samples": 1,
       "per_user_value": 77.675
      },
      {
       "concurrent_sequences": 8,
       "value": 564.7,
       "samples": 1,
       "per_user_value": 70.5875
      }
     ]
    },
    {
     "id": "http-fp8-mtp0-graph-tp1-identity-qualified-aggregate-vs-concurrent-users",
     "label": "Qwen3.5-9B FP8 one card, no speculation, XPU graph capture, identity-qualified aggregate decode vs concurrent users (c1-c8; warm pass)",
     "metric": "aggregate_decode",
     "unit": "tok/s",
     "x_metric": "concurrent_sequences",
     "x_label": "Concurrent HTTP users",
     "scope": "Campaign c2 MTP0 ladder (as above without speculation). c1-c8 output-identity-qualified in both passes. Measured but withheld: c16 634.7 (15/16), c32 1055.2 (31/32), c64 1253.8 (59/64).",
     "evidence": "experiments/qwen35-9b-b70/data/2026-09-07-qwen35-9b-fp8-matrix-result.json",
     "points": [
      {
       "concurrent_sequences": 1,
       "value": 50.1,
       "samples": 1,
       "per_user_value": 50.1
      },
      {
       "concurrent_sequences": 2,
       "value": 97.2,
       "samples": 1,
       "per_user_value": 48.6
      },
      {
       "concurrent_sequences": 4,
       "value": 187.5,
       "samples": 1,
       "per_user_value": 46.875
      },
      {
       "concurrent_sequences": 8,
       "value": 355.0,
       "samples": 1,
       "per_user_value": 44.375
      }
     ]
    },
    {
     "id": "http-fp8-mtp3-drafthead-graph-tp2-identity-qualified-aggregate-vs-concurrent-users",
     "label": "Qwen3.5-9B FP8 two cards (TP2), MTP depth 3 with the draft INT4 head and XPU graph capture, identity-qualified aggregate decode vs concurrent users (c1-c8; warm pass)",
     "metric": "aggregate_decode",
     "unit": "tok/s",
     "x_metric": "concurrent_sequences",
     "x_label": "Concurrent HTTP users",
     "scope": "Campaign c5 (2026-09-07): two B70s (TENSOR_PARALLEL_SIZE=2), otherwise as the one-card depth-3 profile. c1-c8 output-identity-qualified in both passes. Measured but withheld: c16 1156.7 (15/16; first pass 16/16 at 969.3), c32 1422.2 (29/32), c64 1578.1 (57/64).",
     "evidence": "experiments/qwen35-9b-b70/data/2026-09-07-qwen35-9b-fp8-matrix-result.json",
     "points": [
      {
       "concurrent_sequences": 1,
       "value": 177.5,
       "samples": 1,
       "per_user_value": 177.5
      },
      {
       "concurrent_sequences": 2,
       "value": 337.2,
       "samples": 1,
       "per_user_value": 168.6
      },
      {
       "concurrent_sequences": 4,
       "value": 480.1,
       "samples": 1,
       "per_user_value": 120.025
      },
      {
       "concurrent_sequences": 8,
       "value": 848.7,
       "samples": 1,
       "per_user_value": 106.0875
      }
     ]
    },
    {
     "id": "http-fp8-mtp0-graph-tp2-identity-qualified-aggregate-vs-concurrent-users",
     "label": "Qwen3.5-9B FP8 two cards (TP2), no speculation, XPU graph capture, identity-qualified aggregate decode vs concurrent users (c1-c8; warm pass)",
     "metric": "aggregate_decode",
     "unit": "tok/s",
     "x_metric": "concurrent_sequences",
     "x_label": "Concurrent HTTP users",
     "scope": "Campaign c5 MTP0 ladder on two B70s. c1-c8 output-identity-qualified in both passes. Measured but withheld: c16 1013.9 (13/16), c32 1680.7 (27/32), c64 2079.9 (59/64).",
     "evidence": "experiments/qwen35-9b-b70/data/2026-09-07-qwen35-9b-fp8-matrix-result.json",
     "points": [
      {
       "concurrent_sequences": 1,
       "value": 79.5,
       "samples": 1,
       "per_user_value": 79.5
      },
      {
       "concurrent_sequences": 2,
       "value": 153.3,
       "samples": 1,
       "per_user_value": 76.65
      },
      {
       "concurrent_sequences": 4,
       "value": 296.6,
       "samples": 1,
       "per_user_value": 74.15
      },
      {
       "concurrent_sequences": 8,
       "value": 560.5,
       "samples": 1,
       "per_user_value": 70.0625
      }
     ]
    },
    {
     "id": "http-fp8-mtp3-drafthead-decode-vs-active-context",
     "label": "Qwen3.5-9B FP8 one card, MTP depth 3 with the draft INT4 head: decode over active context (2K-32K)",
     "metric": "decode",
     "unit": "tok/s",
     "x_label": "Actual prompt / active context tokens",
     "scope": "Campaign c7 (2026-09-07): one B70, R276 image, one-slot vLLM HTTP completions at exactly 2K, 4K, 8K, 16K, 24K and 32K active context of unrepeated real content (technical prose, Python, structured documents; three requests per depth, median shown), 128 output tokens, max-model-len 33024, max-num-batched-tokens 4096, cache zero, canaries before and after; every answer matched the same-configuration MTP0 oracle (18/18). No value is interpolated.",
     "evidence": "experiments/qwen35-9b-b70/data/2026-09-07-qwen35-9b-fp8-matrix-result.json",
     "points": [
      {
       "context_tokens": 2048,
       "value": 105.88,
       "samples": 3
      },
      {
       "context_tokens": 4096,
       "value": 126.208,
       "samples": 3
      },
      {
       "context_tokens": 8192,
       "value": 120.092,
       "samples": 3
      },
      {
       "context_tokens": 16384,
       "value": 103.277,
       "samples": 3
      },
      {
       "context_tokens": 24576,
       "value": 100.235,
       "samples": 3
      },
      {
       "context_tokens": 32768,
       "value": 86.789,
       "samples": 3
      }
     ]
    },
    {
     "id": "http-fp8-mtp0-decode-vs-active-context",
     "label": "Qwen3.5-9B FP8 one card, no speculation: decode over active context (2K-32K)",
     "metric": "decode",
     "unit": "tok/s",
     "x_label": "Actual prompt / active context tokens",
     "scope": "Campaign c7 MTP0 arm (the oracle for the depth-3 profile), same workload and shape.",
     "evidence": "experiments/qwen35-9b-b70/data/2026-09-07-qwen35-9b-fp8-matrix-result.json",
     "points": [
      {
       "context_tokens": 2048,
       "value": 49.647,
       "samples": 3
      },
      {
       "context_tokens": 4096,
       "value": 49.257,
       "samples": 3
      },
      {
       "context_tokens": 8192,
       "value": 48.648,
       "samples": 3
      },
      {
       "context_tokens": 16384,
       "value": 47.527,
       "samples": 3
      },
      {
       "context_tokens": 24576,
       "value": 46.494,
       "samples": 3
      },
      {
       "context_tokens": 32768,
       "value": 45.538,
       "samples": 3
      }
     ]
    },
    {
     "id": "http-fp8-mtp3-drafthead-tp2-decode-vs-active-context",
     "label": "Qwen3.5-9B FP8 two cards (TP2), MTP depth 3 with the draft INT4 head: decode over active context (2K-32K)",
     "metric": "decode",
     "unit": "tok/s",
     "x_label": "Actual prompt / active context tokens",
     "scope": "Campaign c8 (2026-09-07): two B70s (TP2), otherwise as the one-card context profile; 18/18 answers matched the two-card MTP0 oracle.",
     "evidence": "experiments/qwen35-9b-b70/data/2026-09-07-qwen35-9b-fp8-matrix-result.json",
     "points": [
      {
       "context_tokens": 2048,
       "value": 171.271,
       "samples": 3
      },
      {
       "context_tokens": 4096,
       "value": 202.015,
       "samples": 3
      },
      {
       "context_tokens": 8192,
       "value": 192.086,
       "samples": 3
      },
      {
       "context_tokens": 16384,
       "value": 166.413,
       "samples": 3
      },
      {
       "context_tokens": 24576,
       "value": 161.533,
       "samples": 3
      },
      {
       "context_tokens": 32768,
       "value": 140.559,
       "samples": 3
      }
     ]
    }
   ],
   "hardware": {
    "accelerator": "Intel Arc Pro B70 32 GiB",
    "cards": 1,
    "host_ram_min_bytes": 16106127360,
    "host_ram_plus_swap_min_bytes": 21474836480,
    "model_weight_bytes": 14026775199,
    "note": "One card is the package shape; TP2 rows (two cards) are measured in the same recipe and reported as separate profiles."
   },
   "model": {
    "repository": "RedHatAI/Qwen3.5-9B-FP8-dynamic",
    "revision": "790f0576d2d77dd5322aa0603a470bd9e3a3d1f6",
    "manifest": "repro/qwen35-9b-fp8-b70/manifests/model-direct-redhatai-qwen35-9b-fp8-dynamic-790f0576.json",
    "note": "Served exactly as published; the launcher verifies every LFS file against the manifest before the container starts."
   },
   "runtime": {
    "kind": "container",
    "image": "ghcr.io/steveseguin/vllm-openai-xpu-qwen38-int4@sha256:521eb277c0733f8c2ce47aea1bb98ed576c6f1ad63bf5baf22d38fc07abf54ad",
    "image_id_validated": "sha256:521eb277c0733f8c2ce47aea1bb98ed576c6f1ad63bf5baf22d38fc07abf54ad",
    "local_tag": "neural-download/vllm-openai-xpu:qwen38-int4-gdn-spec-group-sync-free-r276",
    "engine": "vLLM 0.27.2rc1.dev77+gac7509e2b (XPU) with the lab kernel library (vllm-xpu-kernels 1e90ffa6 + patches, oneDNN 0e2a5bfe + patches)",
    "launcher": "repro/qwen35-9b-fp8-b70/scripts/run-qwen35-9b-fp8-server.sh"
   },
   "project_patches": {
    "required": true,
    "items": [
     "experiments/qwen38-27b-b70/docker/r213b-w4a16-determinism-pad-op.py",
     "experiments/qwen38-27b-b70/docker/r224-fp16-linear-rowchunk.py",
     "experiments/qwen38-27b-b70/docker/r228-gdn-spec-group.py",
     "experiments/qwen38-27b-b70/docker/r256-draft-int4-head-fallback.py",
     "experiments/qwen38-27b-b70/patches/onednn-qwen38-w4a16-fixed-k-two-tier-r221-20260905.patch",
     "experiments/qwen38-27b-b70/patches/onednn-qwen38-w4a16-strategy-override-dump-r220-20260905.patch",
     "experiments/qwen38-27b-b70/patches/onednn-qwen38-w8a16-c-default-align-r137b-20260902.patch",
     "experiments/qwen38-27b-b70/patches/onednn-qwen38-w8a16-fixed-k-align16-r137a-20260902.patch"
    ],
    "note": "The image is the Qwen3.8 INT4 lane's R276 image; its overlays (determinism pad op inert here, FP16 row-chunk linears, grouped GDN speculative rows, draft-head fallback unused, sync-free GDN grouping) ship inside the container. No 9B-specific patch."
   },
   "commands": {
    "preflight": "docker pull ghcr.io/steveseguin/vllm-openai-xpu-qwen38-int4@sha256:521eb277c0733f8c2ce47aea1bb98ed576c6f1ad63bf5baf22d38fc07abf54ad && docker tag ghcr.io/steveseguin/vllm-openai-xpu-qwen38-int4@sha256:521eb277c0733f8c2ce47aea1bb98ed576c6f1ad63bf5baf22d38fc07abf54ad neural-download/vllm-openai-xpu:qwen38-int4-gdn-spec-group-sync-free-r276",
    "launch": "MODEL_DIR=/models/Qwen3.5-9B-FP8-dynamic VLLM_CACHE_DIR=/tmp/qwen35-cache MTP_DEPTH=3 repro/qwen35-9b-fp8-b70/scripts/run-qwen35-9b-fp8-server.sh",
    "health": "curl -fsS http://127.0.0.1:18131/health",
    "benchmark": "OUT_DIR=<new dir> BASE_URL=http://127.0.0.1:18131 MODEL_NAME=qwen35-9b-fp8-mtp3 repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/bench-w8a16-mtp1-strict.sh",
    "stop": "docker stop qwen35-9b-fp8-mtp3"
   },
   "dependencies": [
    "experiments/qwen35-9b-b70/data/2026-09-07-qwen35-9b-fp8-matrix-result.json",
    "experiments/qwen35-9b-b70/notes/2026-09-07-qwen35-9b-fp8-one-card-quick-lane.md",
    "experiments/qwen35-9b-b70/scripts/run-20260907-qwen35-9b-campaign-v2.sh",
    "experiments/qwen38-27b-b70/data/2026-08-25-qwen38-q4km-tp2-http-smallctx-suite.json",
    "experiments/qwen38-27b-b70/docker/r213b-w4a16-determinism-pad-op.py",
    "experiments/qwen38-27b-b70/docker/r224-fp16-linear-rowchunk.py",
    "experiments/qwen38-27b-b70/docker/r228-gdn-spec-group.py",
    "experiments/qwen38-27b-b70/docker/r256-draft-int4-head-fallback.py",
    "experiments/qwen38-27b-b70/patches/onednn-qwen38-w4a16-fixed-k-two-tier-r221-20260905.patch",
    "experiments/qwen38-27b-b70/patches/onednn-qwen38-w4a16-strategy-override-dump-r220-20260905.patch",
    "experiments/qwen38-27b-b70/patches/onednn-qwen38-w8a16-c-default-align-r137b-20260902.patch",
    "experiments/qwen38-27b-b70/patches/onednn-qwen38-w8a16-fixed-k-align16-r137a-20260902.patch",
    "packages/qwen35-9b-fp8-b70/compose.yaml",
    "packages/qwen35-9b-fp8-b70/env/one-gpu.env",
    "packages/qwen35-9b-fp8-b70/env/two-gpu.env",
    "packages/qwen35-9b-fp8-b70/scripts/download-model.sh",
    "packages/qwen35-9b-fp8-b70/scripts/preflight.sh",
    "packages/qwen35-9b-fp8-b70/scripts/render-compose.sh",
    "packages/qwen35-9b-fp8-b70/scripts/smoke-test.sh",
    "packages/qwen35-9b-fp8-b70/scripts/verify.sh",
    "repro/gemma4-26b-a4b-q8-b70/realistic-suite-v1.json",
    "repro/qwen35-9b-fp8-b70/README.md",
    "repro/qwen35-9b-fp8-b70/manifests/model-direct-redhatai-qwen35-9b-fp8-dynamic-790f0576.json",
    "repro/qwen35-9b-fp8-b70/scripts/run-qwen35-9b-fp8-server.sh",
    "repro/qwen35-9b-w4a16-b70/README.md",
    "repro/qwen36-27b-autoround-int4-b70/realistic-suite-v1.json",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/bench-w8a16-mtp1-strict.sh",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/run-w8a16-mtp0-strict-server.sh",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/run-w8a16-mtp1-strict-server.sh",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/verify-model-direct.sh",
    "scripts/bench-openai-concurrency-oracle.py",
    "scripts/bench-openai-realistic-suite.py",
    "scripts/compare-strict-attempt-outputs.py",
    "scripts/neural-download-canaries.py",
    "tools/check-container-packet.py",
    "tools/container-packet/download-model.sh",
    "tools/container-packet/preflight.sh",
    "tools/container-packet/smoke-test.sh",
    "tools/container-packet/verify.sh",
    "tools/render-container-compose.py",
    "tools/render-container-packet.sh"
   ],
   "evidence": [
    "experiments/qwen35-9b-b70/data/2026-09-07-qwen35-9b-fp8-matrix-result.json"
   ],
   "missing": [
    "clean-host replay"
   ],
   "container_packet": {
    "level": 2,
    "compose": "packages/qwen35-9b-fp8-b70/compose.yaml",
    "generated": true,
    "generator": "packages/qwen35-9b-fp8-b70/scripts/render-compose.sh",
    "generated_from": "repro/qwen35-9b-fp8-b70/scripts/run-qwen35-9b-fp8-server.sh",
    "ci_check": "tools/check-container-packet.py",
    "profiles": {
     "one-gpu": {
      "cards": 1,
      "tensor_parallel_size": 1
     },
     "two-gpu": {
      "cards": 2,
      "tensor_parallel_size": 2
     }
    },
    "env_files": [
     "packages/qwen35-9b-fp8-b70/env/one-gpu.env",
     "packages/qwen35-9b-fp8-b70/env/two-gpu.env"
    ],
    "scripts": {
     "preflight": "packages/qwen35-9b-fp8-b70/scripts/preflight.sh",
     "download": "packages/qwen35-9b-fp8-b70/scripts/download-model.sh",
     "verify": "packages/qwen35-9b-fp8-b70/scripts/verify.sh",
     "smoke_test": "packages/qwen35-9b-fp8-b70/scripts/smoke-test.sh"
    },
    "shared_implementation": "tools/container-packet/",
    "note": "compose.yaml is rendered from the argv repro/qwen35-9b-fp8-b70/scripts/run-qwen35-9b-fp8-server.sh hands to docker run, so it carries the measured container's 73 environment variables and full serve command verbatim rather than a hand-copied approximation. The model is mounted read-only, the image is pinned by digest, and the port is bound to loopback; CI fails the build if any of that stops holding."
   }
  },
  {
   "manifest": "packages/qwen35-9b-w4a16-b70/package.json",
   "format": "b70-model-package-v1",
   "id": "qwen35-9b-w4a16-b70",
   "name": "Qwen3.5 9B W4A16 with its own MTP head on one or two Intel Arc Pro B70",
   "status": "candidate",
   "audience": "intermediate",
   "guide": "repro/qwen35-9b-w4a16-b70/README.md",
   "clean_host_tested": false,
   "library": {
    "model_family": "Qwen3.5",
    "publisher": "RedHatAI (W4A16 of Alibaba / Qwen)",
    "variant": "9B W4A16 (compressed-tensors INT4 weights, FP16 activations) with the publisher MTP head",
    "summary": "RedHatAI's W4A16 Qwen3.5-9B served by vLLM XPU on one B70 through the lab's R276 image, with the publisher's MTP head as a lossless speculative draft. vLLM's compressed-tensors path selects the lab's wNa16 INT4 kernel, whose fixed-K two-tier strategy is row-count invariant: without speculation this route is byte-exact against a single request at every concurrency through 64 users, where the FP8 route on the same model flips a near-tie token from 16 users up. It is also 28% faster without speculation and 15% faster with it. The kernel is not the only reduction on that path: the RMSNorm this route runs is itself row-count dependent (measured 2026-09-08), so the route is exact in the regimes measured rather than exact by construction.",
    "quantization": "W4A16 (compressed-tensors INT4 weights, FP16 activations)",
    "runtime_label": "vLLM XPU",
    "operating_systems": [
     "Linux"
    ],
    "delivery": [
     "container"
    ],
    "modalities": [
     "text"
    ],
    "use_cases": [
     "general",
     "coding"
    ],
    "tags": [
     "qwen3.5",
     "9b",
     "int4",
     "w4a16",
     "compressed-tensors",
     "vllm",
     "xpu",
     "b70",
     "one card",
     "mtp",
     "draft int4 head",
     "batch-invariant",
     "xpu graph"
    ],
    "published_at": "2026-09-07",
    "featured_metric": {
     "value": 113.265,
     "unit": "tok/s",
     "label": "MTP depth 3 with the draft-only INT4 lm_head, one card, strict completions suite (center of two fresh servers, 113.627 / 112.904)",
     "scope": "Center of two fresh-server class-balanced medians over the strict fixed 12-prompt six-class suite through the completions API, 512-token cap, separate empty compile caches, 12/12 complete token arrays exact versus a same-configuration MTP0 oracle (64.332 / 64.338), canaries on every server, cache zero; one B70, R276 image, qwen3_5_mtp depth 3, draft-only INT4 lm_head, full decode-only XPU graph capture.",
     "evidence": "experiments/qwen35-9b-b70/data/2026-09-07-qwen35-9b-w4a16-matrix-result.json"
    },
    "benchmark_status": "Candidate (2026-09-07, campaign w1, one B70, R276 image): two fresh depth-3 servers 113.627 / 112.904 tok/s class-balanced median over tokens 1-100 on the strict 12-prompt completions suite, 12/12 vs each other and 12/12 vs the MTP0 oracle; two MTP0 servers 64.332 / 64.338 matched 12/12; canaries on every server; cache zero. Identity ladders (128 tokens, two passes): without speculation exact at every rung in both passes through 64 users (c16 746.7, c32 1184.0, c64 1268.4, all 64/64); with depth 3 exact through c16 (750.8), c32 32/32 warm (827.3), c64 61/64 (789.2). LocalMaxxing cmtrhoyl1000cps01o43bhl72 at 113.265. Two cards (w3): MTP0 97.586/97.535 and depth 3 172.271/172.321 tok/s, all gates 12/12, LocalMaxxing cmtrn9hoy001ops01qzd4axry at 172.296; ladders exact through 32 users without speculation (1836.9 tok/s) and 16 with it (1174.6), one flip at 64 users on two cards where one card held 64/64."
   },
   "contributors": [
    {
     "id": "b70-optimization-lab",
     "name": "neural.download lab",
     "kind": "lab",
     "profile": "https://github.com/steveseguin/b70-optimization-lab",
     "contribution": "Fixed-K two-tier W4A16 kernel strategy (built for the Qwen3.8 INT4 lane and reused here unchanged), lane composition on the R276 image and strict launchers, speculative and identity gating, packaging.",
     "status": "integrated",
     "validated_effect": "The row-invariant W4A16 kernel keeps every concurrency rung through 64 users byte-exact without speculation (1268.4 tok/s at 64 users, 64/64 in both passes) where the FP8 route on the same model flips from 16 users; MTP depth 3 with the draft-only INT4 head raises single-user decode from 64.3 to 113.6 tok/s, 12/12 exact against the oracle.",
     "evidence": "experiments/qwen35-9b-b70/data/2026-09-07-qwen35-9b-w4a16-matrix-result.json"
    },
    {
     "id": "redhatai-qwen35-9b-w4a16",
     "name": "RedHatAI",
     "kind": "external",
     "profile": "https://huggingface.co/RedHatAI/Qwen3.5-9B-quantized.w4a16",
     "contribution": "W4A16 quantization of Qwen3.5-9B and the MTP head checkpoint served here unchanged.",
     "status": "credited",
     "validated_effect": "The published tensors serve as-is through vLLM's compressed-tensors path, which selects the lab's wNa16 INT4 kernel with no config relabel.",
     "evidence": "repro/qwen35-9b-w4a16-b70/manifests/model-direct-redhatai-qwen35-9b-w4a16-a398088c.json"
    }
   ],
   "performance_profiles": [
    {
     "id": "http-w4a16-mtp0-graph-tp1-identity-qualified-aggregate-vs-concurrent-users",
     "label": "Qwen3.5-9B W4A16 one card, no speculation, XPU graph capture, identity-qualified aggregate decode vs concurrent users (c1-c64; warm pass)",
     "metric": "aggregate_decode",
     "unit": "tok/s",
     "x_metric": "concurrent_sequences",
     "x_label": "Concurrent HTTP users",
     "scope": "Campaign w1 (2026-09-07): one B70, R276 image, strict launcher env, FULL_DECODE_ONLY capture sizes 1-64, no speculation, max-model-len 256, max-num-seqs 64, max-num-batched-tokens 512, 128 returned raw token IDs per response on the small-context suite, two passes on one server, warm pass shown. Every rung is output-identity-qualified in both passes, including c64 (64/64 twice). Nothing is withheld on this profile.",
     "evidence": "experiments/qwen35-9b-b70/data/2026-09-07-qwen35-9b-w4a16-matrix-result.json",
     "points": [
      {
       "concurrent_sequences": 1,
       "value": 64.2,
       "samples": 1,
       "per_user_value": 64.2
      },
      {
       "concurrent_sequences": 2,
       "value": 124.0,
       "samples": 1,
       "per_user_value": 62.0
      },
      {
       "concurrent_sequences": 4,
       "value": 236.2,
       "samples": 1,
       "per_user_value": 59.05
      },
      {
       "concurrent_sequences": 8,
       "value": 437.7,
       "samples": 1,
       "per_user_value": 54.7125
      },
      {
       "concurrent_sequences": 16,
       "value": 746.7,
       "samples": 1,
       "per_user_value": 46.6688
      },
      {
       "concurrent_sequences": 32,
       "value": 1184.0,
       "samples": 1,
       "per_user_value": 37.0
      },
      {
       "concurrent_sequences": 64,
       "value": 1268.4,
       "samples": 1,
       "per_user_value": 19.8188
      }
     ]
    },
    {
     "id": "http-w4a16-mtp3-drafthead-graph-tp1-identity-qualified-aggregate-vs-concurrent-users",
     "label": "Qwen3.5-9B W4A16 one card, MTP depth 3 with the draft INT4 head, identity-qualified aggregate decode vs concurrent users (c1-c16; warm pass)",
     "metric": "aggregate_decode",
     "unit": "tok/s",
     "x_metric": "concurrent_sequences",
     "x_label": "Concurrent HTTP users",
     "scope": "As above with qwen3_5_mtp depth 3 and the draft-only INT4 lm_head. c1-c16 output-identity-qualified in both passes. Measured but withheld: c32 827.3 (32/32 warm, 31/32 cold) and c64 789.2 (61/64).",
     "evidence": "experiments/qwen35-9b-b70/data/2026-09-07-qwen35-9b-w4a16-matrix-result.json",
     "points": [
      {
       "concurrent_sequences": 1,
       "value": 116.5,
       "samples": 1,
       "per_user_value": 116.5
      },
      {
       "concurrent_sequences": 2,
       "value": 224.2,
       "samples": 1,
       "per_user_value": 112.1
      },
      {
       "concurrent_sequences": 4,
       "value": 371.3,
       "samples": 1,
       "per_user_value": 92.825
      },
      {
       "concurrent_sequences": 8,
       "value": 649.8,
       "samples": 1,
       "per_user_value": 81.225
      },
      {
       "concurrent_sequences": 16,
       "value": 750.8,
       "samples": 1,
       "per_user_value": 46.925
      }
     ]
    },
    {
     "id": "http-w4a16-mtp0-graph-tp2-identity-qualified-aggregate-vs-concurrent-users",
     "label": "Qwen3.5-9B W4A16 two cards (TP2), no speculation, identity-qualified aggregate decode vs concurrent users (c1-c32; warm pass)",
     "metric": "aggregate_decode",
     "unit": "tok/s",
     "x_metric": "concurrent_sequences",
     "x_label": "Concurrent HTTP users",
     "scope": "Campaign w3 (2026-09-07): two B70s, otherwise as the one-card no-speculation profile. c1-c32 output-identity-qualified in both passes. Measured but withheld: c64 2092.9 (63/64 in both passes); the one-card run of the same kernel held 64/64 there, so the residual flip comes from the cross-card reduction.",
     "evidence": "experiments/qwen35-9b-b70/data/2026-09-07-qwen35-9b-w4a16-matrix-result.json",
     "points": [
      {
       "concurrent_sequences": 1,
       "value": 97.5,
       "samples": 1,
       "per_user_value": 97.5
      },
      {
       "concurrent_sequences": 2,
       "value": 186.6,
       "samples": 1,
       "per_user_value": 93.3
      },
      {
       "concurrent_sequences": 4,
       "value": 356.1,
       "samples": 1,
       "per_user_value": 89.025
      },
      {
       "concurrent_sequences": 8,
       "value": 660.8,
       "samples": 1,
       "per_user_value": 82.6
      },
      {
       "concurrent_sequences": 16,
       "value": 1151.8,
       "samples": 1,
       "per_user_value": 71.9875
      },
      {
       "concurrent_sequences": 32,
       "value": 1836.9,
       "samples": 1,
       "per_user_value": 57.4031
      }
     ]
    },
    {
     "id": "http-w4a16-mtp3-drafthead-graph-tp2-identity-qualified-aggregate-vs-concurrent-users",
     "label": "Qwen3.5-9B W4A16 two cards (TP2), MTP depth 3 with the draft INT4 head, identity-qualified aggregate decode vs concurrent users (c1-c16; warm pass)",
     "metric": "aggregate_decode",
     "unit": "tok/s",
     "x_metric": "concurrent_sequences",
     "x_label": "Concurrent HTTP users",
     "scope": "As above with qwen3_5_mtp depth 3 and the draft-only INT4 lm_head. c1-c16 output-identity-qualified in both passes. Measured but withheld: c32 1350.9 (32/32 warm, 31/32 cold) and c64 1464.6 (60/64).",
     "evidence": "experiments/qwen35-9b-b70/data/2026-09-07-qwen35-9b-w4a16-matrix-result.json",
     "points": [
      {
       "concurrent_sequences": 1,
       "value": 176.1,
       "samples": 1,
       "per_user_value": 176.1
      },
      {
       "concurrent_sequences": 2,
       "value": 341.5,
       "samples": 1,
       "per_user_value": 170.75
      },
      {
       "concurrent_sequences": 4,
       "value": 567.3,
       "samples": 1,
       "per_user_value": 141.825
      },
      {
       "concurrent_sequences": 8,
       "value": 982.4,
       "samples": 1,
       "per_user_value": 122.8
      },
      {
       "concurrent_sequences": 16,
       "value": 1174.6,
       "samples": 1,
       "per_user_value": 73.4125
      }
     ]
    },
    {
     "id": "http-w4a16-mtp3-drafthead-decode-vs-active-context",
     "label": "Qwen3.5-9B W4A16 one card, MTP depth 3 with the draft INT4 head: decode over active context (2K-32K)",
     "metric": "decode",
     "unit": "tok/s",
     "x_label": "Actual prompt / active context tokens",
     "scope": "Campaign w4 (2026-09-07): one B70, one-slot vLLM HTTP completions at exactly 2K, 4K, 8K, 16K, 24K and 32K active context of unrepeated real content (technical prose, Python, structured documents; three requests per depth, median shown), 128 output tokens, max-model-len 33024, cache zero, canaries before and after; every answer matched the same-configuration MTP0 oracle (18/18). No value is interpolated.",
     "evidence": "experiments/qwen35-9b-b70/data/2026-09-07-qwen35-9b-w4a16-matrix-result.json",
     "points": [
      {
       "context_tokens": 2048,
       "value": 116.653,
       "samples": 3
      },
      {
       "context_tokens": 4096,
       "value": 145.691,
       "samples": 3
      },
      {
       "context_tokens": 8192,
       "value": 118.726,
       "samples": 3
      },
      {
       "context_tokens": 16384,
       "value": 153.456,
       "samples": 3
      },
      {
       "context_tokens": 24576,
       "value": 134.488,
       "samples": 3
      },
      {
       "context_tokens": 32768,
       "value": 89.507,
       "samples": 3
      }
     ]
    },
    {
     "id": "http-w4a16-mtp0-decode-vs-active-context",
     "label": "Qwen3.5-9B W4A16 one card, no speculation: decode over active context (2K-32K)",
     "metric": "decode",
     "unit": "tok/s",
     "x_label": "Actual prompt / active context tokens",
     "scope": "Campaign w4 MTP0 arm (the oracle for the depth-3 profile), same workload and shape.",
     "evidence": "experiments/qwen35-9b-b70/data/2026-09-07-qwen35-9b-w4a16-matrix-result.json",
     "points": [
      {
       "context_tokens": 2048,
       "value": 63.959,
       "samples": 3
      },
      {
       "context_tokens": 4096,
       "value": 63.02,
       "samples": 3
      },
      {
       "context_tokens": 8192,
       "value": 62.008,
       "samples": 3
      },
      {
       "context_tokens": 16384,
       "value": 60.108,
       "samples": 3
      },
      {
       "context_tokens": 24576,
       "value": 58.446,
       "samples": 3
      },
      {
       "context_tokens": 32768,
       "value": 56.886,
       "samples": 3
      }
     ]
    },
    {
     "id": "http-w4a16-dynamic-draft-schedule-fullgraph-tp1-aggregate-vs-concurrent-users",
     "label": "Qwen3.5-9B W4A16 one card, draft depth scheduled by batch size (3 to 8 users, 1 to 16, none above), full decode graphs per depth, draft-state catch-up, aggregate decode vs concurrent users (c1-c64; warm pass)",
     "metric": "aggregate_decode",
     "unit": "tok/s",
     "x_metric": "concurrent_sequences",
     "x_label": "Concurrent HTTP users",
     "scope": "Campaign cudynm1 (2026-09-11, repeated as cudynm1r on the same boot): one B70, R276 image plus the three overlays in repro/qwen35-9b-w4a16-b70/docker, FULL_DECODE_ONLY capture sizes 1-64 with one full decode graph per scheduled draft depth, schedule [[1,8,3],[9,16,1],[17,64,0]], max-model-len 256, max-num-seqs 64, max-num-batched-tokens 512, 128 returned raw token IDs per response on the small-context suite, two passes on one server, warm pass shown. Identity by the two-run rule: exact against the sequential oracle at every rung through 32 users in both passes of both runs; at 64 users near-exact (63-64/64), the same band as the no-speculation server on this route, so that point is a measured rate, not an identity-qualified one. One user on the strict suite: 110.69/110.60 and 110.65/110.60 tok/s, G1/G2/G3 12/12 in both runs; the 2K-32K real-content ladder is 18/18 exact.",
     "evidence": "experiments/qwen35-9b-b70/data/2026-09-10-qwen35-9b-w4a16-dynamic-schedule-ladders.json",
     "points": [
      {
       "concurrent_sequences": 1,
       "value": 114.0,
       "samples": 1,
       "per_user_value": 114.0
      },
      {
       "concurrent_sequences": 2,
       "value": 218.0,
       "samples": 1,
       "per_user_value": 109.0
      },
      {
       "concurrent_sequences": 4,
       "value": 358.0,
       "samples": 1,
       "per_user_value": 89.5
      },
      {
       "concurrent_sequences": 8,
       "value": 623.0,
       "samples": 1,
       "per_user_value": 77.9
      },
      {
       "concurrent_sequences": 16,
       "value": 851.0,
       "samples": 1,
       "per_user_value": 53.2
      },
      {
       "concurrent_sequences": 32,
       "value": 1113.0,
       "samples": 1,
       "per_user_value": 34.8
      },
      {
       "concurrent_sequences": 64,
       "value": 1184.0,
       "samples": 1,
       "per_user_value": 18.5
      }
     ]
    },
    {
     "id": "http-w4a16-r293-classpad-mtp0-graph-tp1-identity-qualified-aggregate-vs-concurrent-users",
     "label": "Qwen3.5-9B W4A16 one card, no speculation, CLASSPAD=1 (R293): identity-qualified aggregate decode vs concurrent users",
     "metric": "aggregate_decode",
     "unit": "tok/s",
     "x_metric": "concurrent_sequences",
     "x_label": "Concurrent HTTP users",
     "scope": "Chain s1 (2026-09-11): one B70, R293 image with VLLM_XPU_FP16_LINEAR_CLASSPAD=1 (verified in the container and by the op's census line), strict launcher env, FULL_DECODE_ONLY capture sizes 1-64, no speculation, max-model-len 256, max-num-seqs 64, max-num-batched-tokens 512, 128 returned raw token IDs per response on the small-context suite, four passes, pass 2 shown. Points are rungs output-identity-qualified in the shown pass (c1-c24 and c64). Measured but withheld: c32 1208.7 (127/128). With max-num-seqs 128 and max-num-batched-tokens 1024 (s6) c64/c96/c128 are exact at 1641.6 / 1852.8 / 1955.3.",
     "evidence": "experiments/qwen35-9b-b70/data/2026-09-11-qwen35-9b-r293-classpad.json",
     "points": [
      {
       "concurrent_sequences": 1,
       "value": 61.6,
       "samples": 1,
       "per_user_value": 61.6
      },
      {
       "concurrent_sequences": 2,
       "value": 119.2,
       "samples": 1,
       "per_user_value": 59.6
      },
      {
       "concurrent_sequences": 4,
       "value": 229.0,
       "samples": 1,
       "per_user_value": 57.25
      },
      {
       "concurrent_sequences": 8,
       "value": 429.5,
       "samples": 1,
       "per_user_value": 53.6875
      },
      {
       "concurrent_sequences": 12,
       "value": 579.4,
       "samples": 1,
       "per_user_value": 48.2833
      },
      {
       "concurrent_sequences": 16,
       "value": 743.8,
       "samples": 1,
       "per_user_value": 46.4875
      },
      {
       "concurrent_sequences": 20,
       "value": 866.6,
       "samples": 1,
       "per_user_value": 43.33
      },
      {
       "concurrent_sequences": 24,
       "value": 992.7,
       "samples": 1,
       "per_user_value": 41.3625
      },
      {
       "concurrent_sequences": 64,
       "value": 1643.6,
       "samples": 1,
       "per_user_value": 25.6812
      }
     ]
    },
    {
     "id": "http-w4a16-r293-classpad-mtp0-graph-tp2-identity-qualified-aggregate-vs-concurrent-users",
     "label": "Qwen3.5-9B W4A16 two cards, no speculation, CLASSPAD=1 (R293): identity-qualified aggregate decode vs concurrent users",
     "metric": "aggregate_decode",
     "unit": "tok/s",
     "x_metric": "concurrent_sequences",
     "x_label": "Concurrent HTTP users",
     "scope": "Chain s2 (2026-09-11): two B70s (TP2), otherwise as the one-card R293 profile. Points are rungs output-identity-qualified in the shown pass (c1-c32). Measured but withheld: c64 2614.1 (255/256). With max-num-seqs 128 and max-num-batched-tokens 1024 (s7): c64 2614.2 (256/256), c96 2978.7 (383/384), c128 3227.4 (512/512). With the 5 ms admission stagger on this lane's tie-site suite (s5) c64 is 1280/1280 over twenty passes at 2557.1, harness-certified.",
     "evidence": "experiments/qwen35-9b-b70/data/2026-09-11-qwen35-9b-r293-classpad.json",
     "points": [
      {
       "concurrent_sequences": 1,
       "value": 93.4,
       "samples": 1,
       "per_user_value": 93.4
      },
      {
       "concurrent_sequences": 2,
       "value": 179.2,
       "samples": 1,
       "per_user_value": 89.6
      },
      {
       "concurrent_sequences": 4,
       "value": 343.7,
       "samples": 1,
       "per_user_value": 85.925
      },
      {
       "concurrent_sequences": 8,
       "value": 643.3,
       "samples": 1,
       "per_user_value": 80.4125
      },
      {
       "concurrent_sequences": 12,
       "value": 881.1,
       "samples": 1,
       "per_user_value": 73.425
      },
      {
       "concurrent_sequences": 16,
       "value": 1137.0,
       "samples": 1,
       "per_user_value": 71.0625
      },
      {
       "concurrent_sequences": 20,
       "value": 1327.2,
       "samples": 1,
       "per_user_value": 66.36
      },
      {
       "concurrent_sequences": 24,
       "value": 1523.0,
       "samples": 1,
       "per_user_value": 63.4583
      },
      {
       "concurrent_sequences": 32,
       "value": 1855.6,
       "samples": 1,
       "per_user_value": 57.9875
      }
     ]
    },
    {
     "id": "http-w4a16-r293-dynamic-draft-schedule-tp1-aggregate-vs-concurrent-users",
     "label": "Qwen3.5-9B W4A16 one card, scheduled drafts (depth 3 to 8 users, 1 to 16, none above) on R293 with CLASSPAD=1: aggregate decode vs concurrent users (exact rungs)",
     "metric": "aggregate_decode",
     "unit": "tok/s",
     "x_metric": "concurrent_sequences",
     "x_label": "Concurrent HTTP users",
     "scope": "dyn293 (2026-09-11): one B70, the scheduled-draft overlays rebuilt on the R293 image, VLLM_XPU_FP16_LINEAR_CLASSPAD=1, strict launcher env, full decode graphs per scheduled K, max-model-len 256, max-num-seqs 64, 128 returned raw token IDs per response on the small-context suite, four passes, pass 2 shown. Points are rungs output-identity-qualified in the shown pass (c1-c16). Measured but withheld: c32 1192.2 (120/128), c64 1630.6 (255/256; the no-speculation R293 server in the same arm reads 1641.1). Same-host R276 baseline (dyn276): c16 942.7, c32 1153.1 (115/128), c64 1243.4 (249/256).",
     "evidence": "experiments/qwen35-9b-b70/data/2026-09-11-qwen35-9b-r293-classpad.json",
     "points": [
      {
       "concurrent_sequences": 1,
       "value": 113.9,
       "samples": 1,
       "per_user_value": 113.9
      },
      {
       "concurrent_sequences": 2,
       "value": 218.8,
       "samples": 1,
       "per_user_value": 109.4
      },
      {
       "concurrent_sequences": 4,
       "value": 366.3,
       "samples": 1,
       "per_user_value": 91.575
      },
      {
       "concurrent_sequences": 8,
       "value": 652.1,
       "samples": 1,
       "per_user_value": 81.5125
      },
      {
       "concurrent_sequences": 16,
       "value": 960.1,
       "samples": 1,
       "per_user_value": 60.0063
      }
     ]
    }
   ],
   "hardware": {
    "accelerator": "Intel Arc Pro B70 32 GiB",
    "cards": 1,
    "host_ram_min_bytes": 16106127360,
    "host_ram_plus_swap_min_bytes": 21474836480,
    "model_weight_bytes": 11456765607,
    "note": "One card is the default profile; the packet also ships a measured two-card (TP2) profile at 172.3 tok/s. Both profiles are generated from the same recipe launcher."
   },
   "model": {
    "repository": "RedHatAI/Qwen3.5-9B-quantized.w4a16",
    "revision": "a398088c4228b0ae0c8c78df88fd1e4bf445f068",
    "manifest": "repro/qwen35-9b-w4a16-b70/manifests/model-direct-redhatai-qwen35-9b-w4a16-a398088c.json",
    "note": "Served exactly as published; the launcher verifies every LFS file against the manifest before the container starts. No config relabel is needed: compressed-tensors selects the wNa16 INT4 kernel directly."
   },
   "runtime": {
    "kind": "container",
    "image": "ghcr.io/steveseguin/vllm-openai-xpu-qwen38-int4@sha256:40d46730c9a24f9396cc67c0e5578dd80d11dfae7a4d23a55f97620140a0b3e6",
    "image_id_validated": "sha256:40d46730c9a24f9396cc67c0e5578dd80d11dfae7a4d23a55f97620140a0b3e6",
    "local_tag": "neural-download/vllm-openai-xpu:qwen38-int4-fp16-linear-classpad-cheapest-r293",
    "engine": "vLLM 0.27.2rc1.dev77+gac7509e2b (XPU) with the lab kernel library (vllm-xpu-kernels 1e90ffa6 + patches, oneDNN 0e2a5bfe + patches)",
    "launcher": "repro/qwen35-9b-fp8-b70/scripts/run-qwen35-9b-fp8-server.sh",
    "base": "R276 (sync-free grouped GDN branch) + R290-R293: the class-consistent FP16 linear mode (VLLM_XPU_FP16_LINEAR_CLASSPAD; CLASSPAD=0 runs the R276 code path unchanged). The scheduled-draft overlays in repro/qwen35-9b-w4a16-b70/docker/ build on the R276 digest and are not combined with R293."
   },
   "project_patches": {
    "required": true,
    "items": [
     "experiments/qwen38-27b-b70/docker/r213b-w4a16-determinism-pad-op.py",
     "experiments/qwen38-27b-b70/docker/r224-fp16-linear-rowchunk.py",
     "experiments/qwen38-27b-b70/docker/r228-gdn-spec-group.py",
     "experiments/qwen38-27b-b70/docker/r256-draft-int4-head-fallback.py",
     "experiments/qwen38-27b-b70/patches/onednn-qwen38-w4a16-fixed-k-two-tier-r221-20260905.patch",
     "experiments/qwen38-27b-b70/patches/onednn-qwen38-w4a16-strategy-override-dump-r220-20260905.patch",
     "experiments/qwen38-27b-b70/patches/onednn-qwen38-w8a16-c-default-align-r137b-20260902.patch",
     "experiments/qwen38-27b-b70/patches/onednn-qwen38-w8a16-fixed-k-align16-r137a-20260902.patch",
     "repro/qwen35-9b-w4a16-b70/docker/r276-dynamic-mamba-alloc.patch",
     "repro/qwen35-9b-w4a16-b70/docker/r276-dynsd-fullgraph.patch",
     "repro/qwen35-9b-w4a16-b70/docker/r276-dynamic-mamba-alloc.Dockerfile",
     "repro/qwen35-9b-w4a16-b70/docker/r276-dynsd-fullgraph.Dockerfile",
     "repro/qwen35-9b-w4a16-b70/docker/r276-dynsd-catchup.Dockerfile",
     "repro/qwen35-9b-w4a16-b70/docker/r276-dynsd-catchup.patch"
    ],
    "note": "The image is the Qwen3.8 INT4 lane's R276 image; its overlays (determinism pad op inert here, FP16 row-chunk linears, grouped GDN speculative rows, draft-head fallback unused, sync-free GDN grouping) ship inside the container. No 9B-specific patch. The scheduled-draft configuration (2026-09-10) adds three pure-Python overlays on the same image, built locally from the two COPY-only Dockerfiles in repro/qwen35-9b-w4a16-b70/docker: dynamic-Mamba-allocation, full decode graphs per scheduled draft depth, and draft-state catch-up on zero-draft steps. They change nothing for the static profiles; the dynamic launcher pins them by file content."
   },
   "commands": {
    "preflight": "docker pull ghcr.io/steveseguin/vllm-openai-xpu-qwen38-int4@sha256:40d46730c9a24f9396cc67c0e5578dd80d11dfae7a4d23a55f97620140a0b3e6 && docker tag ghcr.io/steveseguin/vllm-openai-xpu-qwen38-int4@sha256:40d46730c9a24f9396cc67c0e5578dd80d11dfae7a4d23a55f97620140a0b3e6 neural-download/vllm-openai-xpu:qwen38-int4-fp16-linear-classpad-cheapest-r293",
    "launch": "MODEL_DIR=/models/Qwen3.5-9B-quantized.w4a16 VLLM_CACHE_DIR=/tmp/qwen35-w4a16-cache MTP_DEPTH=3 CLASSPAD=0 repro/qwen35-9b-w4a16-b70/scripts/run-qwen35-9b-w4a16-server.sh",
    "health": "curl -fsS http://127.0.0.1:18131/health",
    "benchmark": "OUT_DIR=<new dir> BASE_URL=http://127.0.0.1:18131 MODEL_NAME=qwen35-9b-w4a16-mtp3 repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/bench-w8a16-mtp1-strict.sh",
    "stop": "docker stop qwen35-9b-w4a16-mtp3"
   },
   "dependencies": [
    "experiments/qwen35-9b-b70/data/2026-09-07-qwen35-9b-fp8-matrix-result.json",
    "experiments/qwen35-9b-b70/data/2026-09-07-qwen35-9b-w4a16-c128-identity-ladder.json",
    "experiments/qwen35-9b-b70/data/2026-09-07-qwen35-9b-w4a16-depth-sweep.json",
    "experiments/qwen35-9b-b70/data/2026-09-07-qwen35-9b-w4a16-matrix-result.json",
    "experiments/qwen35-9b-b70/notes/2026-09-07-qwen35-9b-fp8-one-card-quick-lane.md",
    "experiments/qwen35-9b-b70/scripts/run-20260907-qwen35-9b-campaign-v2.sh",
    "experiments/qwen35-9b-b70/scripts/run-20260907-qwen35-campaign.sh",
    "experiments/qwen38-27b-b70/data/2026-08-25-qwen38-q4km-tp2-http-smallctx-suite.json",
    "experiments/qwen38-27b-b70/docker/r213b-w4a16-determinism-pad-op.py",
    "experiments/qwen38-27b-b70/docker/r224-fp16-linear-rowchunk.py",
    "experiments/qwen38-27b-b70/docker/r228-gdn-spec-group.py",
    "experiments/qwen38-27b-b70/docker/r256-draft-int4-head-fallback.py",
    "experiments/qwen38-27b-b70/patches/onednn-qwen38-w4a16-fixed-k-two-tier-r221-20260905.patch",
    "experiments/qwen38-27b-b70/patches/onednn-qwen38-w4a16-strategy-override-dump-r220-20260905.patch",
    "experiments/qwen38-27b-b70/patches/onednn-qwen38-w8a16-c-default-align-r137b-20260902.patch",
    "experiments/qwen38-27b-b70/patches/onednn-qwen38-w8a16-fixed-k-align16-r137a-20260902.patch",
    "packages/qwen35-9b-w4a16-b70/compose.yaml",
    "packages/qwen35-9b-w4a16-b70/env/one-gpu.env",
    "packages/qwen35-9b-w4a16-b70/env/two-gpu.env",
    "packages/qwen35-9b-w4a16-b70/scripts/download-model.sh",
    "packages/qwen35-9b-w4a16-b70/scripts/preflight.sh",
    "packages/qwen35-9b-w4a16-b70/scripts/render-compose.sh",
    "packages/qwen35-9b-w4a16-b70/scripts/smoke-test.sh",
    "packages/qwen35-9b-w4a16-b70/scripts/verify.sh",
    "repro/gemma4-26b-a4b-q8-b70/realistic-suite-v1.json",
    "repro/qwen35-9b-fp8-b70/README.md",
    "repro/qwen35-9b-fp8-b70/manifests/model-direct-redhatai-qwen35-9b-fp8-dynamic-790f0576.json",
    "repro/qwen35-9b-fp8-b70/scripts/run-qwen35-9b-fp8-server.sh",
    "repro/qwen35-9b-w4a16-b70/README.md",
    "repro/qwen35-9b-w4a16-b70/manifests/model-direct-redhatai-qwen35-9b-w4a16-a398088c.json",
    "repro/qwen35-9b-w4a16-b70/scripts/run-qwen35-9b-w4a16-server.sh",
    "repro/qwen36-27b-autoround-int4-b70/realistic-suite-v1.json",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/bench-w8a16-mtp1-strict.sh",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/run-w8a16-mtp0-strict-server.sh",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/run-w8a16-mtp1-strict-server.sh",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/verify-model-direct.sh",
    "scripts/bench-openai-concurrency-oracle.py",
    "scripts/bench-openai-realistic-suite.py",
    "scripts/compare-strict-attempt-outputs.py",
    "scripts/neural-download-canaries.py",
    "tools/check-container-packet.py",
    "tools/container-packet/download-model.sh",
    "tools/container-packet/preflight.sh",
    "tools/container-packet/smoke-test.sh",
    "tools/container-packet/verify.sh",
    "tools/render-container-compose.py",
    "tools/render-container-packet.sh",
    "repro/qwen35-9b-w4a16-b70/scripts/run-qwen35-9b-w4a16-dynamic-server.sh",
    "repro/qwen35-9b-w4a16-b70/docker/r276-dynamic-mamba-alloc.Dockerfile",
    "repro/qwen35-9b-w4a16-b70/docker/r276-dynamic-mamba-alloc.patch",
    "repro/qwen35-9b-w4a16-b70/docker/r276-dynsd-fullgraph.Dockerfile",
    "repro/qwen35-9b-w4a16-b70/docker/r276-dynsd-fullgraph.patch",
    "experiments/qwen35-9b-b70/data/qwen35-9b-w4a16-tp1-mtp3-dynamic-fullgraph-20260910-fgdynm1-strict-result.json",
    "experiments/qwen35-9b-b70/data/qwen35-9b-w4a16-tp1-mtp3-dynamic-fullgraph-20260910-fgdynm1r-strict-result.json",
    "experiments/qwen35-9b-b70/data/2026-09-10-qwen35-9b-w4a16-dynamic-schedule-ladders.json",
    "experiments/qwen35-9b-b70/notes/2026-09-10-one-server-for-every-batch-size.md",
    "repro/qwen35-9b-w4a16-b70/docker/r276-dynsd-catchup.Dockerfile",
    "repro/qwen35-9b-w4a16-b70/docker/r276-dynsd-catchup.patch",
    "experiments/qwen35-9b-b70/data/qwen35-9b-w4a16-tp1-mtp3-dynamic-catchup-20260911-cudynm1-strict-result.json",
    "experiments/qwen35-9b-b70/data/qwen35-9b-w4a16-tp1-mtp3-dynamic-catchup-20260911-cudynm1r-strict-result.json",
    "experiments/qwen35-9b-b70/notes/2026-09-10-prereg-draft-state-catch-up-at-schedule-transitions.md",
    "experiments/qwen38-27b-b70/docker/Dockerfile.fp16-linear-classpad-r290",
    "experiments/qwen38-27b-b70/docker/r290-fp16-linear-classpad.py",
    "experiments/qwen38-27b-b70/docker/Dockerfile.fp16-linear-classpad-verified-r291",
    "experiments/qwen38-27b-b70/docker/r291-fp16-linear-classpad-verified.py",
    "experiments/qwen38-27b-b70/docker/Dockerfile.fp16-linear-classpad-nozero-r292",
    "experiments/qwen38-27b-b70/docker/r292-fp16-linear-classpad-nozero.py",
    "experiments/qwen38-27b-b70/docker/Dockerfile.fp16-linear-classpad-cheapest-r293",
    "experiments/qwen38-27b-b70/docker/r293-fp16-linear-classpad-cheapest.py",
    "experiments/qwen35-4b-b70/probes/fp16-linear-mclass-census.py",
    "experiments/qwen35-4b-b70/probes/fp16-linear-mclass-census-extended.py",
    "experiments/qwen35-4b-b70/probes/fp16-linear-classmap.py",
    "experiments/qwen35-4b-b70/probes/census-small-shapes.py",
    "experiments/qwen35-4b-b70/probes/validate-r290-classpad.py",
    "experiments/qwen35-4b-b70/probes/validate-classpad-tp1-shapes.py",
    "experiments/qwen35-4b-b70/probes/validate-r292-stale-pad.py",
    "experiments/qwen35-4b-b70/notes/2026-09-09-the-fp16-linear-chunk-is-a-throughput-tax.md",
    "experiments/qwen35-4b-b70/notes/2026-09-11-r293-class-consistent-fp16-linear-on-the-server.md",
    "repro/qwen38-27b-autoround-int4-b70/scripts/publish-r293-image-ghcr.sh",
    "experiments/qwen35-9b-b70/scripts/run-20260911-qwen35-9b-classpad-chain.sh",
    "experiments/qwen35-9b-b70/data/2026-09-11-qwen35-9b-r293-classpad.json",
    "experiments/qwen35-9b-b70/notes/2026-09-11-r293-on-the-9b.md",
    "repro/qwen35-9b-w4a16-b70/docker/r293-dynamic-mamba-alloc.Dockerfile",
    "repro/qwen35-9b-w4a16-b70/docker/r293-dynsd-fullgraph.Dockerfile",
    "repro/qwen35-9b-w4a16-b70/docker/r293-dynsd-catchup.Dockerfile",
    "experiments/qwen35-9b-b70/scripts/run-20260911-qwen35-9b-dynsd-on-r293-chain.sh"
   ],
   "evidence": [
    "experiments/qwen35-9b-b70/data/2026-09-07-qwen35-9b-w4a16-c128-identity-ladder.json",
    "experiments/qwen35-9b-b70/data/2026-09-07-qwen35-9b-w4a16-depth-sweep.json",
    "experiments/qwen35-9b-b70/data/2026-09-07-qwen35-9b-w4a16-matrix-result.json",
    "experiments/qwen35-9b-b70/data/2026-09-10-qwen35-9b-w4a16-dynamic-schedule-ladders.json",
    "experiments/qwen35-9b-b70/data/qwen35-9b-w4a16-tp1-mtp3-dynamic-fullgraph-20260910-fgdynm1-strict-result.json",
    "experiments/qwen35-9b-b70/data/qwen35-9b-w4a16-tp1-mtp3-dynamic-fullgraph-20260910-fgdynm1r-strict-result.json",
    "experiments/qwen35-9b-b70/data/qwen35-9b-w4a16-tp1-mtp3-dynamic-catchup-20260911-cudynm1-strict-result.json",
    "experiments/qwen35-9b-b70/data/qwen35-9b-w4a16-tp1-mtp3-dynamic-catchup-20260911-cudynm1r-strict-result.json"
   ],
   "missing": [
    "clean-host replay"
   ],
   "container_packet": {
    "level": 2,
    "compose": "packages/qwen35-9b-w4a16-b70/compose.yaml",
    "generated": true,
    "generator": "packages/qwen35-9b-w4a16-b70/scripts/render-compose.sh",
    "generated_from": "repro/qwen35-9b-w4a16-b70/scripts/run-qwen35-9b-w4a16-server.sh",
    "ci_check": "tools/check-container-packet.py",
    "profiles": {
     "one-gpu": {
      "cards": 1,
      "tensor_parallel_size": 1
     },
     "two-gpu": {
      "cards": 2,
      "tensor_parallel_size": 2
     }
    },
    "env_files": [
     "packages/qwen35-9b-w4a16-b70/env/one-gpu.env",
     "packages/qwen35-9b-w4a16-b70/env/two-gpu.env"
    ],
    "scripts": {
     "preflight": "packages/qwen35-9b-w4a16-b70/scripts/preflight.sh",
     "download": "packages/qwen35-9b-w4a16-b70/scripts/download-model.sh",
     "verify": "packages/qwen35-9b-w4a16-b70/scripts/verify.sh",
     "smoke_test": "packages/qwen35-9b-w4a16-b70/scripts/smoke-test.sh"
    },
    "note": "compose.yaml is rendered from the argv the recipe launcher hands to docker run, so it carries the measured container's 73 environment variables and full serve command verbatim rather than a hand-copied approximation. The model is mounted read-only, the image is pinned by digest, and the port is bound to loopback; CI fails the build if any of that stops holding.",
    "shared_implementation": "tools/container-packet/"
   }
  },
  {
   "manifest": "packages/qwen38-27b-256k-vision-mtp-b70/package.json",
   "format": "b70-model-package-v1",
   "id": "qwen38-27b-256k-vision-mtp-b70",
   "name": "Qwen3.8 27B 256K + vision + MTP draft on one Intel Arc Pro B70",
   "status": "candidate",
   "audience": "expert",
   "guide": "repro/qwen38-27b-256k-vision-mtp-b70/README.md",
   "clean_host_tested": false,
   "library": {
    "model_family": "Qwen3.8",
    "publisher": "Alibaba / Qwen (GGUF by unsloth)",
    "variant": "27B flagship package",
    "summary": "Alibaba's Qwen3.8 27B on one Arc Pro B70 with its full 262K-token context, image input, and draft head all resident at once. A capability-first fit, not a speed record.",
    "quantization": "UD-Q5_K_S (shipped) / UD-Q4_K_XL (alternative)",
    "runtime_label": "llama.cpp SYCL",
    "operating_systems": [
     "Linux"
    ],
    "delivery": [
     "native"
    ],
    "modalities": [
     "text",
     "vision"
    ],
    "use_cases": [
     "long context",
     "vision",
     "general",
     "coding"
    ],
    "tags": [
     "one card",
     "256k context",
     "vision",
     "speculative decode",
     "q8_0 kv",
     "stock upstream"
    ],
    "published_at": "2026-08-22",
    "featured_metric": {
     "value": 26.668277,
     "unit": "tok/s",
     "label": "decode (MTP-assisted, full package resident)",
     "scope": "Conventional 99-interval median, cold 12-prompt suite, up to 512-token responses, cache-zero, at 262144 configured capacity with q8_0 K/V, vision mmproj, and MTP draft loaded; active prompts were 48\u201378 tokens and the second fresh-server run measured 26.640510. Speculation-assisted and labeled as such; target-only tg128 at depth 0 is 22.64 tok/s (raw engine).",
     "evidence": "repro/qwen38-27b-256k-vision-mtp-b70/README.md"
    }
   },
   "contributors": [
    {
     "id": "b70-optimization-lab",
     "name": "neural.download lab",
     "kind": "lab",
     "profile": "https://github.com/steveseguin/b70-optimization-lab",
     "contribution": "Package design, KV budget arithmetic, hardware fit-off, vision smoke, operating points for both quants, canary battery.",
     "status": "integrated",
     "validated_effect": "Fit-off on hardware overturned the conservative paper margin: Q5_K_S completes the suite at 262144 with 2.86 GiB free; Q4_K_XL measures 27.510236/27.493910 tok/s (+3.2%) at the identical config.",
     "evidence": "experiments/qwen38-27b-b70/data/2026-08-22-neural-download-firstwave-baselines.json"
    },
    {
     "id": "unsloth",
     "name": "unsloth",
     "kind": "external",
     "profile": "https://huggingface.co/unsloth",
     "contribution": "Published the pinned GGUF quantizations, vision projector, and MTP draft artifacts used verbatim.",
     "status": "integrated",
     "validated_effect": "Artifact provider; no runtime delta adopted.",
     "evidence": "repro/qwen38-27b-256k-vision-mtp-b70/README.md"
    }
   ],
   "performance_profiles": [
    {
     "id": "decode-vs-context-depth",
     "label": "Raw decode over existing context depth",
     "metric": "decode",
     "unit": "tok/s",
     "x_label": "Existing context depth before tg128",
     "scope": "llama-bench raw engine rates, TARGET-ONLY Q5_K_S with q8_0 KV (no draft; llama-bench has no speculation). The MTP-assisted serving rate above is the package headline. The directly measured zero-depth point remains in the linked raw evidence; no missing depth is interpolated.",
     "evidence": "repro/qwen38-27b-256k-vision-mtp-b70/qwen38-27b-q5ks-flagship.sweep.json",
     "points": [
      {
       "context_tokens": 2048,
       "value": 21.013731,
       "samples": 5
      },
      {
       "context_tokens": 4096,
       "value": 19.763244,
       "samples": 5
      },
      {
       "context_tokens": 8192,
       "value": 17.627474,
       "samples": 5
      },
      {
       "context_tokens": 16384,
       "value": 14.139247,
       "samples": 5
      },
      {
       "context_tokens": 24576,
       "value": 11.904055,
       "samples": 5
      },
      {
       "context_tokens": 32768,
       "value": 10.32272,
       "samples": 5
      }
     ]
    },
    {
     "id": "prefill-vs-context-depth",
     "label": "Raw pp2048 over existing context depth",
     "metric": "prefill",
     "unit": "tok/s",
     "x_label": "Existing context depth before pp2048",
     "scope": "llama-bench raw engine rates, TARGET-ONLY Q5_K_S with q8_0 KV (no draft; llama-bench has no speculation). The MTP-assisted serving rate above is the package headline. The directly measured zero-depth point remains in the linked raw evidence; no missing depth is interpolated.",
     "evidence": "repro/qwen38-27b-256k-vision-mtp-b70/qwen38-27b-q5ks-flagship.sweep.json",
     "points": [
      {
       "context_tokens": 2048,
       "value": 831.413858,
       "samples": 5
      },
      {
       "context_tokens": 4096,
       "value": 822.288881,
       "samples": 5
      },
      {
       "context_tokens": 8192,
       "value": 807.562129,
       "samples": 5
      },
      {
       "context_tokens": 16384,
       "value": 760.357526,
       "samples": 5
      },
      {
       "context_tokens": 24576,
       "value": 750.98526,
       "samples": 5
      },
      {
       "context_tokens": 32768,
       "value": 678.867511,
       "samples": 5
      }
     ]
    }
   ],
   "known_limitations": [
    "256K KV requires q8_0 K/V on one card; f16 KV limits context to roughly 96K at these quants.",
    "MTP-assisted rate varies with content (p10 24.08 vs median 26.67); target-only numbers published alongside.",
    "Vision verified by objective two-question smoke (dominant color, corner color) on a deterministic test image; broader multimodal quality untested.",
    "Above ~1024-token prompts this package relies on multi-chunk prefill paths that the lab's separate vLLM lane found corruption-prone; the llama.cpp path here passed its canaries but has not had equivalent deep validation at depth."
   ],
   "hardware": {
    "accelerator": "Intel Arc Pro B70 32 GiB",
    "cards": 1,
    "host_ram_min_bytes": 34359738368,
    "host_ram_plus_swap_min_bytes": 34359738368,
    "model_weight_bytes": 18665753504
   },
   "model": {
    "repository": "unsloth/Qwen3.8-27B-GGUF",
    "revision": "4ca720788d1e01f1bff70c033e0d0028fd02e502",
    "manifest": "repro/qwen38-27b-256k-vision-mtp-b70/model-manifest.json"
   },
   "runtime": {
    "kind": "native",
    "project": "ggml-org/llama.cpp",
    "revision": "9fee29e9435f865ec0b811a783a6471a136d9317",
    "build": "cmake -G Ninja -B build-sycl-aot-bmg-g31 -DCMAKE_BUILD_TYPE=Release -DCMAKE_CXX_COMPILER=icpx -DCMAKE_C_COMPILER=icx -DGGML_SYCL=ON -DGGML_SYCL_F16=ON -DGGML_SYCL_DEVICE_ARCH=bmg_g31 -DGGML_SYCL_MAX_PARALLEL_LINK_JOBS=32 -DLLAMA_CURL=OFF && cmake --build build-sycl-aot-bmg-g31 --target llama-server llama-bench -j 24"
   },
   "project_patches": {
    "required": false,
    "items": []
   },
   "commands": {
    "preflight": "python3 scripts/verify-neural-download-model.py repro/qwen38-27b-256k-vision-mtp-b70/model-manifest.json \"$MODEL_DIR\" && source /opt/intel/oneapi/setvars.sh --force && export ONEAPI_DEVICE_SELECTOR=level_zero:0",
    "launch": "BUILD_DIR/bin/llama-server --model $MODEL_DIR/Qwen3.8-27B-UD-Q5_K_S.gguf --alias qwen38pkg --mmproj $MODEL_DIR/mmproj-F16.gguf --model-draft $MODEL_DIR/MTP/mtp-Qwen3.8-27B-Q4_0.gguf --device-draft SYCL0 --gpu-layers-draft 99 --ctx-size 262144 --cache-type-k q8_0 --cache-type-v q8_0 --cache-type-k-draft q8_0 --cache-type-v-draft q8_0 --device SYCL0 --gpu-layers 99 --flash-attn auto --parallel 1 --cache-ram 0 --ctx-checkpoints 0 --fit off --metrics --no-webui --host 127.0.0.1 --port 18100",
    "health": "curl -fsS http://127.0.0.1:18100/health",
    "benchmark": "python3 scripts/bench-openai-realistic-suite.py --base-url http://127.0.0.1:18100 --model qwen38pkg --api-mode completions --suite repro/qwen36-27b-autoround-int4-b70/realistic-suite-v1.json --max-tokens 512 --metric-tokens 100 --seed 1 --request-extra-json '{\"cache_prompt\":false,\"seed\":42,\"temperature\":0}' --out bench.json",
    "stop": "pkill -x llama-server"
   },
   "dependencies": [
    "repro/qwen38-27b-256k-vision-mtp-b70/README.md",
    "repro/gemma4-26b-a4b-q8-b70/realistic-suite-v1.json",
    "repro/qwen38-27b-256k-vision-mtp-b70/model-manifest.json",
    "repro/qwen38-27b-256k-vision-mtp-b70/qwen38-27b-q5ks-flagship.sweep.json",
    "repro/qwen38-27b-256k-vision-mtp-b70/qwen38-27b-q5ks-flagship.meta.json",
    "repro/qwen38-27b-256k-vision-mtp-b70/depth-sweep.svg",
    "scripts/verify-neural-download-model.py",
    "docs/neural-download-packet-standard.md",
    "experiments/qwen38-27b-b70/data/2026-08-22-neural-download-firstwave-baselines.json",
    "scripts/bench-openai-realistic-suite.py",
    "repro/qwen36-27b-autoround-int4-b70/realistic-suite-v1.json"
   ],
   "missing": [
    "tested clean-host platform installation",
    "broader multimodal quality beyond the objective smoke",
    "deep-context validation beyond canaries on the multi-chunk prefill path"
   ]
  },
  {
   "manifest": "packages/qwen38-27b-int4-fixed-k-tp2-b70/package.json",
   "format": "b70-model-package-v1",
   "id": "qwen38-27b-autoround-int4-b70",
   "name": "Qwen3.8 27B AutoRound INT4, fixed-K batch-invariant profile, on two Intel Arc Pro B70 cards",
   "status": "candidate",
   "audience": "researcher",
   "guide": "repro/qwen38-27b-autoround-int4-b70/README.md",
   "clean_host_tested": false,
   "library": {
    "model_family": "Qwen3.8",
    "publisher": "devan-carlin (AutoRound INT4 of Alibaba / Qwen)",
    "variant": "27B AutoRound INT4 W4A16",
    "summary": "devan-carlin's AutoRound INT4 tensors served through vLLM's plain-GPTQ oneDNN W4A16 path with a rebuilt kernel library that pins a two-tier fixed-K GEMM strategy, FP16 linears in 32-row pieces, single-split attention and size-independent Inductor reductions, on the FP8 lane's whole-graph deterministic stack. Single-request output is repeat-exact and speculative decoding is lossless against the MTP0 oracle at every depth measured; MTP0 output is byte-identical to a single request through 64 concurrent users (a near-tie prompt can differ in some runs), speculative depths through 16.",
    "quantization": "AutoRound INT4 W4A16 (group 128, symmetric)",
    "runtime_label": "vLLM XPU",
    "operating_systems": [
     "Linux"
    ],
    "delivery": [
     "container"
    ],
    "modalities": [
     "text"
    ],
    "use_cases": [
     "general",
     "coding",
     "long context"
    ],
    "tags": [
     "qwen3.8",
     "int4",
     "autoround",
     "gptq",
     "vllm",
     "xpu",
     "b70",
     "tp2",
     "mtp",
     "batch-invariant",
     "fixed-k"
    ],
    "published_at": "2026-09-05",
    "benchmark_status": "Research-status (2026-09-06). Headline: TP2 depth 4 with XPU graph capture and the draft-only INT4 lm_head (R257, R256 image) 112.362/112.325 tok/s, G2 12/12 and G3 12/12 vs the eager MTP0 oracle, acceptance 3.51; depth 5 109.971/110.069, depth 6 108.340/108.373 (R258). Graphs without the draft head (R247/R250/R253): MTP0 49.833/49.887, depth 1 76.723/76.629, depth 4 91.004/91.012, depth 5 88.844/89.117, depth 6 84.060/83.982. Eager R239 matrix: TP2 MTP0 34.210/35.640, d1 51.098/50.088, d2 61.140/61.541, d3 67.613/67.831, d4 68.552/67.789; TP1 MTP0 32.960/32.945, d1 49.640/49.468, d2 56.505/56.449, d3 58.471/58.474, d4 56.292/56.251; every pair 12/12 vs each other and vs the MTP0 oracle. Identity ladders: MTP0 exact c1-c64 (eager and under graphs); speculative depths exact through c16 (eager TP2) / c8 (TP1, graphs), c32 >= 29/32, c64 >= 58/64. R258/R259 (depths 5-6 and ladders on the headline configuration) pending. No promotion or LocalMaxxing submission yet."
   },
   "contributors": [
    {
     "id": "b70-optimization-lab",
     "name": "neural.download lab",
     "kind": "lab",
     "profile": "https://github.com/steveseguin/b70-optimization-lab",
     "contribution": "INT4 kernel routing diagnosis and gptq relabel, two-tier fixed-K oneDNN W4A16 strategy (rebuilt kernel library), FP16 row-chunk linears, batch-invariant attention and Inductor reductions, strict determinism/lossless/concurrency qualification on the R187 stack",
     "status": "integrated",
     "validated_effect": "The deterministic compiled TP2 work historically raised MTP0 from 18.910242 to 34.031596 tok/s, and block W8A16 separately raised matched target-only single/c128 shapes by 60.07%/29.30%. Clean-boot R119 qualified the draft-only INT4 R62 MTP1 profile at 54.424603 tok/s, 5.0504% above the prior qualified MTP1 profile, with both candidate and MTP0-oracle comparisons exact on 12/12 complete arrays in scope.",
     "evidence": "experiments/qwen38-27b-b70/notes/2026-09-05-qwen38-int4-concurrency-identity-r222-r226.md"
    }
   ],
   "hardware": {
    "accelerator": "Intel Arc Pro B70 32 GiB",
    "cards": 2,
    "host_ram_min_bytes": 16106127360,
    "host_ram_plus_swap_min_bytes": 21474836480,
    "model_weight_bytes": 19016930167
   },
   "model": {
    "repository": "devan-carlin/Qwen3.8-27B-int4-AutoRound",
    "revision": "bce40cacab0a4535b92fb3d57615c2bea9adf3d1",
    "manifest": "repro/qwen38-27b-autoround-int4-b70/manifests/model-gptq-relabel-r212.json",
    "note": "Identical tensors to the published revision, hard-linked into a directory whose config.json/quantization_config.json are relabelled from auto-round to plain gptq by repro/qwen38-27b-autoround-int4-b70/scripts/make-gptq-relabel.py; the manifest verifies that directory."
   },
   "runtime": {
    "kind": "container",
    "image": "ghcr.io/steveseguin/vllm-openai-xpu-qwen38-int4@sha256:40d46730c9a24f9396cc67c0e5578dd80d11dfae7a4d23a55f97620140a0b3e6",
    "image_id_validated": "sha256:40d46730c9a24f9396cc67c0e5578dd80d11dfae7a4d23a55f97620140a0b3e6",
    "local_tag": "neural-download/vllm-openai-xpu:qwen38-int4-fp16-linear-classpad-cheapest-r293",
    "xpu_extension_sha256": "271db0d4882124e21ac6a4d080bfeab303fbb08b9ec10e11f21d10fb0723998f",
    "xpu_ops_sha256": "6ee6b8db18759873246aca28e85ca6d2ba177eb08bfd3b9b0f0feea168cee9b3",
    "base": "R228 (fixed-K _xpu_C + FP16 row chunks + GDN spec grouping) + R256 draft-only INT4 lm_head fallback + R276 sync-free grouped GDN branch (graph capture to 320) + R293 class-consistent FP16 linear mode (VLLM_XPU_FP16_LINEAR_CLASSPAD; off = R276 code path)",
    "vllm": "0.27.2rc1.dev77+gac7509e2b.xpu",
    "kernel_head": "1e90ffa672ba02f17a909da11838a4c55b199783",
    "onednn_head": "0e2a5bfeef1bfbffc3137464606540233086ce9b"
   },
   "project_patches": {
    "required": true,
    "items": [
     "experiments/qwen38-27b-b70/docker/r213b-w4a16-determinism-pad-op.py",
     "experiments/qwen38-27b-b70/docker/r224-fp16-linear-rowchunk.py",
     "experiments/qwen38-27b-b70/docker/r228-gdn-spec-group.py",
     "experiments/qwen38-27b-b70/docker/r256-draft-int4-head-fallback.py",
     "experiments/qwen38-27b-b70/patches/onednn-qwen38-w4a16-fixed-k-two-tier-r221-20260905.patch",
     "experiments/qwen38-27b-b70/patches/onednn-qwen38-w4a16-strategy-override-dump-r220-20260905.patch",
     "experiments/qwen38-27b-b70/patches/onednn-qwen38-w8a16-c-default-align-r137b-20260902.patch",
     "experiments/qwen38-27b-b70/patches/onednn-qwen38-w8a16-fixed-k-align16-r137a-20260902.patch",
     "experiments/qwen38-27b-b70/docker/r276-gdn-spec-group-sync-free.py",
     "experiments/qwen38-27b-b70/docker/r290-fp16-linear-classpad.py",
     "experiments/qwen38-27b-b70/docker/r291-fp16-linear-classpad-verified.py",
     "experiments/qwen38-27b-b70/docker/r292-fp16-linear-classpad-nozero.py",
     "experiments/qwen38-27b-b70/docker/r293-fp16-linear-classpad-cheapest.py"
    ]
   },
   "commands": {
    "preflight": "docker pull ghcr.io/steveseguin/vllm-openai-xpu-qwen38-int4@sha256:40d46730c9a24f9396cc67c0e5578dd80d11dfae7a4d23a55f97620140a0b3e6 && docker tag ghcr.io/steveseguin/vllm-openai-xpu-qwen38-int4@sha256:40d46730c9a24f9396cc67c0e5578dd80d11dfae7a4d23a55f97620140a0b3e6 neural-download/vllm-openai-xpu:qwen38-int4-fp16-linear-classpad-cheapest-r293 && IMAGE=neural-download/vllm-openai-xpu:qwen38-int4-fp16-linear-classpad-cheapest-r293 EXPECTED_XPU_EXTENSION_SHA256=271db0d4882124e21ac6a4d080bfeab303fbb08b9ec10e11f21d10fb0723998f EXPECTED_XPU_OPS_SHA256=6ee6b8db18759873246aca28e85ca6d2ba177eb08bfd3b9b0f0feea168cee9b3 EXPECTED_LAYERNORM_SHA256=50cf5f4f9c72f679e4318cd3e3e021a844f59ac188a891d9a4f9638188f4bce8 repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/verify-image-contract.sh mtp1-serial-fa-split-gdn neural-download/vllm-openai-xpu:qwen38-int4-fp16-linear-classpad-cheapest-r293 && python3 repro/qwen38-27b-autoround-int4-b70/scripts/make-gptq-relabel.py /path/to/qwen3.8-27b-int4-autoround /path/to/qwen3.8-27b-int4-autoround-gptq-relabel --manifest repro/qwen38-27b-autoround-int4-b70/manifests/model-gptq-relabel-r212.json",
    "launch": "PORT=18134 MTP_DEPTH=4 XPU_GRAPH=1 DRAFT_HEAD_INT4=1 CLASSPAD=0 MODEL_DIR=/path/to/qwen3.8-27b-int4-autoround-gptq-relabel VLLM_CACHE_DIR=/path/to/new-runtime-cache repro/qwen38-27b-autoround-int4-b70/scripts/run-fixed-k-mtp-server.sh",
    "health": "curl -fsS http://127.0.0.1:18134/health",
    "benchmark": "OUT_DIR=/path/to/new-strict-attempt PORT=18134 MODEL_NAME=qwen38-int4-fixed-k-mtp4 PROFILE_LABEL=int4-fixed-k-mtp4 repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/bench-w8a16-mtp1-strict.sh",
    "stop": "docker stop -t 30 qwen38-int4-fixed-k-mtp4"
   },
   "dependencies": [
    "experiments/qwen38-27b-b70/data/2026-09-05-qwen38-int4-concurrency-ladders-r222-r225-result.json",
    "experiments/qwen38-27b-b70/data/2026-09-05-qwen38-int4-graph-capture-tp2-mtp4-r247-result.json",
    "experiments/qwen38-27b-b70/data/2026-09-05-qwen38-int4-r239-matrix-result.json",
    "experiments/qwen38-27b-b70/data/2026-09-06-qwen38-int4-r256-real-content-depth-r260b-result.json",
    "experiments/qwen38-27b-b70/docker/r213b-w4a16-determinism-pad-op.py",
    "experiments/qwen38-27b-b70/docker/r224-fp16-linear-rowchunk.py",
    "experiments/qwen38-27b-b70/docker/r228-gdn-spec-group.py",
    "experiments/qwen38-27b-b70/docker/r256-draft-int4-head-fallback.py",
    "experiments/qwen38-27b-b70/notes/2026-09-05-qwen38-int4-concurrency-identity-r222-r226.md",
    "experiments/qwen38-27b-b70/patches/onednn-qwen38-w4a16-fixed-k-two-tier-r221-20260905.patch",
    "experiments/qwen38-27b-b70/patches/onednn-qwen38-w4a16-strategy-override-dump-r220-20260905.patch",
    "experiments/qwen38-27b-b70/patches/onednn-qwen38-w8a16-c-default-align-r137b-20260902.patch",
    "experiments/qwen38-27b-b70/patches/onednn-qwen38-w8a16-fixed-k-align16-r137a-20260902.patch",
    "repro/gemma4-26b-a4b-q8-b70/realistic-suite-v1.json",
    "repro/qwen36-27b-autoround-int4-b70/realistic-suite-v1.json",
    "repro/qwen38-27b-autoround-int4-b70/README.md",
    "repro/qwen38-27b-autoround-int4-b70/manifests/model-gptq-relabel-r212.json",
    "repro/qwen38-27b-autoround-int4-b70/scripts/make-gptq-relabel.py",
    "repro/qwen38-27b-autoround-int4-b70/scripts/publish-r228-image-ghcr.sh",
    "repro/qwen38-27b-autoround-int4-b70/scripts/publish-r256-image-ghcr.sh",
    "repro/qwen38-27b-autoround-int4-b70/scripts/run-fixed-k-mtp-server.sh",
    "repro/qwen38-27b-autoround-int4-b70/scripts/verify-model-direct.py",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/bench-w8a16-mtp1-strict.sh",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/run-w8a16-mtp0-strict-server.sh",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/run-w8a16-mtp1-server.sh",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/verify-image-contract.sh",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/verify-model-direct.sh",
    "scripts/bench-openai-realistic-suite.py",
    "scripts/neural-download-canaries.py",
    "experiments/qwen38-27b-b70/docker/Dockerfile.gdn-spec-group-sync-free-r276",
    "experiments/qwen38-27b-b70/docker/Dockerfile.fp16-linear-classpad-r290",
    "experiments/qwen38-27b-b70/docker/Dockerfile.fp16-linear-classpad-verified-r291",
    "experiments/qwen38-27b-b70/docker/Dockerfile.fp16-linear-classpad-nozero-r292",
    "experiments/qwen38-27b-b70/docker/Dockerfile.fp16-linear-classpad-cheapest-r293",
    "experiments/qwen38-27b-b70/docker/r276-gdn-spec-group-sync-free.py",
    "experiments/qwen38-27b-b70/docker/r290-fp16-linear-classpad.py",
    "experiments/qwen38-27b-b70/docker/r291-fp16-linear-classpad-verified.py",
    "experiments/qwen38-27b-b70/docker/r292-fp16-linear-classpad-nozero.py",
    "experiments/qwen38-27b-b70/docker/r293-fp16-linear-classpad-cheapest.py",
    "repro/qwen38-27b-autoround-int4-b70/scripts/publish-r293-image-ghcr.sh",
    "experiments/qwen38-27b-b70/scripts/run-20260911-qwen38-int4-r295-r298-classpad-chain.sh",
    "experiments/qwen35-4b-b70/probes/validate-r292-stale-pad.py",
    "experiments/qwen35-4b-b70/probes/fp16-linear-mclass-census-extended.py"
   ],
   "performance_profiles": [
    {
     "id": "http-int4-fixed-k-mtp0-identity-qualified-aggregate-vs-concurrent-users",
     "label": "INT4 fixed-K TP2 MTP0 identity-qualified aggregate decode vs concurrent users (c1-c64)",
     "metric": "aggregate_decode",
     "unit": "tok/s",
     "x_metric": "concurrent_sequences",
     "x_label": "Concurrent HTTP users",
     "scope": "Measured vLLM HTTP completions on two B70s with the R224 image (fixed-K W4A16, FP16 row-chunk linears) and VLLM_BATCH_INVARIANT=0 (corrected 2026-09-06: vLLM's own batch-invariant switch was never in effect on this lane; the strict launchers pin it to 0 and vLLM refuses to boot the GDN backend with it set), no speculation, FP16 activations/KV, cache disabled, 128 returned raw token IDs per response on the 64-prompt small-context suite (R226). Every point is output-identity-qualified: each concurrent output equals its own sequential oracle byte for byte. One server, one pass per point; not a promoted concurrency speed record.",
     "evidence": "experiments/qwen38-27b-b70/data/2026-09-05-qwen38-int4-concurrency-ladders-r222-r225-result.json",
     "points": [
      {
       "concurrent_sequences": 1,
       "value": 34.3,
       "samples": 1,
       "per_user_value": 34.3
      },
      {
       "concurrent_sequences": 2,
       "value": 67.6,
       "samples": 1,
       "per_user_value": 33.8
      },
      {
       "concurrent_sequences": 4,
       "value": 136.0,
       "samples": 1,
       "per_user_value": 34.0
      },
      {
       "concurrent_sequences": 8,
       "value": 261.9,
       "samples": 1,
       "per_user_value": 32.7375
      },
      {
       "concurrent_sequences": 16,
       "value": 504.8,
       "samples": 1,
       "per_user_value": 31.55
      },
      {
       "concurrent_sequences": 32,
       "value": 841.3,
       "samples": 1,
       "per_user_value": 26.2906
      },
      {
       "concurrent_sequences": 64,
       "value": 998.3,
       "samples": 1,
       "per_user_value": 15.5984
      }
     ]
    },
    {
     "id": "http-int4-fixed-k-mtp4-identity-qualified-aggregate-vs-concurrent-users",
     "label": "INT4 fixed-K TP2 MTP depth-4 identity-qualified aggregate decode vs concurrent users (c1-c16)",
     "metric": "aggregate_decode",
     "unit": "tok/s",
     "x_metric": "concurrent_sequences",
     "x_label": "Concurrent HTTP users",
     "scope": "As above with MTP depth 4 on the R228 image, VLLM_BATCH_INVARIANT=0 (corrected 2026-09-06: vLLM's own batch-invariant switch was never in effect on this lane; the strict launchers pin it to 0 and vLLM refuses to boot the GDN backend with it set) and Inductor split_reductions=false (R232). Points through c16 are output-identity-qualified; c32 (31/32) and c64 (63/64) were measured but are withheld because near-tie prompts took a different valid branch.",
     "evidence": "experiments/qwen38-27b-b70/data/2026-09-05-qwen38-int4-concurrency-ladders-r222-r225-result.json",
     "points": [
      {
       "concurrent_sequences": 1,
       "value": 64.4,
       "samples": 1,
       "per_user_value": 64.4
      },
      {
       "concurrent_sequences": 2,
       "value": 73.0,
       "samples": 1,
       "per_user_value": 36.5
      },
      {
       "concurrent_sequences": 4,
       "value": 219.4,
       "samples": 1,
       "per_user_value": 54.85
      },
      {
       "concurrent_sequences": 8,
       "value": 356.0,
       "samples": 1,
       "per_user_value": 44.5
      },
      {
       "concurrent_sequences": 16,
       "value": 516.8,
       "samples": 1,
       "per_user_value": 32.3
      }
     ]
    },
    {
     "id": "http-int4-fixed-k-tp1-mtp0-identity-qualified-aggregate-vs-concurrent-users",
     "label": "INT4 fixed-K TP1 (one card) MTP0 identity-qualified aggregate decode vs concurrent users (c1-c64)",
     "metric": "aggregate_decode",
     "unit": "tok/s",
     "x_metric": "concurrent_sequences",
     "x_label": "Concurrent HTTP users",
     "scope": "One B70 (TENSOR_PARALLEL_SIZE=1), R228 image, VLLM_BATCH_INVARIANT=0 (corrected 2026-09-06: vLLM's own batch-invariant switch was never in effect on this lane; the strict launchers pin it to 0 and vLLM refuses to boot the GDN backend with it set), Inductor split_reductions=false, no speculation, cache disabled, 128 returned raw token IDs per response on the 64-prompt small-context suite (R239 TP1 depth-1 campaign MTP0 ladder). Every point is output-identity-qualified against its sequential oracle. One server, one pass per point.",
     "evidence": "experiments/qwen38-27b-b70/data/2026-09-05-qwen38-int4-r239-matrix-result.json",
     "points": [
      {
       "concurrent_sequences": 1,
       "value": 32.8,
       "samples": 1,
       "per_user_value": 32.8
      },
      {
       "concurrent_sequences": 2,
       "value": 62.7,
       "samples": 1,
       "per_user_value": 31.35
      },
      {
       "concurrent_sequences": 4,
       "value": 118.6,
       "samples": 1,
       "per_user_value": 29.65
      },
      {
       "concurrent_sequences": 8,
       "value": 206.5,
       "samples": 1,
       "per_user_value": 25.8125
      },
      {
       "concurrent_sequences": 16,
       "value": 342.9,
       "samples": 1,
       "per_user_value": 21.4312
      },
      {
       "concurrent_sequences": 32,
       "value": 506.0,
       "samples": 1,
       "per_user_value": 15.8125
      },
      {
       "concurrent_sequences": 64,
       "value": 447.6,
       "samples": 1,
       "per_user_value": 6.9938
      }
     ]
    },
    {
     "id": "http-int4-fixed-k-graph-tp2-mtp0-identity-qualified-aggregate-vs-concurrent-users",
     "label": "INT4 fixed-K TP2 MTP0 with XPU graph capture, identity-qualified aggregate decode vs concurrent users (c1-c64)",
     "metric": "aggregate_decode",
     "unit": "tok/s",
     "x_metric": "concurrent_sequences",
     "x_label": "Concurrent HTTP users",
     "scope": "Two B70s, R228 image, VLLM_BATCH_INVARIANT=0 (corrected 2026-09-06: vLLM's own batch-invariant switch was never in effect on this lane; the strict launchers pin it to 0 and vLLM refuses to boot the GDN backend with it set), split_reductions=false, VLLM_XPU_ENABLE_XPU_GRAPH=1 with FULL_DECODE_ONLY capture sizes 1-8, no speculation, cache disabled, 128 returned raw token IDs per response on the 64-prompt small-context suite (R251). Every point is output-identity-qualified against its sequential oracle. One server, one pass per point.",
     "evidence": "experiments/qwen38-27b-b70/data/2026-09-05-qwen38-int4-graph-capture-tp2-mtp4-r247-result.json",
     "points": [
      {
       "concurrent_sequences": 1,
       "value": 50.0,
       "samples": 1,
       "per_user_value": 50.0
      },
      {
       "concurrent_sequences": 2,
       "value": 94.4,
       "samples": 1,
       "per_user_value": 47.2
      },
      {
       "concurrent_sequences": 4,
       "value": 178.7,
       "samples": 1,
       "per_user_value": 44.675
      },
      {
       "concurrent_sequences": 8,
       "value": 317.7,
       "samples": 1,
       "per_user_value": 39.7125
      },
      {
       "concurrent_sequences": 16,
       "value": 501.2,
       "samples": 1,
       "per_user_value": 31.325
      },
      {
       "concurrent_sequences": 32,
       "value": 842.1,
       "samples": 1,
       "per_user_value": 26.3156
      },
      {
       "concurrent_sequences": 64,
       "value": 999.4,
       "samples": 1,
       "per_user_value": 15.6156
      }
     ]
    },
    {
     "id": "http-int4-fixed-k-graph-drafthead-tp2-mtp4-identity-qualified-aggregate-vs-concurrent-users",
     "label": "INT4 fixed-K TP2 MTP depth 4 with graph capture and the draft-only INT4 head, identity-qualified aggregate decode vs concurrent users (c1-c16, warm pass)",
     "metric": "aggregate_decode",
     "unit": "tok/s",
     "x_metric": "concurrent_sequences",
     "x_label": "Concurrent HTTP users",
     "scope": "Two B70s, R256 image, split_reductions=false, VLLM_XPU_ENABLE_XPU_GRAPH=1 (FULL_DECODE_ONLY, sizes 1-8), VLLM_XPU_DRAFT_LM_HEAD_INT4=1, MTP depth 4, cache disabled, 128 returned raw token IDs per response on the 64-prompt small-context suite; every rung run twice on one server and the second (warm) pass shown (R265b, 2026-09-06): the first pass of a fresh server carries a one-time Dynamo recompilation on the first multi-user batch. Every point through c16 is output-identity-qualified (16/16 at every rung). Above 16 users the corrected re-measurement with the W4A16 pad switch forwarded (R281) gives c32 632.6 (30/32) and c64 603.7 (62/64, admission-limited at max-model-len 256), withheld because near-tie prompts take a different valid branch; the R276 image with capture sizes to 320 (R282) adds +22%/+10% at two/four users (191.0/294.7).",
     "evidence": "experiments/qwen38-27b-b70/data/2026-09-05-qwen38-int4-graph-capture-tp2-mtp4-r247-result.json",
     "points": [
      {
       "concurrent_sequences": 1,
       "value": 107.3,
       "samples": 1,
       "per_user_value": 107.3
      },
      {
       "concurrent_sequences": 2,
       "value": 147.6,
       "samples": 1,
       "per_user_value": 73.8
      },
      {
       "concurrent_sequences": 4,
       "value": 252.9,
       "samples": 1,
       "per_user_value": 63.225
      },
      {
       "concurrent_sequences": 8,
       "value": 405.4,
       "samples": 1,
       "per_user_value": 50.675
      },
      {
       "concurrent_sequences": 16,
       "value": 580.4,
       "samples": 1,
       "per_user_value": 36.275
      }
     ]
    },
    {
     "id": "http-int4-mtp0-decode-vs-active-context",
     "label": "INT4 fixed-K TP2 MTP0 (graphs) HTTP decode over exact active context",
     "metric": "decode",
     "unit": "tok/s",
     "x_label": "Exact active prompt tokens",
     "scope": "Measured 2026-09-06 (R260b, clean boot) on the headline configuration (R256 image, two B70s, XPU graph capture FULL_DECODE_ONLY sizes 1-8, strict launcher env): one-slot vLLM HTTP completions at exactly 2K, 4K, 8K, 16K, 24K, and 32K active context, unrepeated technical prose, Python, and structured documents (three requests per depth, median shown), 128 output tokens, max-model-len 33024, max-num-batched-tokens 4096, cache zero, canaries before and after. MTP0 (target only). No value is interpolated.",
     "evidence": "experiments/qwen38-27b-b70/data/2026-09-06-qwen38-int4-r256-real-content-depth-r260b-result.json",
     "points": [
      {
       "context_tokens": 2048,
       "value": 50.023492143506274,
       "samples": 3
      },
      {
       "context_tokens": 4096,
       "value": 48.78019783970367,
       "samples": 3
      },
      {
       "context_tokens": 8192,
       "value": 47.46198372409749,
       "samples": 3
      },
      {
       "context_tokens": 16384,
       "value": 45.92111530825193,
       "samples": 3
      },
      {
       "context_tokens": 24576,
       "value": 44.174870800726,
       "samples": 3
      },
      {
       "context_tokens": 32768,
       "value": 42.832649479370204,
       "samples": 3
      }
     ]
    },
    {
     "id": "http-int4-mtp4-decode-vs-active-context",
     "label": "INT4 fixed-K TP2 MTP depth 4 (graphs, draft-only INT4 head) HTTP decode over exact active context",
     "metric": "decode",
     "unit": "tok/s",
     "x_label": "Exact active prompt tokens",
     "scope": "Measured 2026-09-06 (R260b, clean boot) on the headline configuration (R256 image, two B70s, XPU graph capture FULL_DECODE_ONLY sizes 1-8, strict launcher env): one-slot vLLM HTTP completions at exactly 2K, 4K, 8K, 16K, 24K, and 32K active context, unrepeated technical prose, Python, and structured documents (three requests per depth, median shown), 128 output tokens, max-model-len 33024, max-num-batched-tokens 4096, cache zero, canaries before and after. MTP depth 4 with the FP16 target verifier and draft-only INT4 head; every output matched the same-configuration MTP0 oracle (18/18 complete arrays). Class spread is wide at every depth (Python code accepts more draft tokens); the median of the three classes is shown.",
     "evidence": "experiments/qwen38-27b-b70/data/2026-09-06-qwen38-int4-r256-real-content-depth-r260b-result.json",
     "points": [
      {
       "context_tokens": 2048,
       "value": 120.55879982780635,
       "samples": 3
      },
      {
       "context_tokens": 4096,
       "value": 125.72167735144107,
       "samples": 3
      },
      {
       "context_tokens": 8192,
       "value": 124.3502018967137,
       "samples": 3
      },
      {
       "context_tokens": 16384,
       "value": 90.18189154611792,
       "samples": 3
      },
      {
       "context_tokens": 24576,
       "value": 84.62792054167177,
       "samples": 3
      },
      {
       "context_tokens": 32768,
       "value": 100.271460163712,
       "samples": 3
      }
     ]
    },
    {
     "id": "http-int4-fixed-k-graph-r276-big-admission-tp2-mtp0-identity-qualified-aggregate-vs-concurrent-users",
     "label": "INT4 fixed-K TP2 MTP0, R276 image, 128-sequence admission, identity-qualified aggregate decode vs concurrent users (c16-c64; warm pass)",
     "metric": "aggregate_decode",
     "unit": "tok/s",
     "x_metric": "concurrent_sequences",
     "x_label": "Concurrent HTTP users",
     "scope": "R284 (2026-09-06): two B70s, R276 image, XPU graph capture FULL_DECODE_ONLY sizes to 320, strict launcher env with the W4A16 pad switch forwarded off, max-num-seqs 128, max-model-len 512, max-num-batched-tokens 1024, no speculation, cache disabled, 128 returned raw token IDs per response on the small-context suite; every rung run twice on one server, warm pass shown. c16/c32/c64 are output-identity-qualified in both passes (16/16, 32/32, 64/64). Measured but withheld: c96 915.4 (95/96 warm, 96/96 first pass) and c128 1085.3 (128/128 warm, 127/128 first pass), one near-tie prompt each.",
     "evidence": "experiments/qwen38-27b-b70/data/2026-09-05-qwen38-int4-graph-capture-tp2-mtp4-r247-result.json",
     "points": [
      {
       "concurrent_sequences": 16,
       "value": 533.8,
       "samples": 1,
       "per_user_value": 33.3625
      },
      {
       "concurrent_sequences": 32,
       "value": 815.0,
       "samples": 1,
       "per_user_value": 25.4688
      },
      {
       "concurrent_sequences": 64,
       "value": 991.4,
       "samples": 1,
       "per_user_value": 15.4906
      }
     ]
    },
    {
     "id": "http-int4-fixed-k-graph-drafthead-r276-big-admission-tp2-mtp4-identity-qualified-aggregate-vs-concurrent-users",
     "label": "INT4 fixed-K TP2 MTP depth 4 (draft-only INT4 head), R276 image, 128-sequence admission, identity-qualified aggregate decode vs concurrent users (c16-c32; warm pass)",
     "metric": "aggregate_decode",
     "unit": "tok/s",
     "x_metric": "concurrent_sequences",
     "x_label": "Concurrent HTTP users",
     "scope": "R284 (2026-09-06): as the MTP0 profile above with qwen3_next_mtp depth 4, FP16 target verifier and the draft-only INT4 lm_head. c16 and c32 are output-identity-qualified in both passes (16/16, 32/32). Measured but withheld: c64 591.4 (58/64), c96 590.6 (89/96), c128 584.7 (117/128); the same prompts diverge at the same token positions in every rung of 64 and above (near-tie flips in the >32-row W4A16 GEMM tier), and the depth-4 aggregate plateaus at 585-641 tok/s from c32 while MTP0 keeps scaling to 991 (c64) and 1085 (c128). Serve more than about 32 users without speculation.",
     "evidence": "experiments/qwen38-27b-b70/data/2026-09-05-qwen38-int4-graph-capture-tp2-mtp4-r247-result.json",
     "points": [
      {
       "concurrent_sequences": 16,
       "value": 574.3,
       "samples": 1,
       "per_user_value": 35.8937
      },
      {
       "concurrent_sequences": 32,
       "value": 641.3,
       "samples": 1,
       "per_user_value": 20.0406
      }
     ]
    },
    {
     "id": "http-int4-fixed-k-graph-r276-big-admission-tp1-mtp0-identity-qualified-aggregate-vs-concurrent-users",
     "label": "INT4 fixed-K TP1 (one card) MTP0, R276 image, 128-sequence admission, identity-qualified aggregate decode vs concurrent users (c16-c128; warm pass)",
     "metric": "aggregate_decode",
     "unit": "tok/s",
     "x_metric": "concurrent_sequences",
     "x_label": "Concurrent HTTP users",
     "scope": "R285 (2026-09-06): one B70 (TENSOR_PARALLEL_SIZE=1, GPU_MEMORY_UTILIZATION=0.96), R276 image, XPU graph capture FULL_DECODE_ONLY sizes to 320, strict launcher env with the W4A16 pad switch forwarded off, max-num-seqs 128, max-model-len 512, max-num-batched-tokens 1024, no speculation, cache disabled, 128 returned raw token IDs per response on the small-context suite; every rung run twice on one server, warm pass shown. Every rung is output-identity-qualified in both passes (16/16 through 128/128). Peak 514.0 tok/s at 32 users; from 64 users the batch budget queues prefill (ttft_max 13-27 s).",
     "evidence": "experiments/qwen38-27b-b70/data/2026-09-05-qwen38-int4-graph-capture-tp2-mtp4-r247-result.json",
     "points": [
      {
       "concurrent_sequences": 16,
       "value": 345.9,
       "samples": 1,
       "per_user_value": 21.6187
      },
      {
       "concurrent_sequences": 32,
       "value": 514.0,
       "samples": 1,
       "per_user_value": 16.0625
      },
      {
       "concurrent_sequences": 64,
       "value": 418.5,
       "samples": 1,
       "per_user_value": 6.5391
      },
      {
       "concurrent_sequences": 96,
       "value": 403.7,
       "samples": 1,
       "per_user_value": 4.2052
      },
      {
       "concurrent_sequences": 128,
       "value": 443.8,
       "samples": 1,
       "per_user_value": 3.4672
      }
     ]
    },
    {
     "id": "http-int4-fixed-k-graph-drafthead-r276-big-admission-tp1-mtp4-identity-qualified-aggregate-vs-concurrent-users",
     "label": "INT4 fixed-K TP1 (one card) MTP depth 4 (draft-only INT4 head), R276 image, 128-sequence admission, identity-qualified aggregate decode vs concurrent users (c16-c32; warm pass)",
     "metric": "aggregate_decode",
     "unit": "tok/s",
     "x_metric": "concurrent_sequences",
     "x_label": "Concurrent HTTP users",
     "scope": "R285 (2026-09-06): as the one-card MTP0 profile above with qwen3_next_mtp depth 4, FP16 target verifier and the draft-only INT4 lm_head. c16 and c32 are output-identity-qualified in both passes (16/16, 32/32). Measured but withheld: c64 246.2 (62/64), c96 243.4 (90/96), c128 243.9 (123/128), near-tie divergences as on TP2. One card saturates at ~245 tok/s with depth 4 from 16 users; MTP0 on the same card reaches 514 at 32 users, so serve more than about 8-16 one-card users without speculation.",
     "evidence": "experiments/qwen38-27b-b70/data/2026-09-05-qwen38-int4-graph-capture-tp2-mtp4-r247-result.json",
     "points": [
      {
       "concurrent_sequences": 16,
       "value": 237.3,
       "samples": 1,
       "per_user_value": 14.8313
      },
      {
       "concurrent_sequences": 32,
       "value": 243.7,
       "samples": 1,
       "per_user_value": 7.6156
      }
     ]
    },
    {
     "id": "http-int4-fixed-k-graph-drafthead-r276-big-admission-tp2-mtp1-identity-qualified-aggregate-vs-concurrent-users",
     "label": "INT4 fixed-K TP2 MTP depth 1 (draft-only INT4 head), R276 image, 128-sequence admission, identity-qualified aggregate decode vs concurrent users (c2-c32; warm pass)",
     "metric": "aggregate_decode",
     "unit": "tok/s",
     "x_metric": "concurrent_sequences",
     "x_label": "Concurrent HTTP users",
     "scope": "R287 (2026-09-06): as the R284 profiles (two B70s, R276 image, capture sizes to 320, pad off, max-num-seqs 128, max-model-len 512, max-num-batched-tokens 1024, two passes, warm pass shown) with qwen3_next_mtp depth 1. c2, c4 and c32 are output-identity-qualified in both passes (R288/R287); c16 in the warm pass (16/16; the first pass of the fresh server had one near-tie flip, 15/16). Measured but withheld: c64 842.0 (61/64), c96 906.3 (92/96), c128 894.8 (121/128). Depth 1 is the fastest setting through 32 users (depth 2: 723.4, depth 4: 641.3, no speculation: 815.0 at c32); from 64 users no speculation is faster and exact (992 at c64). R288 adds c2 147.3, c4 268.6 and c8 456.0 (7/8 in both passes, one near-tie prompt, withheld); depth 4 (R282) is faster at 1-4 users (191.0/294.7), depth 1 from 8 users.",
     "evidence": "experiments/qwen38-27b-b70/data/2026-09-05-qwen38-int4-graph-capture-tp2-mtp4-r247-result.json",
     "points": [
      {
       "concurrent_sequences": 2,
       "value": 147.3,
       "samples": 1,
       "per_user_value": 73.65
      },
      {
       "concurrent_sequences": 4,
       "value": 268.6,
       "samples": 1,
       "per_user_value": 67.15
      },
      {
       "concurrent_sequences": 16,
       "value": 711.0,
       "samples": 1,
       "per_user_value": 44.4375
      },
      {
       "concurrent_sequences": 32,
       "value": 854.4,
       "samples": 1,
       "per_user_value": 26.7
      }
     ]
    },
    {
     "id": "http-int4-fixed-k-graph-drafthead-r276-big-admission-tp1-mtp1-identity-qualified-aggregate-vs-concurrent-users",
     "label": "INT4 fixed-K TP1 (one card) MTP depth 1 (draft-only INT4 head), R276 image, 128-sequence admission, identity-qualified aggregate decode vs concurrent users (c4-c32; warm pass)",
     "metric": "aggregate_decode",
     "unit": "tok/s",
     "x_metric": "concurrent_sequences",
     "x_label": "Concurrent HTTP users",
     "scope": "R289 (2026-09-06): as the one-card R285 profiles (TENSOR_PARALLEL_SIZE=1, R276 image, capture sizes to 320, pad off, max-num-seqs 128, max-model-len 512, max-num-batched-tokens 1024, two passes, warm pass shown) with qwen3_next_mtp depth 1. c4, c8 and c16 are output-identity-qualified in both passes; c32 in the warm pass (32/32; first pass 30/32) and already behind no speculation there (371.9 vs 513.8, prefill queueing with ttft_max 7.5 s). One card: depth 1 from 4 to 16 users (464.7 at 16 vs 345.7 without speculation), no speculation from 32 users.",
     "evidence": "experiments/qwen38-27b-b70/data/2026-09-05-qwen38-int4-graph-capture-tp2-mtp4-r247-result.json",
     "points": [
      {
       "concurrent_sequences": 4,
       "value": 181.4,
       "samples": 1,
       "per_user_value": 45.35
      },
      {
       "concurrent_sequences": 8,
       "value": 302.8,
       "samples": 1,
       "per_user_value": 37.85
      },
      {
       "concurrent_sequences": 16,
       "value": 464.7,
       "samples": 1,
       "per_user_value": 29.0437
      },
      {
       "concurrent_sequences": 32,
       "value": 371.9,
       "samples": 1,
       "per_user_value": 11.6219
      }
     ]
    },
    {
     "id": "http-int4-r293-classpad-mtp0-identity-qualified-aggregate-vs-concurrent-users",
     "label": "INT4 fixed-K TP2 MTP0, R293 class-consistent FP16 linears (CLASSPAD=1): identity-qualified aggregate decode vs concurrent users (c1-c64, warm pass)",
     "metric": "aggregate_decode",
     "unit": "tok/s",
     "x_metric": "concurrent_sequences",
     "x_label": "Concurrent HTTP users",
     "scope": "Measured vLLM HTTP completions on two B70s with the R293 image and VLLM_XPU_FP16_LINEAR_CLASSPAD=1 (R295, 2026-09-11): the served R276 configuration (capture sizes to 320, INT4 draft head, pad off) with every unquantized FP16 linear kept in one verified oneDNN M-class instead of 32-row pieces. No speculation, 128 returned raw token IDs per response on the 64-prompt small-context suite, second of two passes. Every point is output-identity-qualified: each concurrent output equals its own sequential oracle byte for byte (128/128 at 64 users). One server; not a promoted concurrency speed record. With the 5 ms admission stagger the same rung is 640/640 over ten passes at 1014.4 tok/s (R297).",
     "evidence": "experiments/qwen38-27b-b70/data/2026-09-05-qwen38-int4-graph-capture-tp2-mtp4-r247-result.json",
     "points": [
      {
       "concurrent_sequences": 1,
       "value": 49.6,
       "samples": 1,
       "per_user_value": 49.6,
       "exact": "1/1"
      },
      {
       "concurrent_sequences": 2,
       "value": 94.9,
       "samples": 1,
       "per_user_value": 47.45,
       "exact": "2/2"
      },
      {
       "concurrent_sequences": 4,
       "value": 177.9,
       "samples": 1,
       "per_user_value": 44.475,
       "exact": "4/4"
      },
      {
       "concurrent_sequences": 8,
       "value": 326.5,
       "samples": 1,
       "per_user_value": 40.8125,
       "exact": "8/8"
      },
      {
       "concurrent_sequences": 16,
       "value": 536.3,
       "samples": 1,
       "per_user_value": 33.5187,
       "exact": "16/16"
      },
      {
       "concurrent_sequences": 32,
       "value": 815.4,
       "samples": 1,
       "per_user_value": 25.4812,
       "exact": "32/32"
      },
      {
       "concurrent_sequences": 64,
       "value": 1019.2,
       "samples": 1,
       "per_user_value": 15.925,
       "exact": "64/64"
      }
     ]
    }
   ],
   "missing": [
    "clean-host replay"
   ]
  },
  {
   "manifest": "packages/qwen38-27b-fp8-tp2-b70/package.json",
   "format": "b70-model-package-v1",
   "id": "qwen38-27b-fp8-vllm-tp2-asrock-b70",
   "name": "Qwen3.8 27B official FP8 on two Intel Arc Pro B70 cards",
   "status": "candidate",
   "audience": "intermediate",
   "guide": "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/README.md",
   "clean_host_tested": false,
   "library": {
    "model_family": "Qwen3.8",
    "publisher": "Alibaba / Qwen",
    "variant": "27B official FP8",
    "summary": "Alibaba's Qwen3.8 27B official block-FP8 weights on two Arc Pro B70 cards. The R187 profile (the R156 row-invariant W8A16 kernel and mixed-step GDN split, served with one whole-graph torch.compile instead of vLLM's piecewise split) is clean-boot-qualified at 54.935 tok/s MTP1 (FP16 target verifier, draft-only INT4 head), 70.142 tok/s MTP depth 2, 79.183 tok/s MTP depth 3, 82.396 tok/s MTP depth 4, 86.182 tok/s MTP depth 5, and 33.097 tok/s MTP0; every pair 12/12 against a same-configuration MTP0 oracle, repeat-exact at 224-300-token prompts, MTP0 output-identical to a single request through 64 concurrent users, MTP1 and depths 3-5 through 16, depth 2 through 4. On the piecewise compile MTP depth 2 emitted a phantom first token on one request in 64; on the whole-graph compile no pass has shown it (R182-R193, 2026-09-03). The cause is an unfixed upstream vLLM defect that also occurs on the unmodified image (R192/R194), so this is a configuration that avoids it on our deterministic build, not a fix; no patch, no image rebuild. A prebuilt copy of the exact image is on GitHub Container Registry (ghcr.io/steveseguin/vllm-openai-xpu-qwen38-fp8, digest sha256:173660ec\u2026, equal to the image id the launchers verify); the source build in the guide remains the authoritative route.",
    "quantization": "FP8",
    "runtime_label": "vLLM XPU",
    "operating_systems": [
     "Linux"
    ],
    "delivery": [
     "container"
    ],
    "modalities": [
     "text"
    ],
    "use_cases": [
     "general",
     "coding",
     "long context"
    ],
    "tags": [
     "two cards",
     "Docker",
     "official checkpoint",
     "target only",
     "MTP0",
     "MTP1",
     "dynamic MTP8 to MTP1",
     "W8A16",
     "row-invariant W8A16",
     "mixed-step GDN split"
    ],
    "published_at": "2026-08-27",
    "benchmark_status": "Strict R187 qualification (2026-09-03, clean boot, whole-graph compile splitting_ops=[]): two same-configuration MTP0 controls 33.111/33.082 tok/s matched 12/12; two depth-1 MTP1 candidates 55.006/54.865 tok/s (center 54.935) matched 12/12 against each other and against the MTP0 oracle (R188); two depth-2 candidates 70.146/70.138 tok/s (center 70.142) matched 12/12 (R187); canaries before and after every server; cache zero; repeat-exact at 224, 250 and 300 prompt tokens on both MTP profiles; identity ladders: MTP0 64/64 through c64 (927.9 tok/s aggregate), MTP1 exact through c16 (477.7 tok/s; c32 31/32, c64 59/64), depth 2 exact through c4 (210.8 tok/s; c8 7/8, c32 31/32, c64 60/64). Depth 3 (R191/R193): two candidates 79.163/79.203 tok/s (center 79.183) matched 12/12 vs each other and vs the MTP0 oracle; probe exact; two ladders exact through c16 (557.0 tok/s aggregate). Depth 4 (R197/R201): two candidates 82.447/82.345 tok/s (center 82.396) matched 12/12 vs each other and vs the MTP0 oracle; probe exact; two ladders exact through c16 (529.4 tok/s aggregate). Depth 5 (R200/R204): two candidates 86.266/86.097 tok/s (center 86.182) matched 12/12 vs each other and vs the MTP0 oracle; probe exact; two ladders exact through c16 (493.2 tok/s aggregate). Depth 6 87.12 and depth 7 85.94 on strict pairs only (turnover at 6, +1.1% over depth 5, below the 2% bar).",
    "featured_metric": {
     "value": 86.18172184524545,
     "unit": "tok/s",
     "label": "strict varied-prompt decode (R187 MTP depth 5, whole-graph compile)",
     "scope": "Median of two clean-boot fresh-server class-balanced medians over the complete fixed 12-prompt/six-class natural-512 suite (48-78-token prompts), separate empty compile caches, 12/12 complete token arrays exact versus a same-configuration MTP0 oracle (depth-5 target-verified speculative decoding; MTP1 54.935, depth 4 82.396, MTP0 33.097 on the same line), canaries before and after, cache zero; repeat-exact at 100-300-token prompts; output-identical through 16 concurrent users (MTP0: through 64).",
     "evidence": "experiments/qwen38-27b-b70/data/2026-09-04-qwen38-fp8-r200-whole-graph-depth5-strict-result.json"
    }
   },
   "contributors": [
    {
     "id": "b70-optimization-lab",
     "name": "neural.download lab",
     "kind": "lab",
     "profile": "https://github.com/steveseguin/b70-optimization-lab",
     "contribution": "B70/XPU integration, strict output validation, direct-I/O model verification, direct-P2P concurrency tuning, block-W8A16 dispatch, deterministic GDN state handling, explicit oneCCL completion ordering, exact packed MTP1 Gemma RMSNorm replay, dynamic-width GDN repair, active Mamba-state allocation, and the digest-pinned package.",
     "status": "integrated",
     "validated_effect": "The deterministic compiled TP2 work historically raised MTP0 from 18.910242 to 34.031596 tok/s, and block W8A16 separately raised matched target-only single/c128 shapes by 60.07%/29.30%. Clean-boot R119 qualified the draft-only INT4 R62 MTP1 profile at 54.424603 tok/s, 5.0504% above the prior qualified MTP1 profile, with both candidate and MTP0-oracle comparisons exact on 12/12 complete arrays in scope.",
     "evidence": "experiments/qwen38-27b-b70/notes/2026-09-02-qwen38-fp8-mtp1-draft-int4-r62-cleanboot-r119-promotion.md"
    },
    {
     "id": "vllm-xpu-kernel-contributors",
     "name": "vLLM XPU kernel contributors",
     "kind": "external",
     "profile": "https://github.com/vllm-project/vllm-xpu-kernels/commit/1d5b4f5e5ddd8da96ea23c76d7e7421b00083fdb",
     "contribution": "Split and corrected the native GDN speculative/non-speculative paths so continuous batching can mix MTP decode and newly arriving prefill rows.",
     "status": "integrated",
     "validated_effect": "The older kernel aborted the c16 mixed workload. With the pinned upstream correction integrated, c16 completed and the selected MTP1 profile reached 1091.642460 tok/s at c64 while passing 512/512 concurrent semantic checks.",
     "evidence": "experiments/qwen38-27b-b70/notes/2026-08-26-qwen38-fp8-block-w8a16-mtp1-tp2-result.md"
    }
   ],
   "hardware": {
    "accelerator": "Intel Arc Pro B70 32 GiB",
    "cards": 2,
    "host_ram_min_bytes": 16106127360,
    "host_ram_plus_swap_min_bytes": 21474836480,
    "model_weight_bytes": 30866866928
   },
   "model": {
    "repository": "Qwen/Qwen3.8-27B-FP8",
    "revision": "017b9c7af6b5689d5dd426a76e0bc077eb5ca20a",
    "manifest": "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/model-direct.json"
   },
   "runtime": {
    "kind": "container",
    "image": "vllm/vllm-openai-xpu@sha256:f01e24f6c7ff01f1e0662234255a1372297d1dbd89d003cf13c8fad3eab1ba4f",
    "overlay_image_id_validated": "sha256:ced02d013fe356faac513f2598b4da1f11fd8e20a9bb8fb9a443564fda460556",
    "mtp1_overlay_image_id_validated": "sha256:61bd8edb385c03b40cdadaba068608355b144a5011722597e7ca437f37346ecd",
    "dynamic_mtp_overlay_image_id_validated": "sha256:2b79af686423379e4418aafa92d72e2248e8d09fabe609284dc7e29190cb8cd6",
    "deterministic_eager_overlay_image_id_validated": "sha256:47507a8ca2a78e83666a6f300ec94c5b5c5915740f147bf9d1565c938de8f25b",
    "deterministic_compiled_overlay_image_id_validated": "sha256:d19f802ba702a9cb94b155f807a4674a0100702aee838323372f740d7168e34e",
    "deterministic_mtp1_overlay_image_id_validated": "sha256:ba42e928e69c60d1c9102df6ec1c0e998e9dd8463f74d5dc0a8b4bb45108fa9b",
    "qualified_mtp1_overlay_image_id_validated": "sha256:2932e495b560e79c6301f5cc64584af928a2260f0d0d19c145142b2ef35860d3",
    "clean_rebuild_mtp1_overlay_image_id_validated": "sha256:41aec5da9b124497a9b5dbc6b38f17175bf923d930d5702b9913589f107802d4",
    "qualified_r62_overlay_image_id_validated": "sha256:cac17acf96ebbf65bfbe98e45dcea8eb5626c2b027dcac0228bc3bcba0063374",
    "overlay_build": "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/build-w8a16-image.sh",
    "deterministic_eager_overlay_build": "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/build-deterministic-eager-image.sh",
    "deterministic_compiled_overlay_build": "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/build-deterministic-compiled-image.sh",
    "deterministic_mtp1_overlay_build": "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/build-pinned-mtp1-stack.sh",
    "qualified_r62_overlay_build": "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/build-draft-int4-r62-image.sh",
    "dynamic_mtp_overlay_build": "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/build-w8a16-dynamic-mamba-image.sh"
   },
   "project_patches": {
    "required": true,
    "items": [
     "experiments/qwen38-27b-b70/patches/vllm-qwen38-fp8-block-w8a16-20260826.patch",
     "experiments/qwen38-27b-b70/patches/vllm-qwen38-xpu-deterministic-gdn-ba-state-20260828.patch",
     "experiments/qwen38-27b-b70/patches/vllm-qwen38-xpu-compiled-gdn-state-ccl-wait-20260828.patch",
     "experiments/qwen38-27b-b70/patches/vllm-qwen38-xpu-gemma-rmsnorm-mtp1-serial-exact-r30-20260828.patch",
     "experiments/qwen38-27b-b70/patches/vllm-qwen38-xpu-serial-spec-flash-attn-r38-20260828.patch",
     "experiments/qwen38-27b-b70/patches/vllm-xpu-kernels-qwen38-dynamic-active-width-serial-gdn-r35-20260828.patch",
     "experiments/qwen38-27b-b70/patches/vllm-xpu-kernels-qwen38-gdn-split-serial-gates-r50-20260901.patch",
     "experiments/qwen38-27b-b70/patches/vllm-qwen38-fp8-draft-only-int4-lm-head-r62-20260901.patch",
     "experiments/qwen38-27b-b70/patches/vllm-xpu-kernels-qwen38-dynamic-mtp-active-width-20260826.patch",
     "experiments/qwen38-27b-b70/patches/vllm-qwen38-dynamic-mtp-mamba-active-allocation-20260826.patch",
     "experiments/qwen38-27b-b70/patches/onednn-qwen38-w8a16-fixed-k-align16-r137a-20260902.patch",
     "experiments/qwen38-27b-b70/patches/onednn-qwen38-w8a16-c-default-align-r137b-20260902.patch",
     "experiments/qwen38-27b-b70/patches/vllm-qwen38-xpu-gdn-split-mixed-step-r156-20260903.patch"
    ],
    "reason": "The qualified R187 profile is the R156 profile served with one whole-graph torch.compile (COMPILATION_CONFIG splitting_ops=[]; on the piecewise split at the attention/GDN ops MTP depth 2 emitted a phantom first token on one request in 64, which no whole-graph pass has shown; the cause is an unfixed upstream defect also seen on the unmodified image, R182-R194). R156 is the R62 image plus a Python mixed-step GDN split (vllm/_xpu_ops.py, VLLM_XPU_GDN_SPLIT_MIXED=1) and a rebuilt vllm-xpu-kernels extension whose oneDNN W8A16 GEMM uses a fixed-K row-invariant strategy selector (two oneDNN patches), so greedy output no longer depends on batch shape for batch sizes 1-512 and is repeat-exact at every prompt length. Everything else is the R62 chain: block-W8A16, deterministic GDN B/A and compiler-visible recurrent state, explicit oneCCL Work.wait, packed Gemma RMSNorm serial replay, the content-verified R50 GDN build, the deterministic vLLM compilation contract with XPU Graph disabled, and draft-only INT4 vocabulary projection with an FP16 target verifier. The compiled extension is published as a release binary with whole-file and section digests, and can be rebuilt from the pinned sources with host oneAPI 2026.1. Dynamic MTP additionally requires the active-width GDN and active-lookahead Mamba allocation patches and remains research-only."
   },
   "commands": {
    "preflight": "docker pull ghcr.io/steveseguin/vllm-openai-xpu-qwen38-fp8@sha256:173660ec18c6e98a14b9a4f573922abe9d3414999056f07ab5c3c14b55d6ceb0 && docker tag ghcr.io/steveseguin/vllm-openai-xpu-qwen38-fp8@sha256:173660ec18c6e98a14b9a4f573922abe9d3414999056f07ab5c3c14b55d6ceb0 neural-download/vllm-openai-xpu:qwen38-fp8-mtp1-gdn-split-mixed-r156 && IMAGE=neural-download/vllm-openai-xpu:qwen38-fp8-mtp1-gdn-split-mixed-r156 IMAGE_CONTRACT_PROFILE=mtp1-serial-fa-split-gdn EXPECTED_XPU_EXTENSION_SHA256=f912e12de1d79206221142c9a50af2aba70d2c77c735c9cd2d5d8d9def0740d1 EXPECTED_XPU_OPS_SHA256=6a7761930cd8b9e3f67902648ba5aaaf708567cebf70fcedda595d698f26b064 MODEL_DIR=/path/to/qwen3.8-27b-fp8 repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/preflight.sh",
    "launch": "PORT=18124 MODEL_DIR=/path/to/qwen3.8-27b-fp8 VLLM_CACHE_DIR=/path/to/new-runtime-cache EXPECTED_IMAGE_ID=sha256:replace-with-your-r156-image-id experiments/qwen38-27b-b70/scripts/run-20260903-qwen38-fp8-mtp1-whole-graph-r187-server.sh",
    "health": "curl -fsS http://127.0.0.1:18124/health",
    "benchmark": "OUT_DIR=/path/to/new-strict-attempt MODEL_NAME=qwen38-fp8-block-w8a16-mtp1-whole-graph-r187 PROFILE_LABEL=mtp1-whole-graph-r187 repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/bench-w8a16-mtp1-strict.sh",
    "stop": "docker stop -t 30 qwen38-fp8-w8a16-mtp1-whole-graph-r187"
   },
   "dependencies": [
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/README.md",
    "repro/gemma4-26b-a4b-q8-b70/realistic-suite-v1.json",
    "repro/qwen36-27b-autoround-int4-b70/realistic-suite-v1.json",
    "repro/qwen38-27b-autoround-int4-b70/scripts/verify-model-direct.py",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/verify-model-direct.sh",
    "scripts/bench-openai-realistic-suite.py",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/preflight.sh",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/run-server.sh",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/run-depth-server.sh",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/run-concurrency-server.sh",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/Dockerfile.w8a16",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/Dockerfile.deterministic-eager",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/Dockerfile.deterministic-compiled",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/build-w8a16-image.sh",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/build-deterministic-eager-image.sh",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/build-deterministic-compiled-image.sh",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/build-mtp1-kernel-image.sh",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/build-pinned-mtp1-stack.sh",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/verify-public-source-closure.sh",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/verify-image-contract.sh",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/kernel-wheel-build-info.txt",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/Dockerfile.mtp1-rmsnorm-serial",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/build-mtp1-rmsnorm-serial-image.sh",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/Dockerfile.mtp1-serial-attention",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/build-mtp1-serial-attention-image.sh",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/Dockerfile.mtp1-rebuilt-gdn",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/build-mtp1-rebuilt-gdn-image.sh",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/build-mtp1-published-r55c-image.sh",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/build-pinned-mtp1-published-r55c-stack.sh",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/publication-manifest.json",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/build-dynamic-mtp-active-width-kernel-image.sh",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/build-w8a16-dynamic-mamba-image.sh",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/run-w8a16-concurrency-server.sh",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/run-w8a16-mtp1-server.sh",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/run-w8a16-mtp0-strict-server.sh",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/run-w8a16-mtp1-strict-server.sh",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/run-w8a16-mtp0-depth-server.sh",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/run-w8a16-mtp1-depth-server.sh",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/run-w8a16-depth-server.sh",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/bench-w8a16-concurrency.sh",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/bench-w8a16-mtp1.sh",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/bench-w8a16-mtp1-strict.sh",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/bench-w8a16-strict.sh",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/bench-w8a16-mtp1-depth.sh",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/bench-w8a16-real-content-depth.sh",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/run-w8a16-dynamic-mtp-server.sh",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/bench-w8a16-dynamic-mtp.sh",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/bench-w8a16-dynamic-mtp-realistic.sh",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/bench.sh",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/bench-depth.sh",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/bench-concurrency.sh",
    "experiments/qwen38-27b-b70/notes/2026-08-27-qwen38-fp8-strict-profile-matrix-result.md",
    "experiments/qwen38-27b-b70/data/2026-08-27-qwen38-fp8-strict-profile-matrix-summary.json",
    "experiments/qwen38-27b-b70/scripts/run-20260827-qwen38-fp8-strict-profile-attempt.sh",
    "experiments/qwen38-27b-b70/scripts/summarize-20260827-qwen38-fp8-strict-matrix.py",
    "experiments/qwen38-27b-b70/notes/2026-08-27-qwen38-fp8-tp1-strict-target-control-result.md",
    "experiments/qwen38-27b-b70/data/2026-08-27-qwen38-fp8-tp1-strict-target-control-comparison.json",
    "experiments/qwen38-27b-b70/notes/2026-08-28-qwen38-fp8-deterministic-eager-baseline-and-compiled-closure.md",
    "experiments/qwen38-27b-b70/data/2026-08-28-qwen38-fp8-deterministic-eager-baseline.json",
    "experiments/qwen38-27b-b70/data/2026-08-28-qwen38-fp8-deterministic-compiled-work-wait.json",
    "experiments/qwen38-27b-b70/data/2026-08-28-qwen38-fp8-deterministic-compiled-work-wait-comparison.json",
    "experiments/qwen38-27b-b70/notes/2026-08-28-qwen38-fp8-mtp1-deterministic-r32-result.md",
    "experiments/qwen38-27b-b70/data/2026-08-28-qwen38-fp8-mtp1-deterministic-r32.json",
    "experiments/qwen38-27b-b70/notes/2026-09-01-qwen38-fp8-public-reproduction-audit.md",
    "experiments/qwen38-27b-b70/data/2026-09-01-qwen38-fp8-explicit-deterministic-matrix-r54-prereg.json",
    "experiments/qwen38-27b-b70/data/2026-09-01-qwen38-fp8-explicit-deterministic-matrix-r54-result.json",
    "experiments/qwen38-27b-b70/data/2026-09-01-qwen38-fp8-clean-rebuild-r55c-result.json",
    "experiments/qwen38-27b-b70/data/2026-09-01-qwen38-fp8-real-content-depth-r56-prereg.json",
    "experiments/qwen38-27b-b70/data/2026-09-01-qwen38-fp8-real-content-depth-r56-diagnostic-result.json",
    "experiments/qwen38-27b-b70/notes/2026-09-01-qwen38-fp8-real-content-depth-r56-diagnostic.md",
    "experiments/qwen38-27b-b70/data/2026-09-01-qwen38-fp8-mtp1-prefill-budget-r57-prereg.json",
    "experiments/qwen38-27b-b70/data/2026-09-01-qwen38-fp8-mtp1-prefill-budget-r57-result.json",
    "experiments/qwen38-27b-b70/data/2026-09-01-qwen38-fp8-mtp1-xpugraph-r58-prereg.json",
    "experiments/qwen38-27b-b70/data/2026-09-01-qwen38-fp8-mtp1-xpugraph-r58-result.json",
    "experiments/qwen38-27b-b70/notes/2026-09-01-qwen38-fp8-mtp1-xpugraph-r58-negative.md",
    "experiments/qwen38-27b-b70/scripts/run-20260901-qwen38-fp8-mtp1-xpugraph-r58-server.sh",
    "experiments/qwen38-27b-b70/data/2026-09-01-qwen38-fp8-mtp1-profiler-r59-result.json",
    "experiments/qwen38-27b-b70/notes/2026-09-01-qwen38-fp8-mtp1-profiler-r59.md",
    "experiments/qwen38-27b-b70/data/2026-09-01-qwen38-fp8-mtp1-compiled-allreduce-r60-prereg.json",
    "experiments/qwen38-27b-b70/data/2026-09-01-qwen38-fp8-mtp1-compiled-allreduce-r60-result.json",
    "experiments/qwen38-27b-b70/notes/2026-09-01-qwen38-fp8-mtp1-compiled-allreduce-r60-negative.md",
    "experiments/qwen38-27b-b70/patches/vllm-qwen38-xpu-compiled-allreduce-custom-op-r60-20260901.patch",
    "experiments/qwen38-27b-b70/scripts/run-20260901-qwen38-fp8-mtp1-compiled-allreduce-r60-server.sh",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/Dockerfile.compiled-allreduce-custom-op-r60",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/build-compiled-allreduce-custom-op-r60-image.sh",
    "experiments/qwen38-27b-b70/data/2026-09-01-qwen38-fp8-mtp1-shape-profiler-r61-result.json",
    "experiments/qwen38-27b-b70/notes/2026-09-01-qwen38-fp8-mtp1-shape-profiler-r61.md",
    "experiments/qwen38-27b-b70/scripts/summarize-torch-xpu-shape-trace.py",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-mtp1-clean-rebuild-r55c",
    "experiments/qwen38-27b-b70/scripts/validate-20260901-qwen38-fp8-explicit-deterministic-r54.py",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-mtp0-explicit-deterministic-r54a-r50",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-mtp0-explicit-deterministic-r54c-r50",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-mtp1-explicit-deterministic-r53a",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-mtp1-explicit-deterministic-r53b",
    "experiments/qwen38-27b-b70/data/2026-09-01-qwen38-fp8-mtp1-serial-linear-r48-prereg.json",
    "experiments/qwen38-27b-b70/data/2026-09-01-qwen38-fp8-mtp1-serial-attention-r49-prereg.json",
    "experiments/qwen38-27b-b70/patches/vllm-qwen38-fp8-mtp1-packed-linear-serial-r48-20260901.patch",
    "experiments/qwen38-27b-b70/patches/vllm-qwen38-xpu-serial-spec-flash-attn-r38-20260828.patch",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/Dockerfile.mtp1-serial-linear",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/build-mtp1-serial-linear-image.sh",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-mtp0-compiler-stable-r47a",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-mtp1-serial-linear-r48a",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-mtp1-serial-attention-r49a",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-mtp0-compiler-stable-r47a/performance.json",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-mtp1-serial-linear-r48a/performance.json",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-mtp1-serial-linear-r48a/compare-target-r47a.json",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-mtp1-serial-attention-r49a/performance.json",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-mtp1-serial-attention-r49a/compare-target-r47a.json",
    "experiments/qwen38-27b-b70/notes/2026-08-28-qwen38-fp8-w8a16-mtp1-exact-depth-r33-result.md",
    "experiments/qwen38-27b-b70/data/2026-08-28-qwen38-fp8-w8a16-mtp1-exact-depth-r33-result.json",
    "experiments/qwen38-27b-b70/scripts/validate-20260828-qwen38-fp8-w8a16-mtp1-exact-depth-r33.py",
    "experiments/qwen38-27b-b70/data/2026-08-28-qwen38-fp8-dynamic-exactness-r34-r38b-summary.json",
    "experiments/qwen38-27b-b70/data/2026-08-28-qwen38-gemma-rmsnorm-mtp1-serial-r30-operator-proof.json",
    "experiments/qwen38-27b-b70/patches/vllm-qwen38-xpu-gemma-rmsnorm-mtp1-serial-exact-r30-20260828.patch",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-deterministic-compiled-workwait-20260828-r15a/performance.json",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-deterministic-compiled-workwait-20260828-r15a/canaries.json",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-deterministic-compiled-workwait-20260828-r15b/performance.json",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-deterministic-compiled-workwait-20260828-r15b/canaries.json",
    "experiments/qwen38-27b-b70/patches/vllm-qwen38-xpu-deterministic-gdn-ba-state-20260828.patch",
    "experiments/qwen38-27b-b70/patches/vllm-qwen38-xpu-compiled-gdn-state-ccl-wait-20260828.patch",
    "scripts/compare-strict-attempt-outputs.py",
    "scripts/bench-openai-token-depth-suite.py",
    "scripts/neural-download-canaries.py",
    "data/qwen27-exact-depth/qwen38-bce40ca-mixed-content-depth-v1.json",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/model-direct.json",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/preflight-evidence-20260821.json",
    "experiments/qwen38-27b-b70/docker/Dockerfile.fp8-kernel-1e90-r13",
    "experiments/qwen38-27b-b70/notes/2026-08-16-official-fp8-vllm-graph-tp2.md",
    "experiments/qwen38-27b-b70/data/2026-08-16-official-fp8-vllm-graph-tp2.json",
    "experiments/qwen38-27b-b70/patches/vllm-qwen38-fp8-block-w8a16-20260826.patch",
    "experiments/qwen38-27b-b70/patches/vllm-xpu-kernels-qwen38-dynamic-mtp-active-width-20260826.patch",
    "experiments/qwen38-27b-b70/patches/vllm-qwen38-dynamic-mtp-mamba-active-allocation-20260826.patch",
    "experiments/qwen38-27b-b70/docker/Dockerfile.fp8-w8a16-dynamic-mtp-active-width-r1",
    "experiments/qwen38-27b-b70/docker/Dockerfile.fp8-w8a16-dynamic-mamba-allocation-r1",
    "experiments/qwen38-27b-b70/scripts/build-w8a16-dynamic-mamba-allocation-image.sh",
    "experiments/qwen38-27b-b70/notes/2026-08-27-qwen38-fp8-w8a16-dynamic-mamba-r5-replication-result.md",
    "experiments/qwen38-27b-b70/data/2026-08-27-qwen38-fp8-w8a16-mtp2-dynamic-mamba-r5-summary.json",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-w8a16-mtp2-dynamic-mamba-20260826-r4/bench/c64-screen.json",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-w8a16-mtp2-dynamic-mamba-20260827-r5/c64-screen.json",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-w8a16-mtp2-dynamic-mamba-20260827-r5/c64-quality-512.json",
    "experiments/qwen38-27b-b70/notes/2026-08-27-qwen38-fp8-w8a16-dynamic-mtp3-r7-replication-result.md",
    "experiments/qwen38-27b-b70/data/2026-08-27-qwen38-fp8-w8a16-mtp3-dynamic-mtp1-r7-summary.json",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-w8a16-mtp3-dynamic-mtp1-20260827-r6/bench/c64-screen.json",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-w8a16-mtp3-dynamic-mtp1-20260827-r6/bench/c64-quality-512.json",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-w8a16-mtp3-dynamic-mtp1-20260827-r7/bench/c64-screen.json",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-w8a16-mtp3-dynamic-mtp1-20260827-r7/bench/c64-quality-512.json",
    "experiments/qwen38-27b-b70/notes/2026-08-27-qwen38-fp8-w8a16-dynamic-mtp4-r9-replication-result.md",
    "experiments/qwen38-27b-b70/data/2026-08-27-qwen38-fp8-w8a16-mtp4-dynamic-mtp1-r9-summary.json",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-w8a16-mtp4-dynamic-mtp1-20260827-r8/c64-screen.json",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-w8a16-mtp4-dynamic-mtp1-20260827-r8/c64-quality-512.json",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-w8a16-mtp4-dynamic-mtp1-20260827-r9/c64-screen.json",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-w8a16-mtp4-dynamic-mtp1-20260827-r9/c64-quality-512.json",
    "experiments/qwen38-27b-b70/notes/2026-08-27-qwen38-fp8-w8a16-dynamic-mtp5-r11-replication-result.md",
    "experiments/qwen38-27b-b70/data/2026-08-27-qwen38-fp8-w8a16-mtp5-dynamic-mtp1-r11-summary.json",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-w8a16-mtp5-dynamic-mtp1-20260827-r10/c64-screen.json",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-w8a16-mtp5-dynamic-mtp1-20260827-r10/c64-quality-512.json",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-w8a16-mtp5-dynamic-mtp1-20260827-r11/c64-screen.json",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-w8a16-mtp5-dynamic-mtp1-20260827-r11/c64-quality-512.json",
    "experiments/qwen38-27b-b70/notes/2026-08-27-qwen38-fp8-w8a16-dynamic-mtp7-r14-replication-result.md",
    "experiments/qwen38-27b-b70/data/2026-08-27-qwen38-fp8-w8a16-mtp7-dynamic-mtp1-r14-summary.json",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-w8a16-mtp7-dynamic-mtp1-20260827-r13/c64-screen.json",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-w8a16-mtp7-dynamic-mtp1-20260827-r13/c64-quality-512.json",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-w8a16-mtp7-dynamic-mtp1-20260827-r14/c64-screen.json",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-w8a16-mtp7-dynamic-mtp1-20260827-r14/c64-quality-512.json",
    "experiments/qwen38-27b-b70/notes/2026-08-27-qwen38-fp8-w8a16-dynamic-mtp8-r15-result.md",
    "experiments/qwen38-27b-b70/data/2026-08-27-qwen38-fp8-w8a16-mtp8-dynamic-mtp1-r15-summary.json",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-w8a16-mtp8-dynamic-mtp1-20260827-r15/c64-screen.json",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-w8a16-mtp8-dynamic-mtp1-20260827-r15/c64-quality-512.json",
    "experiments/qwen38-27b-b70/notes/2026-08-27-qwen38-fp8-w8a16-dynamic-mtp8-r16-replication-result.md",
    "experiments/qwen38-27b-b70/data/2026-08-27-qwen38-fp8-w8a16-mtp8-dynamic-mtp1-r16-summary.json",
    "experiments/qwen38-27b-b70/notes/2026-08-27-qwen38-fp8-w8a16-mtp8-realistic-cold-result.md",
    "experiments/qwen38-27b-b70/data/2026-08-27-qwen38-fp8-w8a16-mtp8-realistic-cold-summary.json",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-w8a16-mtp8-realistic-cold-20260827/r1/realistic-suite.json",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-w8a16-mtp8-realistic-cold-20260827/r1/server.log",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-w8a16-mtp8-realistic-cold-20260827/r2/realistic-suite.json",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-w8a16-mtp8-realistic-cold-20260827/r2/server.log",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-w8a16-mtp8-realistic-cold-20260827/identity.env",
    "data/localmaxxing-qwen38-27b-official-fp8-w8a16-tp2-dynamic-mtp8-realistic-58tok-20260827.queue.json",
    "data/localmaxxing-qwen38-27b-official-fp8-w8a16-tp2-mtp1-strict-51tok-20260901.queue.json",
    "data/localmaxxing-responses/qwen38-27b-official-fp8-w8a16-tp2-mtp1-strict-20260901.response.json",
    "data/localmaxxing-responses/qwen38-27b-official-fp8-w8a16-tp2-dynamic-mtp8-realistic-20260827.response.json",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-w8a16-mtp8-dynamic-mtp1-20260827-r16/c64-screen.json",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-w8a16-mtp8-dynamic-mtp1-20260827-r16/c64-quality-512.json",
    "experiments/qwen38-27b-b70/notes/2026-08-27-qwen38-fp8-w8a16-dynamic-mtp9-r17-negative.md",
    "experiments/qwen38-27b-b70/data/2026-08-27-qwen38-fp8-w8a16-mtp9-dynamic-mtp1-r17-summary.json",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-w8a16-mtp9-dynamic-mtp1-20260827-r17/single-p40-o128.json",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-w8a16-mtp9-dynamic-mtp1-20260827-r17/c64-screen.json",
    "experiments/qwen38-27b-b70/notes/2026-08-27-qwen38-fp8-w8a16-dynamic-mtp9-p64-r18-negative.md",
    "experiments/qwen38-27b-b70/data/2026-08-27-qwen38-fp8-w8a16-mtp9-p64-dynamic-mtp1-r18-summary.json",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-w8a16-mtp9-p64-dynamic-mtp1-20260827-r18/single-p40-o128.json",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-w8a16-mtp9-p64-dynamic-mtp1-20260827-r18/c64-screen.json",
    "experiments/qwen38-27b-b70/notes/2026-08-27-qwen38-fp8-w8a16-dynamic-mtp9-latch-r19-negative.md",
    "experiments/qwen38-27b-b70/data/2026-08-27-qwen38-fp8-w8a16-mtp9-latch-r19-summary.json",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-w8a16-mtp9-latch-dynamic-mtp1-20260827-r19/single-p40-o128.json",
    "experiments/qwen38-27b-b70/notes/2026-08-27-qwen38-fp8-w8a16-dynamic-mtp9-latch2-r20-negative.md",
    "experiments/qwen38-27b-b70/data/2026-08-27-qwen38-fp8-w8a16-mtp9-latch2-r20-summary.json",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-w8a16-mtp9-latch2-dynamic-mtp1-20260827-r20/single-p40-o128.json",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-w8a16-mtp9-latch2-dynamic-mtp1-20260827-r20/c64-screen.json",
    "experiments/qwen38-27b-b70/patches/vllm-qwen38-dynamic-sd-busy-period-latch-20260827.patch",
    "experiments/qwen38-27b-b70/patches/vllm-qwen38-dynamic-sd-busy-period-latch-reset-after-free-20260827.patch",
    "experiments/qwen38-27b-b70/docker/Dockerfile.fp8-w8a16-dynamic-sd-latch-r1",
    "experiments/qwen38-27b-b70/docker/Dockerfile.fp8-w8a16-dynamic-sd-latch-r2",
    "experiments/qwen38-27b-b70/scripts/build-w8a16-dynamic-sd-latch-image.sh",
    "experiments/qwen38-27b-b70/scripts/build-w8a16-dynamic-sd-latch-r2-image.sh",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/run-w8a16-mtp9-latch-dynamic-mtp1-r19-server.sh",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/bench-w8a16-mtp9-latch-dynamic-mtp1-r19.sh",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/run-w8a16-mtp9-latch2-dynamic-mtp1-r20-server.sh",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/bench-w8a16-mtp9-latch2-dynamic-mtp1-r20.sh",
    "experiments/qwen38-27b-b70/notes/2026-08-27-qwen38-fp8-w8a16-dynamic-mtp8-c2-r21-negative.md",
    "experiments/qwen38-27b-b70/data/2026-08-27-qwen38-fp8-w8a16-mtp8-c2-r21-summary.json",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-w8a16-mtp8-c2-dynamic-mtp1-20260827-r21/single-p40-o128.json",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-w8a16-mtp8-c2-dynamic-mtp1-20260827-r21/c64-screen.json",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/run-w8a16-mtp8-c2-dynamic-mtp1-r21-server.sh",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/bench-w8a16-mtp8-c2-dynamic-mtp1-r21.sh",
    "experiments/qwen38-27b-b70/notes/2026-08-26-qwen38-fp8-block-w8a16-tp2-p128-result.md",
    "experiments/qwen38-27b-b70/data/2026-08-26-qwen38-fp8-block-w8a16-tp2-p128-summary.json",
    "experiments/qwen38-27b-b70/notes/2026-08-26-qwen38-fp8-block-w8a16-mtp1-tp2-result.md",
    "experiments/qwen38-27b-b70/data/2026-08-26-qwen38-fp8-block-w8a16-mtp1-tp2-summary.json",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-block-w8a16-mtp1-tp2-p128-screen-20260826-r1/mbt512-single-p40-o128.json",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-block-w8a16-mtp1-tp2-p128-screen-20260826-r1/mbt512-replay-c8.json",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-block-w8a16-mtp1-tp2-p128-screen-20260826-r1/mbt512-replay-c16.json",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-block-w8a16-mtp1-tp2-p128-screen-20260826-r1/mbt512-replay-c32.json",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-block-w8a16-mtp1-tp2-p128-screen-20260826-r1/mbt512-c64-replication-x3.json",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-block-w8a16-mtp1-tp2-p128-screen-20260826-r1/mbt512-c128-replication-x3.json",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-block-w8a16-mtp1-tp2-p128-screen-20260826-r1/mbt512-sequential-quality.json",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-block-w8a16-mtp1-tp2-p128-screen-20260826-r1/mbt512-c64-quality-512.json",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-block-w8a16-mtp1-tp2-p128-screen-20260826-r1/r121-server.log",
    "experiments/qwen38-27b-b70/notes/2026-08-26-qwen38-fp8-block-w8a16-mtp2-reuse-result.md",
    "experiments/qwen38-27b-b70/data/2026-08-26-qwen38-fp8-block-w8a16-mtp2-reuse-summary.json",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-block-w8a16-mtp2-reuse-screen-20260826-r1/single-p40-o128.json",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-block-w8a16-mtp2-reuse-screen-20260826-r1/c64-screen.json",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-block-w8a16-mtp2-reuse-screen-20260826-r1/sequential-quality.json",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-block-w8a16-mtp2-reuse-mbt768-screen-20260826-r1/c64-screen.json",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-block-w8a16-mtp2-reuse-mbt768-screen-20260826-r1/sequential-quality.json",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-block-w8a16-tp2-p128-20260826-r1/w8a16-c128-measured-x5.json",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-block-w8a16-tp2-p128-20260826-r1/w8a16-c128-quality-1024.json",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-block-w8a16-tp2-http-depth-20260826-r2/summary.json",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-tp2-http-depth-20260826-r1-attempt1/summary.json",
    "experiments/qwen38-27b-b70/notes/2026-08-26-qwen38-fp8-tp2-http-depth-r1-result.md",
    "experiments/qwen38-27b-b70/data/2026-08-26-qwen38-fp8-tp2-http-concurrency-r3-prereg.json",
    "experiments/qwen38-27b-b70/notes/2026-08-26-qwen38-fp8-tp2-http-concurrency-r3-preregistration.md",
    "experiments/qwen38-27b-b70/data/2026-08-26-qwen38-fp8-tp2-http-concurrency-r3-result.json",
    "experiments/qwen38-27b-b70/notes/2026-08-26-qwen38-fp8-tp2-http-concurrency-r3-result.md",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-tp2-http-concurrency-20260826-r3-attempt1/result.json",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-tp2-http-concurrency-20260826-r3-attempt1/qualification.json",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-tp2-http-concurrency-20260826-r3-attempt2/result.json",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-tp2-http-concurrency-20260826-r3-attempt2/qualification.json",
    "experiments/qwen38-27b-b70/data/2026-08-26-qwen38-fp8-tp2-http-p32-confirmation-r3-prereg.json",
    "experiments/qwen38-27b-b70/notes/2026-08-26-qwen38-fp8-tp2-http-p32-confirmation-r3-preregistration.md",
    "experiments/qwen38-27b-b70/data/2026-08-26-qwen38-fp8-tp2-http-p32-confirmation-r3-result.json",
    "experiments/qwen38-27b-b70/notes/2026-08-26-qwen38-fp8-tp2-http-p32-confirmation-r3-result.md",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-tp2-http-p32-confirmation-20260826-r3-attempt1/result.json",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-tp2-http-p32-confirmation-20260826-r3-attempt1/qualification.json",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-tp2-http-p32-confirmation-20260826-r3-attempt2/result.json",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-tp2-http-p32-confirmation-20260826-r3-attempt2/qualification.json",
    "experiments/qwen38-27b-b70/data/2026-08-26-qwen38-fp8-tp2-http-p64-confirmation-r5-prereg.json",
    "experiments/qwen38-27b-b70/notes/2026-08-26-qwen38-fp8-tp2-http-p64-confirmation-r5-preregistration.md",
    "experiments/qwen38-27b-b70/data/2026-08-26-qwen38-fp8-tp2-http-p64-confirmation-r5-result.json",
    "experiments/qwen38-27b-b70/notes/2026-08-26-qwen38-fp8-tp2-http-p64-confirmation-r5-result.md",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-tp2-http-p64-confirmation-20260826-r5-attempt1/result.json",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-tp2-http-p64-confirmation-20260826-r5-attempt1/qualification.json",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-tp2-http-p64-confirmation-20260826-r5-attempt2/result.json",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-tp2-http-p64-confirmation-20260826-r5-attempt2/qualification.json",
    "experiments/qwen38-27b-b70/data/2026-08-26-qwen38-fp8-tp2-http-p64-p2p1-confirmation-r10-prereg.json",
    "experiments/qwen38-27b-b70/notes/2026-08-26-qwen38-fp8-tp2-http-p64-p2p1-confirmation-r10-result.md",
    "experiments/qwen38-27b-b70/data/2026-08-26-qwen38-fp8-tp2-http-p64-p2p1-confirmation-r10-result.json",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-tp2-http-p64-p2p1-confirmation-20260826-r10-attempt1/result.json",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-tp2-http-p64-p2p1-confirmation-20260826-r10-attempt1/qualification.json",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-tp2-http-p64-p2p1-confirmation-20260826-r10-attempt2/result.json",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-tp2-http-p64-p2p1-confirmation-20260826-r10-attempt2/qualification.json",
    "experiments/qwen38-27b-b70/data/qwen38-fp8-tp2-http-concurrency-oracle-pilot-20260826-r1-attempt1/oracle-digests.json",
    "experiments/qwen38-27b-b70/data/2026-08-25-qwen38-q4km-tp2-http-smallctx-suite.json",
    "scripts/bench-openai-concurrency-oracle.py",
    "scripts/qualify-openai-concurrency-attempt.py",
    "scripts/summarize-http-concurrency-attempts.py",
    "experiments/qwen38-27b-b70/patches/vllm-xpu-kernels-qwen38-dynamic-active-width-serial-gdn-r35-20260828.patch",
    "experiments/qwen38-27b-b70/patches/vllm-xpu-kernels-qwen38-gdn-split-serial-gates-r50-20260901.patch",
    "experiments/qwen38-27b-b70/patches/vllm-qwen38-fp8-draft-only-int4-lm-head-r62-20260901.patch",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/Dockerfile.draft-int4-r62",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/build-draft-int4-r62-image.sh",
    "experiments/qwen38-27b-b70/scripts/run-20260901-qwen38-fp8-mtp1-draft-int4-r62-server.sh",
    "experiments/qwen38-27b-b70/data/2026-09-01-qwen38-fp8-mtp1-draft-int4-r62-prereg.json",
    "experiments/qwen38-27b-b70/data/2026-09-01-qwen38-fp8-mtp1-draft-int4-r62-diagnostic-result.json",
    "experiments/qwen38-27b-b70/notes/2026-09-01-qwen38-fp8-mtp1-draft-int4-r62-diagnostic.md",
    "experiments/qwen38-27b-b70/data/2026-09-02-qwen38-fp8-mtp1-draft-int4-r62-cleanboot-r119-prereg.json",
    "experiments/qwen38-27b-b70/data/2026-09-02-qwen38-fp8-mtp1-draft-int4-r62-cleanboot-r119-result.json",
    "experiments/qwen38-27b-b70/notes/2026-09-02-qwen38-fp8-mtp1-draft-int4-r62-cleanboot-r119-promotion.md",
    "experiments/qwen38-27b-b70/scripts/probe-qwen38-fp8-c1-prefill-length-determinism.py",
    "experiments/qwen38-27b-b70/notes/2026-09-02-qwen38-fp8-mtp1-c2-identity-review-kernel-census-cr1.md",
    "experiments/qwen38-27b-b70/data/2026-09-01-qwen38-fp8-mtp1-draft-int4-r63-concurrency-prereg.json",
    "experiments/qwen38-27b-b70/data/2026-09-01-qwen38-fp8-mtp1-draft-int4-r63-control-prereg.json",
    "experiments/qwen38-27b-b70/data/2026-09-01-qwen38-fp8-mtp1-draft-int4-r63-concurrency-result.json",
    "experiments/qwen38-27b-b70/notes/2026-09-01-qwen38-fp8-mtp1-draft-int4-r63-concurrency-negative.md",
    "scripts/test_bench_openai_concurrency_oracle.py",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/build-fixed-k-w8a16-r139-published-image.sh",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/build-fixed-k-w8a16-r139-image.sh",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/Dockerfile.fixed-k-w8a16-r139",
    "scripts/build-vllm-xpu-kernels-xpu-c-only.sh",
    "experiments/qwen38-27b-b70/scripts/run-20260902-qwen38-fp8-mtp1-fixed-k-r139-server.sh",
    "experiments/qwen38-27b-b70/scripts/run-20260902-qwen38-fp8-mtp0-fixed-k-r139-server.sh",
    "experiments/qwen38-27b-b70/patches/onednn-qwen38-w8a16-fixed-k-align16-r137a-20260902.patch",
    "experiments/qwen38-27b-b70/patches/onednn-qwen38-w8a16-c-default-align-r137b-20260902.patch",
    "experiments/qwen38-27b-b70/data/2026-09-02-qwen38-fp8-mtp1-fixed-k-serving-r139-prereg.json",
    "experiments/qwen38-27b-b70/data/2026-09-02-qwen38-fp8-mtp1-fixed-k-serving-r139-result.json",
    "experiments/qwen38-27b-b70/data/2026-09-02-qwen38-fp8-mtp1-fixed-k-regenerated-oracle-r147-prereg.json",
    "experiments/qwen38-27b-b70/data/2026-09-02-qwen38-fp8-mtp1-fixed-k-regenerated-oracle-r147-result.json",
    "experiments/qwen38-27b-b70/data/2026-09-02-qwen38-fp8-mtp0-fixed-k-probe-ladder-r147c-prereg.json",
    "experiments/qwen38-27b-b70/data/2026-09-02-qwen38-fp8-mtp0-fixed-k-probe-ladder-r147c-result.json",
    "experiments/qwen38-27b-b70/data/2026-09-02-qwen38-fp8-fixed-k-ladder-batched-2048-r148-result.json",
    "experiments/qwen38-27b-b70/data/2026-09-02-qwen38-fp8-lm-head-chunk-rows-r149-result.json",
    "experiments/qwen38-27b-b70/notes/2026-09-02-qwen38-fp8-fixed-k-identity-ladders-r147-r149.md",
    "experiments/qwen38-27b-b70/notes/2026-09-02-qwen38-fp8-w8a16-row-invariant-kernel-track-r120-r146.md",
    "experiments/qwen38-27b-b70/scripts/run-20260902-qwen38-fp8-mtp1-fixed-k-regenerated-oracle-r147.sh",
    "experiments/qwen38-27b-b70/scripts/run-20260902-qwen38-fp8-mtp0-fixed-k-probe-ladder-r147c.sh",
    "experiments/qwen38-27b-b70/data/2026-09-02-qwen38-fp8-fixed-k-real-content-depth-r150-result.json",
    "experiments/qwen38-27b-b70/data/2026-09-03-qwen38-fp8-gdn-split-mixed-promotion-r156f-result.json",
    "experiments/qwen38-27b-b70/scripts/run-20260903-qwen38-fp8-mtp1-split-mixed-r156-server.sh",
    "experiments/qwen38-27b-b70/scripts/run-20260903-qwen38-fp8-mtp0-split-mixed-r156-server.sh",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/build-gdn-split-mixed-r156-image.sh",
    "repro/qwen38-27b-fp8-vllm-tp2-asrock-b70/Dockerfile.gdn-split-mixed-r156",
    "tools/validate-xpu-gdn-split-mixed-r156.py",
    "experiments/qwen38-27b-b70/patches/vllm-qwen38-xpu-gdn-split-mixed-step-r156-20260903.patch",
    "experiments/qwen38-27b-b70/data/2026-09-03-qwen38-fp8-gdn-split-mixed-r156-prereg.json",
    "experiments/qwen38-27b-b70/data/2026-09-03-qwen38-fp8-gdn-split-mixed-r156-result.json",
    "experiments/qwen38-27b-b70/data/2026-09-03-qwen38-fp8-gdn-fa-mixed-step-census-r155-result.json",
    "experiments/qwen38-27b-b70/notes/2026-09-03-qwen38-fp8-c32-identity-source-census-r151-r162.md",
    "experiments/qwen38-27b-b70/data/2026-09-03-qwen38-fp8-r187-whole-graph-depth2-result.json",
    "experiments/qwen38-27b-b70/data/2026-09-03-qwen38-fp8-r188-whole-graph-depth1-result.json",
    "experiments/qwen38-27b-b70/notes/2026-09-03-qwen38-fp8-mtp2-no-splitting-full-campaign-r187-result.md",
    "experiments/qwen38-27b-b70/notes/2026-09-03-qwen38-fp8-mtp1-depth1-no-splitting-r188-result.md",
    "experiments/qwen38-27b-b70/notes/2026-09-03-qwen38-fp8-mtp2-phantom-inductor-knobs-r184-result.md",
    "experiments/qwen38-27b-b70/scripts/run-20260903-qwen38-fp8-mtp1-whole-graph-r187-server.sh",
    "experiments/qwen38-27b-b70/scripts/run-20260903-qwen38-fp8-mtp0-whole-graph-r187-server.sh",
    "experiments/qwen38-27b-b70/scripts/run-20260903-qwen38-fp8-mtp2-whole-graph-r187-server.sh",
    "experiments/qwen38-27b-b70/data/2026-09-03-qwen38-fp8-r187-real-content-depth-r189-result.json",
    "experiments/qwen38-27b-b70/data/2026-09-03-qwen38-fp8-r191-whole-graph-depth3-result.json",
    "experiments/qwen38-27b-b70/notes/2026-09-03-qwen38-fp8-mtp3-whole-graph-r191-result.md",
    "experiments/qwen38-27b-b70/scripts/run-20260903-qwen38-fp8-mtp3-whole-graph-r187-server.sh",
    "experiments/qwen38-27b-b70/notes/2026-09-03-qwen38-fp8-r187-real-content-depth-r189-result.md",
    "experiments/qwen38-27b-b70/data/2026-09-04-qwen38-fp8-r187-mtp2-real-content-depth-r195-result.json",
    "experiments/qwen38-27b-b70/data/2026-09-04-qwen38-fp8-r187-mtp3-real-content-depth-r196-result.json",
    "experiments/qwen38-27b-b70/notes/2026-09-04-qwen38-fp8-r187-depth-curves-mtp2-mtp3-r195-r196.md",
    "experiments/qwen38-27b-b70/data/2026-09-04-qwen38-fp8-r197-whole-graph-depth4-result.json",
    "experiments/qwen38-27b-b70/notes/2026-09-04-qwen38-fp8-mtp4-whole-graph-r197-result.md",
    "experiments/qwen38-27b-b70/scripts/run-20260903-qwen38-fp8-mtp4-whole-graph-r187-server.sh",
    "experiments/qwen38-27b-b70/data/2026-09-04-qwen38-fp8-r200-whole-graph-depth5-strict-result.json",
    "experiments/qwen38-27b-b70/notes/2026-09-04-qwen38-fp8-mtp-depth-turnover-r200-r203.md",
    "experiments/qwen38-27b-b70/scripts/run-20260903-qwen38-fp8-mtp5-whole-graph-r187-server.sh",
    "experiments/qwen38-27b-b70/data/2026-09-04-qwen38-fp8-r202-r203-depth6-depth7-result.json"
   ],
   "performance_profiles": [
    {
     "id": "http-r156-mtp1-identity-qualified-aggregate-vs-concurrent-users",
     "label": "R187 FP8 TP2 MTP1 identity-qualified aggregate decode vs concurrent users (c1-c16)",
     "metric": "aggregate_decode",
     "unit": "tok/s",
     "x_metric": "concurrent_sequences",
     "x_label": "Concurrent HTTP users",
     "scope": "Measured vLLM HTTP completions on two B70s with the R187 image (MTP1, FP16 verifier, draft-only INT4 head), FP16 activations/KV, 64 service slots, 256-token total request capacity, max_num_batched_tokens=512, cache disabled, 128 returned raw token IDs per response on the 64-prompt small-context suite. Every point shown is output-identity-qualified: each concurrent output equals its own sequential oracle byte for byte (harness --require-output-identity). c32 and c64 were measured but are withheld: 1/32 and 8/64 near-tie prompts took a different valid branch. One server, one pass per point; not a promoted concurrency speed record.",
     "evidence": "experiments/qwen38-27b-b70/data/2026-09-03-qwen38-fp8-r188-whole-graph-depth1-result.json",
     "points": [
      {
       "concurrent_sequences": 1,
       "value": 53.5361089749907,
       "samples": 1,
       "per_user_value": 53.5361089749907
      },
      {
       "concurrent_sequences": 2,
       "value": 72.1595380864749,
       "samples": 1,
       "per_user_value": 36.07976904323745
      },
      {
       "concurrent_sequences": 4,
       "value": 193.3635682197048,
       "samples": 1,
       "per_user_value": 48.3408920549262
      },
      {
       "concurrent_sequences": 8,
       "value": 353.233907423836,
       "samples": 1,
       "per_user_value": 44.1542384279795
      },
      {
       "concurrent_sequences": 16,
       "value": 477.6588917840359,
       "samples": 1,
       "per_user_value": 29.853680736502245
      }
     ]
    },
    {
     "id": "http-r156-mtp0-identity-qualified-aggregate-vs-concurrent-users",
     "label": "R187 FP8 TP2 MTP0 identity-qualified aggregate decode vs concurrent users (c1-c64)",
     "metric": "aggregate_decode",
     "unit": "tok/s",
     "x_metric": "concurrent_sequences",
     "x_label": "Concurrent HTTP users",
     "scope": "Measured vLLM HTTP completions on two B70s with the R187 image (MTP0, target only), FP16 activations/KV, 64 service slots, 256-token total request capacity, max_num_batched_tokens=512, cache disabled, 128 returned raw token IDs per response on the 64-prompt small-context suite. Every point shown is output-identity-qualified: each concurrent output equals its own sequential oracle byte for byte (harness --require-output-identity). All seven points including c32 and c64 passed 64/64. One server, one pass per point; not a promoted concurrency speed record.",
     "evidence": "experiments/qwen38-27b-b70/data/2026-09-03-qwen38-fp8-r187-whole-graph-depth2-result.json",
     "points": [
      {
       "concurrent_sequences": 1,
       "value": 33.375431781866325,
       "samples": 1,
       "per_user_value": 33.375431781866325
      },
      {
       "concurrent_sequences": 2,
       "value": 64.35204956704715,
       "samples": 1,
       "per_user_value": 32.17602478352357
      },
      {
       "concurrent_sequences": 4,
       "value": 123.93764901565976,
       "samples": 1,
       "per_user_value": 30.98441225391494
      },
      {
       "concurrent_sequences": 8,
       "value": 234.38391319439305,
       "samples": 1,
       "per_user_value": 29.29798914929913
      },
      {
       "concurrent_sequences": 16,
       "value": 419.9261644481194,
       "samples": 1,
       "per_user_value": 26.24538527800746
      },
      {
       "concurrent_sequences": 32,
       "value": 672.0107142920494,
       "samples": 1,
       "per_user_value": 21.000334821626545
      },
      {
       "concurrent_sequences": 64,
       "value": 927.9028136957172,
       "samples": 1,
       "per_user_value": 14.498481463995581
      }
     ]
    },
    {
     "id": "http-decode-vs-active-context",
     "label": "R187 FP8 TP2 MTP0 HTTP decode over exact active context",
     "metric": "decode",
     "unit": "tok/s",
     "x_label": "Exact active prompt tokens",
     "scope": "Measured on the published R187 line (R156 image, whole-graph torch.compile; 2026-09-03, clean boot): one-slot vLLM HTTP completions on two B70s at exactly 2K, 4K, 8K, 16K, 24K, and 32K active context, unrepeated technical prose, Python, and structured documents (three requests per depth, median shown), 128 output tokens, cache zero, canaries before and after. MTP0 (target only). No value is interpolated.",
     "evidence": "experiments/qwen38-27b-b70/data/2026-09-03-qwen38-fp8-r187-real-content-depth-r189-result.json",
     "points": [
      {
       "context_tokens": 2048,
       "value": 33.00186467133787,
       "samples": 3
      },
      {
       "context_tokens": 4096,
       "value": 32.74192382357966,
       "samples": 3
      },
      {
       "context_tokens": 8192,
       "value": 31.896309461629123,
       "samples": 3
      },
      {
       "context_tokens": 16384,
       "value": 31.181138974276305,
       "samples": 3
      },
      {
       "context_tokens": 24576,
       "value": 30.44503893662358,
       "samples": 3
      },
      {
       "context_tokens": 32768,
       "value": 29.77809869783394,
       "samples": 3
      }
     ]
    },
    {
     "id": "http-effective-prompt-proxy-vs-active-context",
     "label": "R139 MTP0 effective prompt throughput (prompt tokens divided by HTTP TTFT)",
     "metric": "prefill",
     "unit": "tok/s",
     "x_label": "Exact submitted prompt tokens",
     "scope": "Measured on the published R139 row-invariant W8A16 image (2026-09-02, clean boot): one-slot vLLM HTTP completions on two B70s at exactly 2K, 4K, 8K, 16K, 24K, and 32K active context, unrepeated technical prose, Python, and structured documents (three requests per depth, median shown), 128 output tokens, cache zero, canaries before and after. Derived as active context tokens divided by HTTP TTFT; MTP0.",
     "evidence": "experiments/qwen38-27b-b70/data/2026-09-02-qwen38-fp8-fixed-k-real-content-depth-r150-result.json",
     "points": [
      {
       "context_tokens": 2048,
       "value": 3491.2399394447475,
       "samples": 3
      },
      {
       "context_tokens": 4096,
       "value": 3623.1356401914027,
       "samples": 3
      },
      {
       "context_tokens": 8192,
       "value": 3576.999701367745,
       "samples": 3
      },
      {
       "context_tokens": 16384,
       "value": 3432.3821288359386,
       "samples": 3
      },
      {
       "context_tokens": 24576,
       "value": 3296.480331290917,
       "samples": 3
      },
      {
       "context_tokens": 32768,
       "value": 3177.328335630365,
       "samples": 3
      }
     ]
    },
    {
     "id": "http-ttft-vs-active-context",
     "label": "R139 FP8 TP2 MTP0 HTTP TTFT over exact active context",
     "metric": "ttft",
     "unit": "ms",
     "x_label": "Exact active prompt tokens",
     "scope": "Measured on the published R139 row-invariant W8A16 image (2026-09-02, clean boot): one-slot vLLM HTTP completions on two B70s at exactly 2K, 4K, 8K, 16K, 24K, and 32K active context, unrepeated technical prose, Python, and structured documents (three requests per depth, median shown), 128 output tokens, cache zero, canaries before and after. Direct HTTP TTFT, MTP0.",
     "evidence": "experiments/qwen38-27b-b70/data/2026-09-02-qwen38-fp8-fixed-k-real-content-depth-r150-result.json",
     "points": [
      {
       "context_tokens": 2048,
       "value": 586.6110710012435,
       "samples": 3
      },
      {
       "context_tokens": 4096,
       "value": 1130.512464000276,
       "samples": 3
      },
      {
       "context_tokens": 8192,
       "value": 2290.1874990002398,
       "samples": 3
      },
      {
       "context_tokens": 16384,
       "value": 4773.361293999187,
       "samples": 3
      },
      {
       "context_tokens": 24576,
       "value": 7455.224217999785,
       "samples": 3
      },
      {
       "context_tokens": 32768,
       "value": 10313.066998000068,
       "samples": 3
      }
     ]
    },
    {
     "id": "http-mtp1-decode-vs-active-context",
     "label": "R187 FP8 TP2 MTP1 HTTP decode over exact active context",
     "metric": "decode",
     "unit": "tok/s",
     "x_label": "Exact active prompt tokens",
     "scope": "Measured on the published R187 line (R156 image, whole-graph torch.compile; 2026-09-03, clean boot): one-slot vLLM HTTP completions on two B70s at exactly 2K, 4K, 8K, 16K, 24K, and 32K active context, unrepeated technical prose, Python, and structured documents (three requests per depth, median shown), 128 output tokens, cache zero, canaries before and after. MTP1 with FP16 target verifier and draft-only INT4 head; every output matched the same-image MTP0 oracle (18/18 complete arrays).",
     "evidence": "experiments/qwen38-27b-b70/data/2026-09-03-qwen38-fp8-r187-real-content-depth-r189-result.json",
     "points": [
      {
       "context_tokens": 2048,
       "value": 54.613745980688705,
       "samples": 3
      },
      {
       "context_tokens": 4096,
       "value": 55.16720004495228,
       "samples": 3
      },
      {
       "context_tokens": 8192,
       "value": 53.74307625029583,
       "samples": 3
      },
      {
       "context_tokens": 16384,
       "value": 52.68395525752048,
       "samples": 3
      },
      {
       "context_tokens": 24576,
       "value": 51.561871965678094,
       "samples": 3
      },
      {
       "context_tokens": 32768,
       "value": 51.55062633347536,
       "samples": 3
      }
     ]
    },
    {
     "id": "http-mtp1-effective-prompt-proxy-vs-active-context",
     "label": "R139 MTP1 effective prompt throughput (prompt tokens divided by HTTP TTFT)",
     "metric": "prefill",
     "unit": "tok/s",
     "x_label": "Exact submitted prompt tokens",
     "scope": "Measured on the published R139 row-invariant W8A16 image (2026-09-02, clean boot): one-slot vLLM HTTP completions on two B70s at exactly 2K, 4K, 8K, 16K, 24K, and 32K active context, unrepeated technical prose, Python, and structured documents (three requests per depth, median shown), 128 output tokens, cache zero, canaries before and after. Derived as active context tokens divided by HTTP TTFT; MTP1.",
     "evidence": "experiments/qwen38-27b-b70/data/2026-09-02-qwen38-fp8-fixed-k-real-content-depth-r150-result.json",
     "points": [
      {
       "context_tokens": 2048,
       "value": 3472.782784745914,
       "samples": 3
      },
      {
       "context_tokens": 4096,
       "value": 3563.377797744992,
       "samples": 3
      },
      {
       "context_tokens": 8192,
       "value": 3488.3291722048984,
       "samples": 3
      },
      {
       "context_tokens": 16384,
       "value": 3355.129932344724,
       "samples": 3
      },
      {
       "context_tokens": 24576,
       "value": 3215.1493189010234,
       "samples": 3
      },
      {
       "context_tokens": 32768,
       "value": 3091.7452393793774,
       "samples": 3
      }
     ]
    },
    {
     "id": "http-mtp1-ttft-vs-active-context",
     "label": "R139 FP8 TP2 MTP1 HTTP TTFT over exact active context",
     "metric": "ttft",
     "unit": "ms",
     "x_label": "Exact active prompt tokens",
     "scope": "Measured on the published R139 row-invariant W8A16 image (2026-09-02, clean boot): one-slot vLLM HTTP completions on two B70s at exactly 2K, 4K, 8K, 16K, 24K, and 32K active context, unrepeated technical prose, Python, and structured documents (three requests per depth, median shown), 128 output tokens, cache zero, canaries before and after. Direct HTTP TTFT, MTP1.",
     "evidence": "experiments/qwen38-27b-b70/data/2026-09-02-qwen38-fp8-fixed-k-real-content-depth-r150-result.json",
     "points": [
      {
       "context_tokens": 2048,
       "value": 589.7287930001767,
       "samples": 3
      },
      {
       "context_tokens": 4096,
       "value": 1149.4711569994251,
       "samples": 3
      },
      {
       "context_tokens": 8192,
       "value": 2348.4022279990313,
       "samples": 3
      },
      {
       "context_tokens": 16384,
       "value": 4883.268407000287,
       "samples": 3
      },
      {
       "context_tokens": 24576,
       "value": 7643.812950000211,
       "samples": 3
      },
      {
       "context_tokens": 32768,
       "value": 10598.544661001142,
       "samples": 3
      }
     ]
    },
    {
     "id": "http-ttft-p50-vs-concurrent-users",
     "label": "Official FP8 TP2 HTTP median TTFT under concurrency",
     "metric": "ttft",
     "unit": "ms",
     "x_metric": "concurrent_sequences",
     "x_label": "Concurrent HTTP users",
     "scope": "Measured on the previous R50/R62 image (natural oneDNN W8A16 kernel), not yet re-measured on R139. Median request TTFT from the same two fresh-server, output-audited direct-P2P attempts. The service has 64 active slots, so c1-c64 are unqueued. Each point is the median of the two per-attempt p50 values; worst latency range across every reported p50/p95 metric was 4.404%. No point is interpolated or extrapolated.",
     "evidence": "experiments/qwen38-27b-b70/data/2026-08-26-qwen38-fp8-tp2-http-p64-p2p1-confirmation-r10-result.json",
     "points": [
      {
       "concurrent_sequences": 1,
       "value": 95.04806052427739,
       "samples": 2
      },
      {
       "concurrent_sequences": 2,
       "value": 122.74269599583931,
       "samples": 2
      },
      {
       "concurrent_sequences": 4,
       "value": 210.9998007363174,
       "samples": 2
      },
      {
       "concurrent_sequences": 8,
       "value": 267.1333530161064,
       "samples": 2
      },
      {
       "concurrent_sequences": 16,
       "value": 262.5558724976145,
       "samples": 2
      },
      {
       "concurrent_sequences": 32,
       "value": 426.0661029838957,
       "samples": 2
      },
      {
       "concurrent_sequences": 64,
       "value": 768.7485652277246,
       "samples": 2
      }
     ]
    },
    {
     "id": "http-ttft-p95-vs-concurrent-users",
     "label": "Official FP8 TP2 HTTP p95 TTFT under concurrency",
     "metric": "ttft",
     "unit": "ms",
     "x_metric": "concurrent_sequences",
     "x_label": "Concurrent HTTP users",
     "scope": "Measured on the previous R50/R62 image (natural oneDNN W8A16 kernel), not yet re-measured on R139. p95 request TTFT from the same two fresh-server, output-audited direct-P2P attempts. The service has 64 active slots, so c1-c64 are unqueued. Each point is the median of the two per-attempt p95 values; c64 reached 1.526 seconds. No point is interpolated or extrapolated.",
     "evidence": "experiments/qwen38-27b-b70/data/2026-08-26-qwen38-fp8-tp2-http-p64-p2p1-confirmation-r10-result.json",
     "points": [
      {
       "concurrent_sequences": 1,
       "value": 95.04806052427739,
       "samples": 2
      },
      {
       "concurrent_sequences": 2,
       "value": 170.9504044643836,
       "samples": 2
      },
      {
       "concurrent_sequences": 4,
       "value": 211.2859381682938,
       "samples": 2
      },
      {
       "concurrent_sequences": 8,
       "value": 267.68026400823146,
       "samples": 2
      },
      {
       "concurrent_sequences": 16,
       "value": 391.23190475220326,
       "samples": 2
      },
      {
       "concurrent_sequences": 32,
       "value": 728.5014198219869,
       "samples": 2
      },
      {
       "concurrent_sequences": 64,
       "value": 1525.9734081308125,
       "samples": 2
      }
     ]
    },
    {
     "id": "http-mtp2-decode-vs-active-context",
     "label": "R187 FP8 TP2 MTP depth-2 HTTP decode over exact active context",
     "metric": "decode",
     "unit": "tok/s",
     "x_label": "Exact active prompt tokens",
     "scope": "Measured on the published R187 line (R156 image, whole-graph torch.compile; 2026-09-04, clean boot): one-slot vLLM HTTP completions on two B70s at exactly 2K, 4K, 8K, 16K, 24K, and 32K active context, unrepeated technical prose, Python, and structured documents (three requests per depth, median shown), 128 output tokens, cache zero, canaries before and after. MTP1 with FP16 target verifier and draft-only INT4 head; every output matched the same-image MTP0 oracle (18/18 complete arrays). MTP depth 2 matched the same-configuration MTP0 oracle on 18/18 complete arrays.",
     "evidence": "experiments/qwen38-27b-b70/data/2026-09-04-qwen38-fp8-r187-mtp2-real-content-depth-r195-result.json",
     "points": [
      {
       "context_tokens": 2048,
       "value": 70.84942380399343,
       "samples": 3
      },
      {
       "context_tokens": 4096,
       "value": 73.3710208747659,
       "samples": 3
      },
      {
       "context_tokens": 8192,
       "value": 70.98388769165258,
       "samples": 3
      },
      {
       "context_tokens": 16384,
       "value": 68.96875043706231,
       "samples": 3
      },
      {
       "context_tokens": 24576,
       "value": 60.16140864752236,
       "samples": 3
      },
      {
       "context_tokens": 32768,
       "value": 68.52798939677955,
       "samples": 3
      }
     ]
    },
    {
     "id": "http-mtp3-decode-vs-active-context",
     "label": "R187 FP8 TP2 MTP depth-3 HTTP decode over exact active context",
     "metric": "decode",
     "unit": "tok/s",
     "x_label": "Exact active prompt tokens",
     "scope": "Measured on the published R187 line (R156 image, whole-graph torch.compile; 2026-09-04, clean boot): one-slot vLLM HTTP completions on two B70s at exactly 2K, 4K, 8K, 16K, 24K, and 32K active context, unrepeated technical prose, Python, and structured documents (three requests per depth, median shown), 128 output tokens, cache zero, canaries before and after. MTP1 with FP16 target verifier and draft-only INT4 head; every output matched the same-image MTP0 oracle (18/18 complete arrays). MTP depth 3 matched the same-configuration MTP0 oracle on 18/18 complete arrays.",
     "evidence": "experiments/qwen38-27b-b70/data/2026-09-04-qwen38-fp8-r187-mtp3-real-content-depth-r196-result.json",
     "points": [
      {
       "context_tokens": 2048,
       "value": 80.6041499563398,
       "samples": 3
      },
      {
       "context_tokens": 4096,
       "value": 83.61436918129985,
       "samples": 3
      },
      {
       "context_tokens": 8192,
       "value": 84.70874552665002,
       "samples": 3
      },
      {
       "context_tokens": 16384,
       "value": 77.85963571992264,
       "samples": 3
      },
      {
       "context_tokens": 24576,
       "value": 66.87441060480666,
       "samples": 3
      },
      {
       "context_tokens": 32768,
       "value": 83.18747425034647,
       "samples": 3
      }
     ]
    },
    {
     "id": "http-r187-mtp2-identity-qualified-aggregate-vs-concurrent-users",
     "label": "R187 FP8 TP2 MTP depth-2 identity-qualified aggregate decode vs concurrent users (c1-c4)",
     "metric": "aggregate_decode",
     "unit": "tok/s",
     "x_metric": "concurrent_sequences",
     "x_label": "Concurrent HTTP users",
     "scope": "Measured vLLM HTTP completions on two B70s with the R187 image (MTP1, FP16 verifier, draft-only INT4 head), FP16 activations/KV, 64 service slots, 256-token total request capacity, max_num_batched_tokens=512, cache disabled, 128 returned raw token IDs per response on the 64-prompt small-context suite. Every point shown is output-identity-qualified: each concurrent output equals its own sequential oracle byte for byte (harness --require-output-identity). c32 and c64 were measured but are withheld: 1/32 and 8/64 near-tie prompts took a different valid branch. One server, one pass per point; not a promoted concurrency speed record.",
     "evidence": "experiments/qwen38-27b-b70/data/2026-09-03-qwen38-fp8-r187-whole-graph-depth2-result.json",
     "points": [
      {
       "concurrent_sequences": 1,
       "value": 63.14486988197364,
       "samples": 1,
       "per_user_value": 63.14486988197364
      },
      {
       "concurrent_sequences": 2,
       "value": 68.048355221868,
       "samples": 1,
       "per_user_value": 34.024177610934
      },
      {
       "concurrent_sequences": 4,
       "value": 212.5199807591929,
       "samples": 1,
       "per_user_value": 53.12999518979822
      }
     ]
    },
    {
     "id": "http-r187-mtp3-identity-qualified-aggregate-vs-concurrent-users",
     "label": "R187 FP8 TP2 MTP depth-3 identity-qualified aggregate decode vs concurrent users (c1-c16)",
     "metric": "aggregate_decode",
     "unit": "tok/s",
     "x_metric": "concurrent_sequences",
     "x_label": "Concurrent HTTP users",
     "scope": "Measured vLLM HTTP completions on two B70s with the R187 image (MTP1, FP16 verifier, draft-only INT4 head), FP16 activations/KV, 64 service slots, 256-token total request capacity, max_num_batched_tokens=512, cache disabled, 128 returned raw token IDs per response on the 64-prompt small-context suite. Every point shown is output-identity-qualified: each concurrent output equals its own sequential oracle byte for byte (harness --require-output-identity). c32 and c64 were measured but are withheld: 1/32 and 8/64 near-tie prompts took a different valid branch. One server, one pass per point; not a promoted concurrency speed record.",
     "evidence": "experiments/qwen38-27b-b70/data/2026-09-03-qwen38-fp8-r191-whole-graph-depth3-result.json",
     "points": [
      {
       "concurrent_sequences": 1,
       "value": 70.93644548357716,
       "samples": 1,
       "per_user_value": 70.93644548357716
      },
      {
       "concurrent_sequences": 2,
       "value": 82.59040753391578,
       "samples": 1,
       "per_user_value": 41.29520376695789
      },
      {
       "concurrent_sequences": 4,
       "value": 237.64012399494973,
       "samples": 1,
       "per_user_value": 59.41003099873743
      },
      {
       "concurrent_sequences": 8,
       "value": 382.76128415386177,
       "samples": 1,
       "per_user_value": 47.84516051923272
      },
      {
       "concurrent_sequences": 16,
       "value": 557.0489881160106,
       "samples": 1,
       "per_user_value": 34.81556175725066
      }
     ]
    },
    {
     "id": "http-r187-mtp4-identity-qualified-aggregate-vs-concurrent-users",
     "label": "R187 FP8 TP2 MTP depth-4 identity-qualified aggregate decode vs concurrent users (c1-c16)",
     "metric": "aggregate_decode",
     "unit": "tok/s",
     "x_metric": "concurrent_sequences",
     "x_label": "Concurrent HTTP users",
     "scope": "Measured vLLM HTTP completions on two B70s with the R187 image (MTP1, FP16 verifier, draft-only INT4 head), FP16 activations/KV, 64 service slots, 256-token total request capacity, max_num_batched_tokens=512, cache disabled, 128 returned raw token IDs per response on the 64-prompt small-context suite. Every point shown is output-identity-qualified: each concurrent output equals its own sequential oracle byte for byte (harness --require-output-identity). c32 and c64 were measured but are withheld: 1/32 and 8/64 near-tie prompts took a different valid branch. One server, one pass per point; not a promoted concurrency speed record.",
     "evidence": "experiments/qwen38-27b-b70/data/2026-09-04-qwen38-fp8-r197-whole-graph-depth4-result.json",
     "points": [
      {
       "concurrent_sequences": 1,
       "value": 79.07277631598423,
       "per_user_value": 79.07277631598423,
       "samples": 1
      },
      {
       "concurrent_sequences": 2,
       "value": 71.14050759403455,
       "per_user_value": 35.57025379701727,
       "samples": 1
      },
      {
       "concurrent_sequences": 4,
       "value": 218.00673313409288,
       "per_user_value": 54.50168328352322,
       "samples": 1
      },
      {
       "concurrent_sequences": 8,
       "value": 384.48818883503094,
       "per_user_value": 48.06102360437887,
       "samples": 1
      },
      {
       "concurrent_sequences": 16,
       "value": 529.4198600731922,
       "per_user_value": 33.088741254574515,
       "samples": 1
      }
     ]
    },
    {
     "id": "http-r187-mtp5-identity-qualified-aggregate-vs-concurrent-users",
     "label": "R187 FP8 TP2 MTP depth-5 identity-qualified aggregate decode vs concurrent users (c1-c16)",
     "metric": "aggregate_decode",
     "unit": "tok/s",
     "x_metric": "concurrent_sequences",
     "x_label": "Concurrent HTTP users",
     "scope": "Measured vLLM HTTP completions on two B70s with the R187 image (MTP1, FP16 verifier, draft-only INT4 head), FP16 activations/KV, 64 service slots, 256-token total request capacity, max_num_batched_tokens=512, cache disabled, 128 returned raw token IDs per response on the 64-prompt small-context suite. Every point shown is output-identity-qualified: each concurrent output equals its own sequential oracle byte for byte (harness --require-output-identity). c32 and c64 were measured but are withheld: 1/32 and 8/64 near-tie prompts took a different valid branch. One server, one pass per point; not a promoted concurrency speed record.",
     "evidence": "experiments/qwen38-27b-b70/data/2026-09-04-qwen38-fp8-r200-whole-graph-depth5-strict-result.json",
     "points": [
      {
       "concurrent_sequences": 1,
       "value": 74.16068912391982,
       "per_user_value": 74.16068912391982,
       "samples": 1
      },
      {
       "concurrent_sequences": 2,
       "value": 72.18136463590626,
       "per_user_value": 36.09068231795313,
       "samples": 1
      },
      {
       "concurrent_sequences": 4,
       "value": 214.91504538380357,
       "per_user_value": 53.72876134595089,
       "samples": 1
      },
      {
       "concurrent_sequences": 8,
       "value": 375.5361131455465,
       "per_user_value": 46.94201414319331,
       "samples": 1
      },
      {
       "concurrent_sequences": 16,
       "value": 493.2246280734996,
       "per_user_value": 30.826539254593726,
       "samples": 1
      }
     ]
    }
   ],
   "missing": [
    "tested clean-host Intel driver and Docker installation",
    "independent host-driver/Docker installation and strict endpoint replay",
    "beginner recovery flow",
    "MTP1 output identity above 16 concurrent users (MTP0 is exact through 64; the MTP1 residual is not in any censused kernel)"
   ]
  },
  {
   "manifest": "packages/qwen38-27b-q4km-mtp2-tp1-b70/package.json",
   "format": "b70-model-package-v1",
   "id": "qwen38-27b-q4km-mtp2-tp1-b70",
   "name": "Qwen3.8 27B Q4_K_M + MTP2 on one Intel Arc Pro B70",
   "status": "candidate",
   "audience": "intermediate",
   "guide": "repro/qwen38-27b-q4km-mtp2-tp1-b70/README.md",
   "clean_host_tested": false,
   "library": {
    "model_family": "Qwen3.8",
    "publisher": "Alibaba / Qwen",
    "variant": "27B",
    "summary": "The one-card Qwen3.8 Q4_K_M target with its separately downloaded Q4_0 MTP draft at depth 2. Two fresh servers measured a 55.75% strict decode gain with target-exact output.",
    "quantization": "Q4_K_M target + Q4_0 MTP draft",
    "runtime_label": "llama.cpp SYCL",
    "operating_systems": [
     "Linux"
    ],
    "delivery": [
     "native"
    ],
    "modalities": [
     "text"
    ],
    "use_cases": [
     "general",
     "coding",
     "interactive"
    ],
    "tags": [
     "one card",
     "MTP2",
     "speculative decoding",
     "cache zero",
     "source build"
    ],
    "published_at": "2026-08-27",
    "featured_metric": {
     "value": 42.63698751409024,
     "unit": "tok/s",
     "label": "strict varied-prompt decode",
     "scope": "Median of two fresh-server class-balanced medians over the fixed 12-prompt/six-class, 512-cap native HTTP suite; one B70, TP1, MTP2, F16 target/draft KV, cache zero, 12/12 complete arrays exact between replicas and against same-build MTP0.",
     "evidence": "experiments/qwen38-27b-b70/data/2026-08-27-qwen38-q4km-q4mtp-tp1-mtp2-strict-result.json"
    }
   },
   "contributors": [
    {
     "id": "b70-optimization-lab",
     "name": "neural.download lab",
     "kind": "lab",
     "profile": "https://github.com/steveseguin/b70-optimization-lab",
     "contribution": "Qwen3.8 Q4_K_M kernel stack, external-MTP integration, strict MTP0/1/2/3/5 depth screen, output oracle, package, and validation.",
     "status": "integrated",
     "validated_effect": "MTP2 measured 42.636988 tok/s versus 27.375682 for the matched target-only control (+55.75%), with 12/12 complete outputs exact across both fresh MTP2 servers and the control.",
     "evidence": "experiments/qwen38-27b-b70/notes/2026-08-27-qwen38-q4km-q4mtp-tp1-mtp2-strict-result.md"
    },
    {
     "id": "mndodd",
     "name": "mndodd",
     "kind": "external",
     "profile": "https://github.com/mndodd",
     "contribution": "Optimized Intel SYCL llama.cpp fork used as the pinned runtime base beneath the lab patch stack.",
     "status": "integrated",
     "validated_effect": "Credited only for the separately matched fork contribution documented by the base Q4 package; the MTP2 screen and gain are this lab's measurements.",
     "evidence": "community/mndodd-qwen36-27b-llamacpp-sycl/STATUS.md"
    }
   ],
   "performance_profiles": [
    {
     "id": "strict-decode-vs-mtp-depth",
     "label": "Strict decode over MTP depth",
     "metric": "decode",
     "unit": "tok/s",
     "x_metric": "speculative_tokens",
     "x_label": "Maximum MTP draft depth",
     "scope": "One-B70 fixed full-suite screen under one target/runtime identity. MTP0/1/3 are one fresh-server screen values; MTP2 is the qualified median of two fresh servers. All displayed speculative arms are 12/12 target-exact. MTP5 measured 32.241 tok/s but is excluded because it matched 0/12 target arrays. No value is interpolated or extrapolated.",
     "evidence": "experiments/qwen38-27b-b70/data/2026-08-27-qwen38-q4km-q4mtp-tp1-mtp2-strict-result.json",
     "points": [
      {
       "speculative_tokens": 0,
       "value": 27.375681662420252,
       "samples": 1
      },
      {
       "speculative_tokens": 1,
       "value": 38.32021307697288,
       "samples": 1
      },
      {
       "speculative_tokens": 2,
       "value": 42.63698751409024,
       "samples": 2
      },
      {
       "speculative_tokens": 3,
       "value": 42.123431279701364,
       "samples": 1
      }
     ]
    },
    {
     "id": "http-decode-vs-active-context",
     "label": "Three-class real-content HTTP decode over active context",
     "metric": "decode",
     "unit": "tok/s",
     "x_metric": "context_tokens",
     "x_label": "Exact active prompt tokens",
     "scope": "One-slot native HTTP completions on the exact MTP2 package identity, F16 target/draft KV, cache zero, and 128 returned token IDs. Each point is the median of two fresh-server class medians over unrepeated technical prose, Python code, and structured documentation; all 36 MTP2 outputs were exact to the fresh matched MTP0 oracle. Raw document continuations are representative real-content context shapes, not a natural retrieval/task suite. No point is interpolated or extrapolated.",
     "evidence": "experiments/qwen38-27b-b70/data/2026-08-27-qwen38-q4km-q4mtp-tp1-mixed-content-depth-r1-result.json",
     "points": [
      {
       "context_tokens": 2048,
       "value": 41.424652885454535,
       "samples": 6
      },
      {
       "context_tokens": 4096,
       "value": 41.71002822382011,
       "samples": 6
      },
      {
       "context_tokens": 8192,
       "value": 41.497569394347856,
       "samples": 6
      },
      {
       "context_tokens": 16384,
       "value": 34.65465932855136,
       "samples": 6
      },
      {
       "context_tokens": 24576,
       "value": 32.25723037788288,
       "samples": 6
      },
      {
       "context_tokens": 32768,
       "value": 36.50506489790905,
       "samples": 6
      }
     ]
    },
    {
     "id": "http-ttft-vs-active-context",
     "label": "Three-class real-content HTTP TTFT over active context",
     "metric": "ttft",
     "unit": "ms",
     "x_metric": "context_tokens",
     "x_label": "Exact active prompt tokens",
     "scope": "Direct TTFT from the same two-fresh-server, three-class, target-oracle-exact receipts as the real-content decode curve. Cache was zero; all displayed prompt depths and token counts are exact; no point is interpolated or extrapolated.",
     "evidence": "experiments/qwen38-27b-b70/data/2026-08-27-qwen38-q4km-q4mtp-tp1-mixed-content-depth-r1-result.json",
     "points": [
      {
       "context_tokens": 2048,
       "value": 2083.700629475061,
       "samples": 6
      },
      {
       "context_tokens": 4096,
       "value": 4179.578055511229,
       "samples": 6
      },
      {
       "context_tokens": 8192,
       "value": 8569.58368344931,
       "samples": 6
      },
      {
       "context_tokens": 16384,
       "value": 18029.892563004978,
       "samples": 6
      },
      {
       "context_tokens": 24576,
       "value": 28347.941487503704,
       "samples": 6
      },
      {
       "context_tokens": 32768,
       "value": 39538.43021352077,
       "samples": 6
      }
     ]
    },
    {
     "id": "http-output-audited-aggregate-vs-concurrent-users",
     "label": "Output-audited MTP2 HTTP aggregate decode",
     "metric": "aggregate_decode",
     "unit": "tok/s",
     "x_metric": "concurrent_sequences",
     "x_label": "Concurrent HTTP users",
     "scope": "llama.cpp native HTTP /completion on one B70 with Q4_K_M target, Q4_0 MTP draft at depth 2, 16 slots, 8K total F16 target/draft context (512 nominal tokens per slot), prompt cache and slot similarity disabled, unique short prompts, and 128 returned raw token IDs per throughput request. Each marker is the median of two fresh-server attempts whose relative range was at most 3.04%; 256/256 separate concurrent exact-answer canaries passed. Multi-user greedy output is batch-shape-dependent. No point is interpolated or extrapolated.",
     "evidence": "experiments/qwen38-27b-b70/data/2026-08-27-qwen38-q4km-q4mtp-tp1-mtp2-http-concurrency-r2-result.json",
     "points": [
      {
       "concurrent_sequences": 1,
       "value": 34.89261578012342,
       "per_user_value": 34.89261578012342,
       "samples": 2
      },
      {
       "concurrent_sequences": 2,
       "value": 41.25496493747313,
       "per_user_value": 20.627482468736567,
       "samples": 2
      },
      {
       "concurrent_sequences": 4,
       "value": 52.3551634776993,
       "per_user_value": 13.088790869424825,
       "samples": 2
      },
      {
       "concurrent_sequences": 8,
       "value": 47.914052836702545,
       "per_user_value": 5.989256604587818,
       "samples": 2
      },
      {
       "concurrent_sequences": 16,
       "value": 68.34116883301498,
       "per_user_value": 4.271323052063436,
       "samples": 2
      }
     ]
    }
   ],
   "hardware": {
    "accelerator": "Intel Arc Pro B70 32 GiB",
    "cards": 1,
    "host_ram_min_bytes": 17179869184,
    "host_ram_plus_swap_min_bytes": 25769803776,
    "model_weight_bytes": 20343461088
   },
   "model": {
    "repository": "ggml-org/Qwen3.8-27B-GGUF",
    "revision": "0669b98607d47046c7c2b3f801011d54a08cfccf",
    "manifest": "repro/qwen38-27b-q4km-tp1-b70/model-direct.json"
   },
   "runtime": {
    "kind": "native",
    "project": "mndodd/llama.cpp",
    "revision": "4302fb59969a5d8cf9f8e5f55fdd4506d0ed2126",
    "build": "repro/qwen38-27b-q4km-mtp2-tp1-b70/restore-and-build.sh"
   },
   "known_limitations": [
    "The 42.636988 tok/s headline is the fixed short-context realistic suite. The separate Grade B context profile uses raw continuations of unrepeated technical prose, Python code, and structured documentation; it measured 36.505065 tok/s and 39.538 s TTFT at exact 32K. It is not a natural retrieval/task suite.",
    "All 36 outputs in the new real-content MTP2 profile matched the fresh MTP0 oracle, but the older repeated-token diagnostic still reproducibly diverged at 2K/generated token 23. Target parity remains workload-scoped rather than universal.",
    "Output-qualified concurrency reaches 68.341 aggregate tok/s at 16 users only in the measured 16-slot/8K-total service profile. Larger 32- and 64-slot MTP2 profiles failed startup with device OOM; this is not a 32K-per-user or 64-user claim.",
    "MTP5 is explicitly unsafe for this identity: it changed all 12 complete target outputs and is not a supported speed mode.",
    "The tested 16 GiB host used swap and a 13 GiB process memory cap. A clean-host Intel/oneAPI install and source-build replay remain pending.",
    "The target and MTP draft come from two pinned repositories/revisions and both downloads are required."
   ],
   "project_patches": {
    "required": true,
    "items": [
     "patches/qwen36-27b-q8-tp2-asrock-b70/llama-cpp-mndodd-4302fb599-lab-tp2-dp4a2-20260815.diff.gz.b64",
     "patches/qwen38-27b-q4km-tp2-asrock-b70/llama-cpp-q4k-mmvq-swiglu-tp2-20260815.diff.gz.b64",
     "patches/qwen38-27b-q4km-tp1-b70s/llama-cpp-tp1-gdn-state-io-widen-20260821.diff.gz.b64",
     "patches/qwen38-27b-q4km-tp1-b70s/llama-cpp-tp1-conv-qk-widen-20260821.diff.gz.b64",
     "patches/qwen38-27b-q4km-tp1-b70s/llama-cpp-tp1-qk-norm-rope-src-widen-20260821.diff.gz.b64",
     "patches/qwen38-27b-q4km-tp1-b70s/llama-cpp-tp1-q8out-rejected-memo320-20260821.diff.gz.b64"
    ],
    "reason": "The measured target and MTP verifier path uses the complete lab TP1 source stack; the guide links and verifies every patch."
   },
   "commands": {
    "preflight": "TARGET_DIR=/path/to/target DRAFT_DIR=/path/to/draft BUILD_DIR=/path/to/build repro/qwen38-27b-q4km-mtp2-tp1-b70/preflight.sh",
    "launch": "TARGET_DIR=/path/to/target DRAFT_DIR=/path/to/draft BUILD_DIR=/path/to/build repro/qwen38-27b-q4km-mtp2-tp1-b70/run-server.sh",
    "health": "curl -fsS http://127.0.0.1:18139/health",
    "benchmark": "OUT_DIR=/path/to/new-result repro/qwen38-27b-q4km-mtp2-tp1-b70/bench.sh",
    "stop": "Ctrl-C in the foreground server terminal; then pgrep -x llama-server must return no process"
   },
   "dependencies": [
    "repro/qwen38-27b-q4km-mtp2-tp1-b70/README.md",
    "experiments/qwen38-27b-b70/data/qwen38-q4km-targetonly-tp1-mtp0-20260827-r1/performance.json",
    "repro/gemma4-26b-a4b-q8-b70/realistic-suite-v1.json",
    "repro/qwen36-27b-autoround-int4-b70/realistic-suite-v1.json",
    "repro/qwen38-27b-autoround-int4-b70/scripts/verify-model-direct.py",
    "repro/qwen38-27b-q4km-mtp2-tp1-b70/draft-model-direct.json",
    "repro/qwen38-27b-q4km-mtp2-tp1-b70/verify-models.sh",
    "repro/qwen38-27b-q4km-mtp2-tp1-b70/preflight.sh",
    "repro/qwen38-27b-q4km-mtp2-tp1-b70/restore-and-build.sh",
    "repro/qwen38-27b-q4km-mtp2-tp1-b70/run-server.sh",
    "repro/qwen38-27b-q4km-mtp2-tp1-b70/bench.sh",
    "experiments/qwen38-27b-b70/data/2026-08-27-qwen38-q4km-q4mtp-tp1-depth-screen-r1-prereg.json",
    "experiments/qwen38-27b-b70/data/2026-08-27-qwen38-q4km-q4mtp-tp1-mtp2-strict-result.json",
    "experiments/qwen38-27b-b70/data/2026-08-27-qwen38-q4km-q4mtp-tp1-mtp2-r2-r3-comparison.json",
    "experiments/qwen38-27b-b70/data/2026-08-27-qwen38-q4km-mtp0-vs-q4mtp-mtp2-r3-comparison.json",
    "experiments/qwen38-27b-b70/data/2026-08-27-qwen38-q4km-q4mtp-tp1-mtp2-exact-depth-r2-result.json",
    "experiments/qwen38-27b-b70/data/2026-08-27-qwen38-q4km-q4mtp-tp1-mtp2-exact-depth-r2-report-recovery-prereg.json",
    "data/qwen27-exact-depth/qwen38-bce40ca-mixed-content-depth-v1.json",
    "scripts/build-qwen27-mixed-content-depth-fixture.py",
    "experiments/qwen38-27b-b70/data/2026-08-27-qwen38-q4km-q4mtp-tp1-mixed-content-depth-r1-prereg.json",
    "experiments/qwen38-27b-b70/data/2026-08-27-qwen38-q4km-q4mtp-tp1-mixed-content-depth-r1-mtp2-amendment.json",
    "experiments/qwen38-27b-b70/data/2026-08-27-qwen38-q4km-q4mtp-tp1-mixed-content-depth-r1-result.json",
    "experiments/qwen38-27b-b70/notes/2026-08-27-qwen38-q4km-q4mtp-tp1-mixed-content-depth-r1-result.md",
    "experiments/qwen38-27b-b70/scripts/run-20260827-qwen38-q4km-q4mtp-tp1-mixed-content-depth-arm.sh",
    "experiments/qwen38-27b-b70/scripts/validate-20260827-qwen38-q4km-q4mtp-tp1-mixed-content-depth-r1.py",
    "experiments/qwen38-27b-b70/data/2026-08-27-qwen38-q4km-q4mtp-tp1-mtp2-http-concurrency-r1-prereg.json",
    "experiments/qwen38-27b-b70/data/2026-08-27-qwen38-q4km-q4mtp-tp1-mtp2-http-concurrency-r2-capacity-amendment.json",
    "experiments/qwen38-27b-b70/data/2026-08-27-qwen38-q4km-q4mtp-tp1-mtp2-http-concurrency-r2-result.json",
    "experiments/qwen38-27b-b70/notes/2026-08-27-qwen38-q4km-q4mtp-tp1-mtp2-http-concurrency-r2-result.md",
    "experiments/qwen38-27b-b70/scripts/run-20260827-qwen38-q4km-q4mtp-tp1-mtp2-http-concurrency-r1.sh",
    "experiments/qwen38-27b-b70/scripts/qwen38-concurrent-quality-canary.py",
    "experiments/qwen38-27b-b70/scripts/recover-20260827-qwen38-q4km-q4mtp-tp1-mtp2-exact-depth-r2.py",
    "experiments/qwen38-27b-b70/scripts/run-20260827-qwen38-q4km-q4mtp-tp1-screen-attempt.sh",
    "scripts/bench-openai-realistic-suite.py",
    "scripts/neural-download-canaries.py",
    "scripts/compare-strict-attempt-outputs.py",
    "repro/qwen38-27b-q4km-tp1-b70/model-direct.json",
    "experiments/qwen38-27b-b70/notes/2026-08-27-qwen38-q4km-q4mtp-tp1-mtp2-strict-result.md",
    "community/mndodd-qwen36-27b-llamacpp-sycl/STATUS.md",
    "patches/qwen36-27b-q8-tp2-asrock-b70/llama-cpp-mndodd-4302fb599-lab-tp2-dp4a2-20260815.diff.gz.b64",
    "patches/qwen38-27b-q4km-tp2-asrock-b70/llama-cpp-q4k-mmvq-swiglu-tp2-20260815.diff.gz.b64",
    "patches/qwen38-27b-q4km-tp1-b70s/llama-cpp-tp1-gdn-state-io-widen-20260821.diff.gz.b64",
    "patches/qwen38-27b-q4km-tp1-b70s/llama-cpp-tp1-conv-qk-widen-20260821.diff.gz.b64",
    "patches/qwen38-27b-q4km-tp1-b70s/llama-cpp-tp1-qk-norm-rope-src-widen-20260821.diff.gz.b64",
    "patches/qwen38-27b-q4km-tp1-b70s/llama-cpp-tp1-q8out-rejected-memo320-20260821.diff.gz.b64"
   ],
   "missing": [
    "natural retrieval/task long-context HTTP suite beyond the measured raw-document continuation profile",
    "tested clean-host Intel driver and oneAPI installation",
    "clean-host source build and endpoint replay",
    "beginner recovery flow"
   ]
  },
  {
   "manifest": "packages/qwen38-27b-q4km-q4mtp-mtp2-tp2-b70/package.json",
   "format": "b70-model-package-v1",
   "id": "qwen38-27b-q4km-q4mtp-mtp2-tp2-b70",
   "name": "Qwen3.8 27B Q4_K_M + MTP2 on two Intel Arc Pro B70 cards",
   "status": "candidate",
   "audience": "expert",
   "guide": "repro/qwen38-27b-q4km-q4mtp-mtp2-tp2-b70/README.md",
   "clean_host_tested": false,
   "library": {
    "model_family": "Qwen3.8",
    "publisher": "Alibaba / Qwen",
    "variant": "27B Q4_K_M target + Q4_0 MTP2 TP2",
    "summary": "Two-B70 Qwen3.8 27B Q4_K_M with target-exact Q4_0 MTP2.",
    "quantization": "Q4_K_M target + Q4_0 MTP draft / F16 KV",
    "runtime_label": "llama.cpp SYCL",
    "operating_systems": [
     "Linux"
    ],
    "delivery": [
     "native"
    ],
    "modalities": [
     "text"
    ],
    "use_cases": [
     "general",
     "coding",
     "interactive",
     "two-card throughput"
    ],
    "tags": [
     "two cards",
     "TP2",
     "MTP2",
     "speculative decoding",
     "cache zero",
     "source build"
    ],
    "published_at": "2026-08-30",
    "featured_metric": {
     "value": 64.23730130993152,
     "unit": "tok/s",
     "label": "strict varied-prompt decode",
     "scope": "Median of two fresh-server class-balanced medians over the fixed 12-prompt/six-class, 512-cap native HTTP suite; two B70s, equal target split, draft on SYCL0, MTP2, F16 target/draft KV, cache zero, and 24/24 complete candidate arrays exact to the fresh target-only oracle.",
     "evidence": "experiments/qwen38-27b-b70/data/2026-08-30-qwen38-q4km-q4mtp-tp2-mtp2-promotion-attestation.json"
    }
   },
   "contributors": [
    {
     "id": "b70-optimization-lab",
     "name": "neural.download lab",
     "kind": "lab",
     "profile": "https://github.com/steveseguin/b70-optimization-lab",
     "contribution": "Qwen3.8 Q4_K_M two-card kernel stack, external-MTP TP2 composition, preregistered strict oracle/replication campaign, exact-output validation, and package.",
     "status": "integrated",
     "validated_effect": "MTP2 measured a two-server median of 64.237301 tok/s versus 49.787366 tok/s for the matched fresh target-only oracle (+29.02%), with 24/24 complete candidate outputs target-exact.",
     "evidence": "experiments/qwen38-27b-b70/notes/2026-08-30-qwen38-q4km-q4mtp-tp2-mtp2-promoted-result.md"
    },
    {
     "id": "mndodd",
     "name": "mndodd",
     "kind": "external",
     "profile": "https://github.com/mndodd",
     "contribution": "Optimized Intel SYCL llama.cpp fork used as the pinned runtime base beneath the lab patch stack.",
     "status": "integrated",
     "validated_effect": "Credited only for the separately matched fork contribution documented by the base Q4 package; the TP2/MTP2 composition and gain are this lab's measurements.",
     "evidence": "community/mndodd-qwen36-27b-llamacpp-sycl/STATUS.md"
    }
   ],
   "performance_profiles": [
    {
     "id": "strict-decode-vs-mtp-depth",
     "label": "Strict TP2 decode with and without MTP2",
     "metric": "decode",
     "unit": "tok/s",
     "x_metric": "speculative_tokens",
     "x_label": "Maximum MTP draft depth",
     "scope": "Same target, runtime, two-card equal split, F16 KV, cache-zero suite, and short-context contract. MTP0 is one fresh oracle; MTP2 is the median of two fresh servers and all 24 candidate arrays were target-exact. Only depths 0 and 2 were measured in this TP2 campaign; no point is interpolated or extrapolated.",
     "evidence": "experiments/qwen38-27b-b70/data/2026-08-30-qwen38-q4km-q4mtp-tp2-mtp2-promotion-attestation.json",
     "points": [
      {
       "speculative_tokens": 0,
       "value": 49.78736600126793,
       "samples": 1
      },
      {
       "speculative_tokens": 2,
       "value": 64.23730130993152,
       "samples": 2
      }
     ]
    }
   ],
   "hardware": {
    "accelerator": "Intel Arc Pro B70 32 GiB",
    "cards": 2,
    "host_ram_min_bytes": 16106127360,
    "host_ram_plus_swap_min_bytes": 25769803776,
    "model_weight_bytes": 20343461088
   },
   "model": {
    "repository": "ggml-org/Qwen3.8-27B-GGUF",
    "revision": "0669b98607d47046c7c2b3f801011d54a08cfccf",
    "manifest": "repro/qwen38-27b-q4km-q4mtp-mtp2-tp2-b70/manifest.sha256"
   },
   "runtime": {
    "kind": "native",
    "project": "mndodd/llama.cpp",
    "revision": "4302fb59969a5d8cf9f8e5f55fdd4506d0ed2126",
    "build": "repro/qwen38-27b-q4km-q4mtp-mtp2-tp2-b70/restore-and-build.sh"
   },
   "project_patches": {
    "required": true,
    "items": [
     "patches/qwen36-27b-q8-tp2-asrock-b70/llama-cpp-mndodd-4302fb599-lab-tp2-dp4a2-20260815.diff.gz.b64",
     "patches/qwen38-27b-q4km-tp2-asrock-b70/llama-cpp-q4k-mmvq-swiglu-tp2-20260815.diff.gz.b64",
     "patches/qwen38-27b-q4km-tp1-b70s/llama-cpp-tp1-gdn-state-io-widen-20260821.diff.gz.b64",
     "patches/qwen38-27b-q4km-tp1-b70s/llama-cpp-tp1-conv-qk-widen-20260821.diff.gz.b64",
     "patches/qwen38-27b-q4km-tp1-b70s/llama-cpp-tp1-qk-norm-rope-src-widen-20260821.diff.gz.b64",
     "patches/qwen38-27b-q4km-tp1-b70s/llama-cpp-tp1-q8out-rejected-memo320-20260821.diff.gz.b64"
    ],
    "reason": "The exact measured runtime uses the complete lab source stack. The guide links every patch directly and the preflight binds the measured runtime hashes."
   },
   "commands": {
    "preflight": "TARGET_DIR=/path/to/target DRAFT_DIR=/path/to/draft/MTP BUILD_DIR=/path/to/build repro/qwen38-27b-q4km-q4mtp-mtp2-tp2-b70/preflight.sh",
    "launch": "TARGET_DIR=/path/to/target DRAFT_DIR=/path/to/draft/MTP BUILD_DIR=/path/to/build MTP_DEPTH=2 repro/qwen38-27b-q4km-q4mtp-mtp2-tp2-b70/run-server.sh",
    "health": "curl -fsS http://127.0.0.1:18142/health",
    "benchmark": "MTP_DEPTH=2 ORACLE_JSON=/path/to/fresh-mtp0/performance.json OUT_DIR=/path/to/new-result repro/qwen38-27b-q4km-q4mtp-mtp2-tp2-b70/bench.sh",
    "stop": "Ctrl-C in the foreground server terminal; then pgrep -x llama-server must return no process"
   },
   "dependencies": [
    "repro/qwen38-27b-q4km-q4mtp-mtp2-tp2-b70/README.md",
    "repro/gemma4-26b-a4b-q8-b70/realistic-suite-v1.json",
    "repro/qwen36-27b-autoround-int4-b70/realistic-suite-v1.json",
    "repro/qwen38-27b-autoround-int4-b70/scripts/verify-model-direct.py",
    "repro/qwen38-27b-q4km-q4mtp-mtp2-tp2-b70/manifest.sha256",
    "repro/qwen38-27b-q4km-q4mtp-mtp2-tp2-b70/preflight.sh",
    "repro/qwen38-27b-q4km-q4mtp-mtp2-tp2-b70/restore-and-build.sh",
    "repro/qwen38-27b-q4km-q4mtp-mtp2-tp2-b70/run-server.sh",
    "repro/qwen38-27b-q4km-q4mtp-mtp2-tp2-b70/bench.sh",
    "repro/qwen38-27b-q4km-q4mtp-mtp2-tp2-b70/verify-evidence.sh",
    "repro/qwen38-27b-q4km-mtp2-tp1-b70/draft-model-direct.json",
    "repro/qwen38-27b-q4km-mtp2-tp1-b70/verify-models.sh",
    "repro/qwen38-27b-q4km-tp1-b70/model-direct.json",
    "experiments/qwen38-27b-b70/data/2026-08-30-qwen38-q4km-q4mtp-tp2-mtp2-screen-r1-prereg.json",
    "experiments/qwen38-27b-b70/data/2026-08-30-qwen38-q4km-q4mtp-tp2-mtp2-screen-r1-result.json",
    "experiments/qwen38-27b-b70/data/2026-08-30-qwen38-q4km-q4mtp-tp2-mtp2-replication-r2-prereg.json",
    "experiments/qwen38-27b-b70/data/2026-08-30-qwen38-q4km-q4mtp-tp2-mtp2-replication-r2-result.json",
    "experiments/qwen38-27b-b70/data/2026-08-30-qwen38-q4km-q4mtp-tp2-mtp2-promotion-attestation.json",
    "experiments/qwen38-27b-b70/data/qwen38-q4km-q4mtp-tp2-mtp2-20260830/manifest.json",
    "experiments/qwen38-27b-b70/notes/2026-08-30-qwen38-q4km-q4mtp-tp2-mtp2-promoted-result.md",
    "experiments/qwen38-27b-b70/scripts/validate-20260830-qwen38-q4km-q4mtp-tp2-mtp2-screen-r1.py",
    "experiments/qwen38-27b-b70/scripts/validate-20260830-qwen38-q4km-q4mtp-tp2-mtp2-replication-r2.py",
    "experiments/qwen38-27b-b70/scripts/verify-20260830-qwen38-q4km-q4mtp-tp2-mtp2-archive.py",
    "scripts/bench-openai-realistic-suite.py",
    "scripts/neural-download-canaries.py",
    "scripts/promotion_evidence.py",
    "community/mndodd-qwen36-27b-llamacpp-sycl/STATUS.md",
    "patches/qwen36-27b-q8-tp2-asrock-b70/llama-cpp-mndodd-4302fb599-lab-tp2-dp4a2-20260815.diff.gz.b64",
    "patches/qwen38-27b-q4km-tp2-asrock-b70/llama-cpp-q4k-mmvq-swiglu-tp2-20260815.diff.gz.b64",
    "patches/qwen38-27b-q4km-tp1-b70s/llama-cpp-tp1-gdn-state-io-widen-20260821.diff.gz.b64",
    "patches/qwen38-27b-q4km-tp1-b70s/llama-cpp-tp1-conv-qk-widen-20260821.diff.gz.b64",
    "patches/qwen38-27b-q4km-tp1-b70s/llama-cpp-tp1-qk-norm-rope-src-widen-20260821.diff.gz.b64",
    "patches/qwen38-27b-q4km-tp1-b70s/llama-cpp-tp1-q8out-rejected-memo320-20260821.diff.gz.b64"
   ],
   "known_limitations": [
    "The 64.237301 tok/s headline is short-context, one-user evidence with 8K configured context. Target-only TP2 and one-card MTP2 context or concurrency values do not transfer.",
    "Only MTP0 and MTP2 were measured under this exact TP2 campaign. No deeper MTP mode is supported by this package.",
    "The tested 16 GiB host used swap and a 13 GiB process memory cap. A clean-host Intel/oneAPI install and source-build replay remain pending.",
    "The target and MTP draft come from two pinned repositories/revisions and both downloads are required.",
    "A locally rebuilt binary is a distinct identity until the paired full-suite oracle and MTP2 validation pass."
   ],
   "missing": [
    "measured 32K active-context profile for this exact MTP2 deployment",
    "output-qualified HTTP concurrency profile for this exact MTP2 deployment",
    "standard 512-token prompt TTFT and prefill measurement",
    "tested clean-host Intel driver and oneAPI installation",
    "clean-host source build and endpoint replay",
    "beginner recovery replay"
   ]
  },
  {
   "manifest": "packages/qwen38-27b-q4km-tp1-b70/package.json",
   "format": "b70-model-package-v1",
   "id": "qwen38-27b-q4km-tp1-b70",
   "name": "Qwen3.8 27B Q4_K_M on one Intel Arc Pro B70",
   "status": "candidate",
   "audience": "intermediate",
   "guide": "repro/qwen38-27b-q4km-tp1-b70/README.md",
   "clean_host_tested": false,
   "library": {
    "model_family": "Qwen3.8",
    "publisher": "Alibaba / Qwen",
    "variant": "27B",
    "summary": "Alibaba's Qwen3.8 27B on one Arc Pro B70 in 4-bit form with llama.cpp, no draft model. The lab's promoted one-card package, rebuilt from pinned source.",
    "quantization": "Q4_K_M",
    "runtime_label": "llama.cpp SYCL",
    "operating_systems": [
     "Linux"
    ],
    "delivery": [
     "native"
    ],
    "modalities": [
     "text"
    ],
    "use_cases": [
     "general",
     "coding",
     "long context"
    ],
    "tags": [
     "one card",
     "target only",
     "cache zero",
     "source build"
    ],
    "published_at": "2026-08-22",
    "featured_metric": {
     "value": 27.825725650072858,
     "unit": "tok/s",
     "label": "decode",
     "scope": "Class-balanced median of per-input-class medians using conventional 99-interval rates on the fixed cold 12-prompt suite; target-only and cache-zero. The all-prompt median is 27.824790 tok/s.",
     "evidence": "experiments/qwen38-27b-b70/data/2026-08-21-q4km-tp1-gpu0-final-j.json"
    }
   },
   "contributors": [
    {
     "id": "b70-optimization-lab",
     "name": "neural.download lab",
     "kind": "lab",
     "profile": "https://github.com/steveseguin/b70-optimization-lab",
     "contribution": "Qwen3.8 Q4_K_M bring-up, the complete lab kernel stack, TP1 fusion increments, package, and validation.",
     "status": "integrated",
     "validated_effect": "The TP1 matcher and fusion ladder moved the registered 26.047863/26.068073 tok/s baseline to 27.813629/27.824790 tok/s (+6.8% to +7.0%) with 24/24 oracle-exact outputs and the full quality battery passing.",
     "evidence": "patches/qwen38-27b-q4km-tp1-b70s/README.md"
    },
    {
     "id": "mndodd",
     "name": "mndodd",
     "kind": "external",
     "profile": "https://github.com/mndodd",
     "contribution": "Optimized Intel SYCL llama.cpp fork used as the pinned runtime base beneath the lab's Qwen3.8 patch stack.",
     "status": "integrated",
     "validated_effect": "On the separate matched Qwen3.6 Q8 TP2 control, the fork measured 31.338765 vs 29.610651 tok/s (+5.836%); that A/B does not assign the later Qwen3.8 package result to the fork.",
     "evidence": "community/mndodd-qwen36-27b-llamacpp-sycl/STATUS.md"
    }
   ],
   "performance_profiles": [
    {
     "id": "decode-f16-kv-vs-context-depth",
     "label": "Q4_K_M raw decode with F16 KV",
     "metric": "decode",
     "unit": "tok/s",
     "x_label": "Existing context depth before tg128",
     "scope": "Direct llama-bench raw-engine tg128 using the exact Q4_K_M TP1 lane, one B70, flash attention on, F16 K/V, and 5 repetitions at every displayed depth. This is a context-shape profile, not the 27.82 tok/s realistic-suite headline. Every marker is measured; no point is interpolated or extrapolated.",
     "evidence": "experiments/qwen38-27b-b70/data/2026-08-22-q4km-tp1-context-kv-sweep.json",
     "points": [
      {
       "context_tokens": 0,
       "value": 24.81,
       "samples": 5
      },
      {
       "context_tokens": 2048,
       "value": 24.46,
       "samples": 5
      },
      {
       "context_tokens": 4096,
       "value": 24.25,
       "samples": 5
      },
      {
       "context_tokens": 8192,
       "value": 23.83,
       "samples": 5
      },
      {
       "context_tokens": 16384,
       "value": 23.1,
       "samples": 5
      },
      {
       "context_tokens": 24576,
       "value": 22.42,
       "samples": 5
      },
      {
       "context_tokens": 32768,
       "value": 21.77,
       "samples": 5
      }
     ]
    },
    {
     "id": "prefill-f16-kv-vs-context-depth",
     "label": "Q4_K_M raw pp2048 with F16 KV",
     "metric": "prefill",
     "unit": "tok/s",
     "x_label": "Existing context depth before pp2048",
     "scope": "Direct llama-bench raw-engine pp2048 using the exact Q4_K_M TP1 lane, one B70, flash attention on, F16 K/V, and 5 repetitions at every displayed depth. Every marker is measured; no point is interpolated or extrapolated.",
     "evidence": "experiments/qwen38-27b-b70/data/2026-08-22-q4km-tp1-context-kv-sweep.json",
     "points": [
      {
       "context_tokens": 0,
       "value": 825.24,
       "samples": 5
      },
      {
       "context_tokens": 2048,
       "value": 919.67,
       "samples": 5
      },
      {
       "context_tokens": 4096,
       "value": 892.64,
       "samples": 5
      },
      {
       "context_tokens": 8192,
       "value": 850.97,
       "samples": 5
      },
      {
       "context_tokens": 16384,
       "value": 779.5,
       "samples": 5
      },
      {
       "context_tokens": 24576,
       "value": 719.39,
       "samples": 5
      },
      {
       "context_tokens": 32768,
       "value": 667.83,
       "samples": 5
      }
     ]
    },
    {
     "id": "decode-q8-kv-vs-context-depth",
     "label": "Q4_K_M raw decode with Q8_0 KV",
     "metric": "decode",
     "unit": "tok/s",
     "x_label": "Existing context depth before tg128",
     "scope": "Alternative Q8_0 K/V operating profile on the same exact Q4_K_M TP1 lane, one B70, flash attention on, raw-engine tg128, and 5 repetitions per depth. It is shown separately because KV precision materially changes decode. Every marker is measured; no point is interpolated or extrapolated.",
     "evidence": "experiments/qwen38-27b-b70/data/2026-08-22-q4km-tp1-context-kv-sweep.json",
     "points": [
      {
       "context_tokens": 0,
       "value": 24.27,
       "samples": 5
      },
      {
       "context_tokens": 2048,
       "value": 22.45,
       "samples": 5
      },
      {
       "context_tokens": 4096,
       "value": 21.05,
       "samples": 5
      },
      {
       "context_tokens": 8192,
       "value": 18.68,
       "samples": 5
      },
      {
       "context_tokens": 16384,
       "value": 14.86,
       "samples": 5
      },
      {
       "context_tokens": 24576,
       "value": 12.4,
       "samples": 5
      },
      {
       "context_tokens": 32768,
       "value": 10.66,
       "samples": 5
      }
     ]
    },
    {
     "id": "prefill-q8-kv-vs-context-depth",
     "label": "Q4_K_M raw pp2048 with Q8_0 KV",
     "metric": "prefill",
     "unit": "tok/s",
     "x_label": "Existing context depth before pp2048",
     "scope": "Alternative Q8_0 K/V operating profile on the same exact Q4_K_M TP1 lane, one B70, flash attention on, raw-engine pp2048, and 5 repetitions per depth. Every marker is measured; no point is interpolated or extrapolated.",
     "evidence": "experiments/qwen38-27b-b70/data/2026-08-22-q4km-tp1-context-kv-sweep.json",
     "points": [
      {
       "context_tokens": 0,
       "value": 817.78,
       "samples": 5
      },
      {
       "context_tokens": 2048,
       "value": 912.17,
       "samples": 5
      },
      {
       "context_tokens": 4096,
       "value": 887.17,
       "samples": 5
      },
      {
       "context_tokens": 8192,
       "value": 843.21,
       "samples": 5
      },
      {
       "context_tokens": 16384,
       "value": 772.01,
       "samples": 5
      },
      {
       "context_tokens": 24576,
       "value": 711.48,
       "samples": 5
      },
      {
       "context_tokens": 32768,
       "value": 662.56,
       "samples": 5
      }
     ]
    },
    {
     "id": "aggregate-decode-vs-concurrent-sequences",
     "label": "Raw aggregate decode over parallel sequences",
     "metric": "aggregate_decode",
     "unit": "tok/s",
     "x_metric": "concurrent_sequences",
     "x_label": "Concurrent engine sequences",
     "scope": "Direct llama-batched-bench raw-engine continuous batching with independent pp128 prompts, interleaved tg256 decode, c32768, flash attention on, F16 KV, one B70, and the exact accepted package stack. This is a mechanism ceiling: it does not emit auditable completions and excludes HTTP, JSON, queueing, and server-scheduler overhead. No point is scaled, interpolated, or extrapolated.",
     "evidence": "experiments/qwen38-27b-b70/data/qwen38-q4km-tp1-batched-ladder-20260825-r1-attempt2/summary.json",
     "points": [
      {
       "concurrent_sequences": 1,
       "value": 24.363621,
       "per_user_value": 24.363621,
       "samples": 1
      },
      {
       "concurrent_sequences": 2,
       "value": 39.62175,
       "per_user_value": 19.810875,
       "samples": 1
      },
      {
       "concurrent_sequences": 4,
       "value": 54.081142,
       "per_user_value": 13.5202855,
       "samples": 1
      },
      {
       "concurrent_sequences": 8,
       "value": 59.088936,
       "per_user_value": 7.386117,
       "samples": 1
      },
      {
       "concurrent_sequences": 16,
       "value": 58.696918,
       "per_user_value": 3.668557375,
       "samples": 1
      },
      {
       "concurrent_sequences": 32,
       "value": 70.999931,
       "per_user_value": 2.21874784375,
       "samples": 1
      },
      {
       "concurrent_sequences": 64,
       "value": 95.411842,
       "per_user_value": 1.49081003125,
       "samples": 1
      }
     ]
    },
    {
     "id": "http-decode-vs-active-context",
     "label": "Qualified HTTP decode over exact active context",
     "metric": "decode",
     "unit": "tok/s",
     "x_label": "Exact active prompt tokens",
     "scope": "One-slot llama-server HTTP completions on the exact accepted Q4_K_M TP1 stack, one B70, F16 KV, 33,024 configured context, cache zero, no truncation or context shift, and 128 returned token IDs. The registered repeated-token fixture is evidence grade C: it fixes context shape but is not natural prose. Every marker is measured; no point is interpolated or extrapolated.",
     "evidence": "experiments/qwen38-27b-b70/data/2026-08-25-qwen38-q4km-tp1-http-depth-r1-result.json",
     "points": [
      {
       "context_tokens": 2048,
       "value": 27.261640383930207,
       "samples": 1
      },
      {
       "context_tokens": 4096,
       "value": 27.19498446044623,
       "samples": 1
      },
      {
       "context_tokens": 8192,
       "value": 26.865065081907545,
       "samples": 1
      },
      {
       "context_tokens": 16384,
       "value": 25.997575234489133,
       "samples": 1
      },
      {
       "context_tokens": 24576,
       "value": 25.235258591473393,
       "samples": 1
      },
      {
       "context_tokens": 32768,
       "value": 24.488129029771436,
       "samples": 1
      }
     ]
    },
    {
     "id": "http-output-audited-aggregate-vs-concurrent-users",
     "label": "Output-audited HTTP aggregate decode",
     "metric": "aggregate_decode",
     "unit": "tok/s",
     "x_metric": "concurrent_sequences",
     "x_label": "Concurrent HTTP users",
     "scope": "llama.cpp native HTTP /completion on one B70 with 64 slots, 32K total F16 KV context, prompt cache and slot similarity disabled, unique short prompts, and 128 returned raw token IDs per request. Each marker is the median of two preregistered fresh-server attempts whose relative range was at most 2.03%. Output isolation passed with no cross-base oracle collision, but multi-user greedy token identity is batch-shape-dependent. No point is interpolated or extrapolated.",
     "evidence": "experiments/qwen38-27b-b70/data/2026-08-25-qwen38-q4km-tp1-http-concurrency-r3-result.json",
     "points": [
      {
       "concurrent_sequences": 1,
       "value": 24.640824267538715,
       "per_user_value": 24.640824267538715,
       "samples": 2
      },
      {
       "concurrent_sequences": 2,
       "value": 36.55301939162611,
       "per_user_value": 18.276509695813054,
       "samples": 2
      },
      {
       "concurrent_sequences": 4,
       "value": 49.31613227816557,
       "per_user_value": 12.329033069541392,
       "samples": 2
      },
      {
       "concurrent_sequences": 8,
       "value": 56.11968943309129,
       "per_user_value": 7.014961179136411,
       "samples": 2
      },
      {
       "concurrent_sequences": 16,
       "value": 54.96560656389367,
       "per_user_value": 3.4353504102433545,
       "samples": 2
      },
      {
       "concurrent_sequences": 32,
       "value": 65.80288089832067,
       "per_user_value": 2.056340028072521,
       "samples": 2
      },
      {
       "concurrent_sequences": 64,
       "value": 83.79674254084222,
       "per_user_value": 1.3093241022006596,
       "samples": 2
      }
     ]
    },
    {
     "id": "http-ttft-vs-active-context",
     "label": "Qualified HTTP TTFT over exact active context",
     "metric": "ttft",
     "unit": "ms",
     "x_label": "Exact active prompt tokens",
     "scope": "TTFT from the same one-slot exact-token HTTP receipts as the qualified decode curve. Cache was zero and prompt counts were exact at every marker. The repeated-token fixture is evidence grade C and is not a natural-prose latency claim. Every marker is measured; no point is interpolated or extrapolated.",
     "evidence": "experiments/qwen38-27b-b70/data/2026-08-25-qwen38-q4km-tp1-http-depth-r1-result.json",
     "points": [
      {
       "context_tokens": 2048,
       "value": 2775.1139199826866,
       "samples": 1
      },
      {
       "context_tokens": 4096,
       "value": 5577.305173967034,
       "samples": 1
      },
      {
       "context_tokens": 8192,
       "value": 11343.381671002135,
       "samples": 1
      },
      {
       "context_tokens": 16384,
       "value": 23472.727512940764,
       "samples": 1
      },
      {
       "context_tokens": 24576,
       "value": 36439.75531402975,
       "samples": 1
      },
      {
       "context_tokens": 32768,
       "value": 50266.55042101629,
       "samples": 1
      }
     ]
    }
   ],
   "hardware": {
    "accelerator": "Intel Arc Pro B70 32 GiB",
    "cards": 1,
    "host_ram_min_bytes": 68719476736,
    "host_ram_plus_swap_min_bytes": 68719476736,
    "model_weight_bytes": 18973870432
   },
   "model": {
    "repository": "ggml-org/Qwen3.8-27B-GGUF",
    "revision": "0669b98607d47046c7c2b3f801011d54a08cfccf",
    "manifest": "repro/qwen38-27b-q4km-tp1-b70/model-direct.json"
   },
   "runtime": {
    "kind": "native",
    "project": "mndodd/llama.cpp",
    "revision": "4302fb59969a5d8cf9f8e5f55fdd4506d0ed2126",
    "build": "repro/qwen38-27b-q4km-tp1-b70/restore-and-build.sh"
   },
   "known_limitations": [
    "The beginner launch remains a conservative one-slot 8K profile. Separate audits measured exact one-slot HTTP active context through 32K and a stable output-audited 1\u219264 HTTP aggregate curve. The 64-slot profile nearly fills the card and multi-user greedy token identity is batch-shape-dependent, so it is a capacity result rather than a deterministic serving recommendation.",
    "Qwen3.8 27B is dense; do not compare this one-card aggregate rate as if it were a sparse Qwen-derived MoE or an NVIDIA NVFP4 deployment."
   ],
   "project_patches": {
    "required": true,
    "items": [
     "patches/qwen36-27b-q8-tp2-asrock-b70/llama-cpp-mndodd-4302fb599-lab-tp2-dp4a2-20260815.diff.gz.b64",
     "patches/qwen38-27b-q4km-tp2-asrock-b70/llama-cpp-q4k-mmvq-swiglu-tp2-20260815.diff.gz.b64",
     "patches/qwen38-27b-q4km-tp1-b70s/llama-cpp-tp1-gdn-state-io-widen-20260821.diff.gz.b64",
     "patches/qwen38-27b-q4km-tp1-b70s/llama-cpp-tp1-conv-qk-widen-20260821.diff.gz.b64",
     "patches/qwen38-27b-q4km-tp1-b70s/llama-cpp-tp1-qk-norm-rope-src-widen-20260821.diff.gz.b64",
     "patches/qwen38-27b-q4km-tp1-b70s/llama-cpp-tp1-q8out-rejected-memo320-20260821.diff.gz.b64"
    ],
    "reason": "The one-card speed and exactness identity depends on the lab's complete source stack and TP1 matcher increments."
   },
   "commands": {
    "preflight": "MODEL_DIR=/path/to/qwen3.8-27b-q4km BUILD_DIR=/path/to/llama.cpp/build-sycl-aot-bmg-g31 repro/qwen38-27b-q4km-tp1-b70/preflight.sh",
    "launch": "MODEL_DIR=/path/to/qwen3.8-27b-q4km BUILD_DIR=/path/to/llama.cpp/build-sycl-aot-bmg-g31 repro/qwen38-27b-q4km-tp1-b70/run-server.sh",
    "health": "curl -fsS http://127.0.0.1:18088/health",
    "benchmark": "OUT=/path/to/result.json repro/qwen38-27b-q4km-tp1-b70/bench.sh",
    "stop": "Ctrl-C in the foreground server terminal; then pgrep -x llama-server must return no process"
   },
   "dependencies": [
    "repro/qwen38-27b-q4km-tp1-b70/README.md",
    "repro/gemma4-26b-a4b-q8-b70/realistic-suite-v1.json",
    "repro/qwen36-27b-autoround-int4-b70/realistic-suite-v1.json",
    "repro/qwen38-27b-autoround-int4-b70/scripts/verify-model-direct.py",
    "repro/qwen38-27b-q4km-tp1-b70/model-direct.json",
    "repro/qwen38-27b-q4km-tp1-b70/model-verification-20260822.json",
    "repro/qwen38-27b-q4km-tp1-b70/verify-model-direct.sh",
    "repro/qwen38-27b-q4km-tp1-b70/restore-and-build.sh",
    "repro/qwen38-27b-q4km-tp1-b70/preflight.sh",
    "repro/qwen38-27b-q4km-tp1-b70/run-server.sh",
    "repro/qwen38-27b-q4km-tp1-b70/bench.sh",
    "repro/qwen38-27b-q4km-tp1-b70/CLEAN-HOST.md",
    "repro/qwen38-27b-q4km-tp1-b70/clean-host-inventory.sh",
    "patches/qwen38-27b-q4km-tp1-b70s/README.md",
    "experiments/qwen38-27b-b70/notes/2026-08-21-qwen38-q4km-tp1-quality-battery-result.md",
    "experiments/qwen38-27b-b70/data/2026-08-21-q4km-tp1-gpu0-final-i.json",
    "experiments/qwen38-27b-b70/data/2026-08-21-q4km-tp1-gpu0-final-j.json",
    "experiments/qwen38-27b-b70/data/2026-08-21-q4km-tp1-gpu0-quality-battery.json",
    "experiments/qwen38-27b-b70/data/2026-08-25-qwen38-q4km-tp1-http-smallctx-r1-prereg.json",
    "experiments/qwen38-27b-b70/data/2026-08-25-qwen38-q4km-tp1-http-smallctx-r1-summary.json",
    "experiments/qwen38-27b-b70/notes/2026-08-25-qwen38-q4km-tp1-http-smallctx-r1-result.md",
    "experiments/qwen38-27b-b70/scripts/run-qwen38-q4km-tp1-http-smallctx.sh",
    "experiments/qwen38-27b-b70/notes/2026-08-25-qwen38-q4km-tp1-batched-ladder-result.md",
    "experiments/qwen38-27b-b70/data/qwen38-q4km-tp1-batched-ladder-20260825-r1-attempt2/summary.json",
    "experiments/qwen38-27b-b70/data/2026-08-25-qwen38-q4km-tp1-http-depth-r1-prereg.json",
    "experiments/qwen38-27b-b70/data/2026-08-25-qwen38-q4km-tp1-http-depth-r1-result.json",
    "experiments/qwen38-27b-b70/notes/2026-08-25-qwen38-q4km-tp1-http-depth-r1-result.md",
    "experiments/qwen38-27b-b70/scripts/run-qwen38-q4km-tp1-http-depth.sh",
    "experiments/qwen38-27b-b70/data/2026-08-25-qwen38-q4km-tp1-http-concurrency-r2-prereg.json",
    "experiments/qwen38-27b-b70/data/2026-08-25-qwen38-q4km-tp1-http-concurrency-r2-result.json",
    "experiments/qwen38-27b-b70/notes/2026-08-25-qwen38-q4km-tp1-http-concurrency-r2-result.md",
    "experiments/qwen38-27b-b70/scripts/run-qwen38-q4km-tp1-http-concurrency-r2.sh",
    "experiments/qwen38-27b-b70/data/2026-08-25-qwen38-q4km-tp1-http-concurrency-oracle-digests.json",
    "experiments/qwen38-27b-b70/data/2026-08-25-qwen38-q4km-tp1-http-concurrency-r3-prereg.json",
    "experiments/qwen38-27b-b70/data/2026-08-25-qwen38-q4km-tp1-http-concurrency-r3-result.json",
    "experiments/qwen38-27b-b70/notes/2026-08-25-qwen38-q4km-tp1-http-concurrency-r3-result.md",
    "experiments/qwen38-27b-b70/scripts/run-qwen38-q4km-tp1-http-concurrency-r3.sh",
    "scripts/build-concurrency-oracle-digests.py",
    "scripts/bench-openai-concurrency-oracle.py",
    "scripts/bench-openai-realistic-suite.py",
    "community/mndodd-qwen36-27b-llamacpp-sycl/STATUS.md",
    "patches/qwen36-27b-q8-tp2-asrock-b70/llama-cpp-mndodd-4302fb599-lab-tp2-dp4a2-20260815.diff.gz.b64",
    "patches/qwen38-27b-q4km-tp2-asrock-b70/llama-cpp-q4k-mmvq-swiglu-tp2-20260815.diff.gz.b64",
    "patches/qwen38-27b-q4km-tp1-b70s/llama-cpp-tp1-gdn-state-io-widen-20260821.diff.gz.b64",
    "patches/qwen38-27b-q4km-tp1-b70s/llama-cpp-tp1-conv-qk-widen-20260821.diff.gz.b64",
    "patches/qwen38-27b-q4km-tp1-b70s/llama-cpp-tp1-qk-norm-rope-src-widen-20260821.diff.gz.b64",
    "patches/qwen38-27b-q4km-tp1-b70s/llama-cpp-tp1-q8out-rejected-memo320-20260821.diff.gz.b64",
    "experiments/qwen38-27b-b70/data/2026-08-22-q4km-tp1-context-kv-sweep.json"
   ],
   "missing": [
    "tested clean-host Intel driver and oneAPI installation",
    "clean-host source build and endpoint replay",
    "batch-shape-invariant greedy output for multi-user serving"
   ]
  },
  {
   "manifest": "packages/qwen38-27b-q4km-tp2-asrock-b70/package.json",
   "format": "b70-model-package-v1",
   "id": "qwen38-27b-q4km-tp2-asrock-b70",
   "name": "Qwen3.8 27B Q4_K_M on two Intel Arc Pro B70 cards",
   "status": "candidate",
   "audience": "expert",
   "guide": "repro/qwen38-27b-q4km-tp2-asrock-b70/README.md",
   "clean_host_tested": false,
   "library": {
    "model_family": "Qwen3.8",
    "publisher": "Alibaba / Qwen",
    "variant": "27B Q4_K_M target-only TP2",
    "summary": "Alibaba's Qwen3.8 27B on two Arc Pro B70 cards in 4-bit form with llama.cpp, no draft model. Roughly 1.8x the one-card speed using the lab's two-card kernel stack.",
    "quantization": "Q4_K_M / F16 KV",
    "runtime_label": "llama.cpp SYCL",
    "operating_systems": [
     "Linux"
    ],
    "delivery": [
     "native"
    ],
    "modalities": [
     "text"
    ],
    "use_cases": [
     "general",
     "coding",
     "long context",
     "two-card throughput"
    ],
    "tags": [
     "two cards",
     "TP2",
     "target only",
     "no speculation",
     "quality gated"
    ],
    "published_at": "2026-08-23",
    "featured_metric": {
     "value": 49.71750333219927,
     "unit": "tok/s",
     "label": "conventional decode median",
     "scope": "99-interval median across the fixed 12-prompt cache-zero suite; target-only, reasoning off, 12/12 exact output hashes.",
     "evidence": "experiments/qwen38-27b-b70/data/2026-08-15-q4km-tp2-q4k-glu-summary.json"
    }
   },
   "contributors": [
    {
     "id": "b70-optimization-lab",
     "name": "neural.download lab",
     "kind": "lab",
     "profile": "https://github.com/steveseguin/b70-optimization-lab",
     "contribution": "Qwen3.8 Q4_K_M transfer, complete B70 TP2 patch stack, Q4_K gate/up SwiGLU fusion, quality gates, benchmarking, and packaging.",
     "status": "integrated",
     "validated_effect": "The Q4_K fusion measured 50.271708 versus 49.460273 tok/s in a same-binary llama-bench A/B (+1.6406%); the final endpoint measured 49.717503 tok/s with 12/12 exact output hashes.",
     "evidence": "experiments/qwen38-27b-b70/data/2026-08-15-q4km-tp2-q4k-glu-summary.json"
    },
    {
     "id": "mndodd",
     "name": "mndodd",
     "kind": "external",
     "profile": "https://github.com/mndodd",
     "contribution": "Provided the optimized Intel SYCL llama.cpp fork used as the clean public base beneath the lab's TP2 patch stack.",
     "status": "integrated",
     "validated_effect": "A separate matched Qwen3.6 Q8 TP2 A/B measured 31.338765 versus 29.610651 tok/s (+5.836%) for that base; the Qwen3.8 Q4 result remains a separately measured lab lane.",
     "evidence": "community/mndodd-qwen36-27b-llamacpp-sycl/STATUS.md"
    }
   ],
   "performance_profiles": [
    {
     "id": "http-decode-vs-active-context",
     "label": "Qualified TP2 HTTP decode over exact active context",
     "metric": "decode",
     "unit": "tok/s",
     "x_label": "Exact active prompt tokens",
     "scope": "One-slot native HTTP completions on the exact promoted Q4_K_M TP2 stack, two B70s, equal tensor split, F16 KV, 33,024 configured context, cache disabled, no truncation or context shift, and 128 returned token IDs. The fixture is grade C repeated-token shape evidence, not natural prose. Every marker is measured; no point is interpolated or extrapolated.",
     "evidence": "experiments/qwen38-27b-b70/data/qwen38-q4km-tp2-http-depth-20260825-r1-attempt1/summary.json",
     "points": [
      {
       "context_tokens": 2048,
       "value": 49.489490115227746,
       "samples": 1
      },
      {
       "context_tokens": 4096,
       "value": 49.01030658650703,
       "samples": 1
      },
      {
       "context_tokens": 8192,
       "value": 48.30039279903885,
       "samples": 1
      },
      {
       "context_tokens": 16384,
       "value": 47.03057165241295,
       "samples": 1
      },
      {
       "context_tokens": 24576,
       "value": 45.53456055073531,
       "samples": 1
      },
      {
       "context_tokens": 32768,
       "value": 44.43728051677345,
       "samples": 1
      }
     ]
    },
    {
     "id": "http-ttft-vs-active-context",
     "label": "Qualified TP2 HTTP TTFT over exact active context",
     "metric": "ttft",
     "unit": "ms",
     "x_label": "Exact active prompt tokens",
     "scope": "TTFT from the same one-slot exact-token HTTP receipts as the TP2 decode curve. Cache was zero and prompt counts were exact at every marker. The fixture is grade C repeated-token shape evidence and is not a natural-prose latency claim. Every marker is measured; no point is interpolated or extrapolated.",
     "evidence": "experiments/qwen38-27b-b70/data/qwen38-q4km-tp2-http-depth-20260825-r1-attempt1/summary.json",
     "points": [
      {
       "context_tokens": 2048,
       "value": 1945.0462919194251,
       "samples": 1
      },
      {
       "context_tokens": 4096,
       "value": 3860.131878987886,
       "samples": 1
      },
      {
       "context_tokens": 8192,
       "value": 7861.680709989741,
       "samples": 1
      },
      {
       "context_tokens": 16384,
       "value": 16299.778147018515,
       "samples": 1
      },
      {
       "context_tokens": 24576,
       "value": 25347.310534096323,
       "samples": 1
      },
      {
       "context_tokens": 32768,
       "value": 35058.737567975186,
       "samples": 1
      }
     ]
    },
    {
     "id": "http-output-audited-aggregate-vs-concurrent-users",
     "label": "Output-audited TP2 HTTP aggregate decode",
     "metric": "aggregate_decode",
     "unit": "tok/s",
     "x_metric": "concurrent_sequences",
     "x_label": "Concurrent HTTP users",
     "scope": "Best qualified llama.cpp native HTTP aggregate results on two B70s with prompt caching disabled, unique short prompts, and 128 returned token IDs/request. The 1\u201332 markers are medians from the original 32K two-attempt curve; c64 is the later exact ffn_down+ffn_gate-cache center at 175.623794 tok/s. The c96 endpoint is a separate near-capacity profile: llama.cpp rounded the requested 32K pool to an effective 49,152 tokens (96x512), two fresh candidates centered at 192.341954 tok/s and matched a frozen same-shape control-batch oracle 96/96 each. Its isolated sequential comparison was only 50/96, so it qualifies candidate-vs-control identity and capacity, not batch-invariant text. Peak used VRAM was about 30.48/30.35 GiB. No point is interpolated or extrapolated.",
     "evidence": "experiments/qwen38-27b-b70/data/2026-08-30-qwen38-q4km-tp2-exact-cache-c96-r14-results.json",
     "points": [
      {
       "concurrent_sequences": 1,
       "value": 42.694235799361806,
       "per_user_value": 42.694235799361806,
       "samples": 2
      },
      {
       "concurrent_sequences": 2,
       "value": 61.88466425158718,
       "per_user_value": 30.94233212579359,
       "samples": 2
      },
      {
       "concurrent_sequences": 4,
       "value": 87.56640520423115,
       "per_user_value": 21.891601301057786,
       "samples": 2
      },
      {
       "concurrent_sequences": 8,
       "value": 108.37157051235434,
       "per_user_value": 13.546446314044292,
       "samples": 2
      },
      {
       "concurrent_sequences": 16,
       "value": 109.14655327814907,
       "per_user_value": 6.821659579884317,
       "samples": 2
      },
      {
       "concurrent_sequences": 32,
       "value": 127.49980108773008,
       "per_user_value": 3.984368783991565,
       "samples": 2
      },
      {
       "concurrent_sequences": 64,
       "value": 175.6237938170031,
       "per_user_value": 2.7441217783906735,
       "samples": 2
      },
      {
       "concurrent_sequences": 96,
       "value": 192.34195380893703,
       "per_user_value": 2.0035620188430943,
       "samples": 2
      }
     ]
    }
   ],
   "hardware": {
    "accelerator": "Intel Arc Pro B70 32 GiB",
    "cards": 2,
    "model_weight_bytes": 18973870432
   },
   "model": {
    "repository": "ggml-org/Qwen3.8-27B-GGUF",
    "revision": "0669b98607d47046c7c2b3f801011d54a08cfccf",
    "manifest": "repro/qwen38-27b-q4km-tp1-b70/model-direct.json"
   },
   "runtime": {
    "kind": "native",
    "repository": "mndodd/llama.cpp",
    "revision": "4302fb59969a5d8cf9f8e5f55fdd4506d0ed2126",
    "build": "patches/qwen38-27b-q4km-tp2-asrock-b70/README.md"
   },
   "project_patches": {
    "required": true,
    "items": [
     "patches/qwen36-27b-q8-tp2-asrock-b70/llama-cpp-mndodd-4302fb599-lab-tp2-dp4a2-20260815.diff.gz.b64",
     "patches/qwen38-27b-q4km-tp2-asrock-b70/llama-cpp-q4k-mmvq-swiglu-tp2-20260815.diff.gz.b64"
    ],
    "reason": "The result depends on the complete lab TP2 source stack and the Q4_K dense-FFN fusion increment."
   },
   "commands": {
    "preflight": "QWEN38_SOURCE_DIR=/path/to/patched/source QWEN38_BUILD_DIR=/path/to/build QWEN38_MODEL=/path/to/Qwen3.8-27B-Q4_K_M.gguf repro/qwen38-27b-q4km-tp2-asrock-b70/preflight.sh",
    "launch": "QWEN38_SOURCE_DIR=/path/to/patched/source QWEN38_BUILD_DIR=/path/to/build QWEN38_MODEL=/path/to/Qwen3.8-27B-Q4_K_M.gguf repro/qwen38-27b-q4km-tp2-asrock-b70/run-server.sh",
    "health": "curl -fsS http://127.0.0.1:18087/health",
    "benchmark": "OUT=/path/to/result.json repro/qwen38-27b-q4km-tp2-asrock-b70/bench.sh",
    "stop": "Stop the foreground server with Ctrl-C, then verify clean teardown with pgrep -af llama-server."
   },
   "dependencies": [
    "repro/qwen38-27b-q4km-tp2-asrock-b70/README.md",
    "repro/gemma4-26b-a4b-q8-b70/realistic-suite-v1.json",
    "repro/qwen36-27b-autoround-int4-b70/realistic-suite-v1.json",
    "repro/qwen38-27b-autoround-int4-b70/scripts/verify-model-direct.py",
    "scripts/bench-openai-realistic-suite.py",
    "repro/qwen38-27b-q4km-tp2-asrock-b70/preflight.sh",
    "repro/qwen38-27b-q4km-tp2-asrock-b70/config.env",
    "repro/qwen38-27b-q4km-tp2-asrock-b70/runtime-common.sh",
    "repro/qwen38-27b-q4km-tp2-asrock-b70/run-server.sh",
    "repro/qwen38-27b-q4km-tp2-asrock-b70/bench.sh",
    "repro/qwen38-27b-q4km-tp1-b70/model-direct.json",
    "repro/qwen38-27b-q4km-tp1-b70/verify-model-direct.sh",
    "patches/qwen36-27b-q8-tp2-asrock-b70/README.md",
    "patches/qwen38-27b-q4km-tp2-asrock-b70/README.md",
    "experiments/qwen38-27b-b70/data/2026-08-15-q4km-tp2-q4k-glu-summary.json",
    "experiments/qwen38-27b-b70/data/2026-08-25-qwen38-q4km-tp2-http-concurrency-r2-result.json",
    "experiments/qwen38-27b-b70/data/2026-08-30-qwen38-q4km-tp2-public-profile-cache-c64-r8-results.json",
    "experiments/qwen38-27b-b70/notes/2026-08-30-qwen38-q4km-tp2-exact-f16-cache-c64-result.md",
    "experiments/qwen38-27b-b70/patches/llama-qwen38-q4k-f16-exact-weight-cache-candidate-20260830.patch",
    "experiments/qwen38-27b-b70/patches/llama-server-fixed-inference-cohort-admission-20260830.patch",
    "experiments/qwen38-27b-b70/data/2026-08-30-qwen38-q4km-tp2-exact-cache-pair-r10-results.json",
    "experiments/qwen38-27b-b70/notes/2026-08-30-qwen38-q4km-tp2-exact-f16-cache-pair-c64-result.md",
    "experiments/qwen38-27b-b70/patches/llama-qwen38-q4k-f16-cache-comma-filter-20260830.patch",
    "experiments/qwen38-27b-b70/data/2026-08-30-qwen38-q4km-tp2-c96-batch-oracle-r14.json",
    "experiments/qwen38-27b-b70/data/2026-08-30-qwen38-q4km-tp2-exact-cache-c96-r14-results.json",
    "experiments/qwen38-27b-b70/notes/2026-08-30-qwen38-q4km-tp2-exact-f16-cache-c96-result.md",
    "community/mndodd-qwen36-27b-llamacpp-sycl/STATUS.md",
    "patches/qwen36-27b-q8-tp2-asrock-b70/llama-cpp-mndodd-4302fb599-lab-tp2-dp4a2-20260815.diff.gz.b64",
    "patches/qwen38-27b-q4km-tp2-asrock-b70/llama-cpp-q4k-mmvq-swiglu-tp2-20260815.diff.gz.b64",
    "experiments/qwen38-27b-b70/data/qwen38-q4km-tp2-http-depth-20260825-r1-attempt1/summary.json"
   ],
   "known_limitations": [
    "The package has not been replayed from a clean host and is not a beginner install guide; source-rebuilt binaries require an explicit preflight override and the full output-oracle gate.",
    "The 49.717503 tok/s headline is a TP2 target-only reasoning-off result and must not be compared as a TP1, speculative, or alternate-accounting row.",
    "The optional large-batch mode improves prefill on a short probe while slightly reducing decode; it is not the headline configuration.",
    "Multi-user greedy token identity is batch-shape-dependent. The qualified HTTP curve proves complete isolated responses with zero cross-base oracle collisions, not sequential byte identity.",
    "The exact F16 ffn_down+ffn_gate cache is aggregate-only and uses approximately 13 GiB of additional device memory per card. Its c96 endpoint reached 192.341954 tok/s but used about 30.48/30.35 GiB per card and matched the same-shape batch oracle 96/96 while matching isolated sequential references only 50/96.",
    "The c96 endpoint includes peak VRAM samples, but no active-context prefill, full memory, or power curve has been captured for this TP2 identity."
   ],
   "missing": [
    "tested platform installation",
    "self-contained model download helper",
    "clean-host replay",
    "beginner recovery flow",
    "active-context prefill and power profiles plus a full memory curve"
   ]
  },
  {
   "manifest": "packages/qwen38-27b-q8-q4mtp-mtp2-tp1-b70/package.json",
   "format": "b70-model-package-v1",
   "id": "qwen38-27b-q8-q4mtp-mtp2-tp1-b70",
   "name": "Qwen3.8 27B Q8_0 + MTP2 on one Intel Arc Pro B70",
   "status": "candidate",
   "audience": "intermediate",
   "guide": "repro/qwen38-27b-q8-q4mtp-mtp2-tp1-b70/README.md",
   "clean_host_tested": false,
   "library": {
    "model_family": "Qwen3.8",
    "publisher": "Alibaba / Qwen",
    "variant": "27B",
    "summary": "The quality-conservative Q8_0 target with a separately downloaded Q4_0 MTP draft at depth 2. Two fresh servers measured 37.062 tok/s with all 24 speculative outputs exact to a matched MTP0 control.",
    "quantization": "Q8_0 target + Q4_0 MTP draft",
    "runtime_label": "llama.cpp SYCL",
    "operating_systems": [
     "Linux"
    ],
    "delivery": [
     "native"
    ],
    "modalities": [
     "text"
    ],
    "use_cases": [
     "general",
     "coding",
     "high quality",
     "interactive"
    ],
    "tags": [
     "one card",
     "Q8",
     "MTP2",
     "speculative decoding",
     "cache zero",
     "source build"
    ],
    "published_at": "2026-08-27",
    "featured_metric": {
     "value": 37.06202846372931,
     "unit": "tok/s",
     "label": "strict varied-prompt decode",
     "scope": "Median of two fresh-server class-balanced medians over the fixed 12-prompt/six-class, 512-cap native HTTP suite; one B70, TP1, 1024-token configured context, MTP2, F16 target/draft KV, cache zero, and 24/24 complete MTP2 arrays exact to matched MTP0.",
     "evidence": "experiments/qwen38-27b-b70/data/2026-08-27-qwen38-q8-q4mtp-tp1-mtp2-strict-r1-result.json"
    }
   },
   "contributors": [
    {
     "id": "b70-optimization-lab",
     "name": "neural.download lab",
     "kind": "lab",
     "profile": "https://github.com/steveseguin/b70-optimization-lab",
     "contribution": "Qwen3.8 Q8 kernel stack, one-card external-MTP fit work, strict MTP0/1/2 screen, target oracle, package, and validation.",
     "status": "integrated",
     "validated_effect": "MTP2 measured 37.062028 tok/s versus 19.582597 for the matched target-only control (+89.26%), with 24/24 complete MTP2 outputs exact to control across two fresh servers.",
     "evidence": "experiments/qwen38-27b-b70/notes/2026-08-27-qwen38-q8-q4mtp-tp1-mtp2-strict-r1-result.md"
    },
    {
     "id": "mndodd",
     "name": "mndodd",
     "kind": "external",
     "profile": "https://github.com/mndodd",
     "contribution": "Optimized Intel SYCL llama.cpp fork used as the pinned runtime base beneath the lab patch stack.",
     "status": "integrated",
     "validated_effect": "Credited for the pinned fork contribution documented by the base Q8 package; the external-MTP integration and measured gain are this lab's work.",
     "evidence": "community/mndodd-qwen36-27b-llamacpp-sycl/STATUS.md"
    }
   ],
   "performance_profiles": [
    {
     "id": "strict-decode-vs-mtp-depth",
     "label": "Strict Q8 decode over MTP depth",
     "metric": "decode",
     "unit": "tok/s",
     "x_metric": "speculative_tokens",
     "x_label": "Maximum MTP draft depth",
     "scope": "One-B70 fixed full-suite screen at 1024 configured context under one target/runtime identity. MTP0 and MTP1 are one fresh-server values; MTP2 is the median of two fresh servers. All speculative outputs were exact to the matched MTP0 oracle. No value is interpolated or extrapolated.",
     "evidence": "experiments/qwen38-27b-b70/data/2026-08-27-qwen38-q8-q4mtp-tp1-mtp2-strict-r1-result.json",
     "points": [
      {
       "speculative_tokens": 0,
       "value": 19.58259717754693,
       "samples": 1
      },
      {
       "speculative_tokens": 1,
       "value": 30.260757687210067,
       "samples": 1
      },
      {
       "speculative_tokens": 2,
       "value": 37.06202846372931,
       "samples": 2
      }
     ]
    }
   ],
   "hardware": {
    "accelerator": "Intel Arc Pro B70 32 GiB",
    "cards": 1,
    "host_ram_min_bytes": 17179869184,
    "host_ram_plus_swap_min_bytes": 25769803776,
    "model_weight_bytes": 29965354208
   },
   "model": {
    "repository": "ggml-org/Qwen3.8-27B-GGUF plus unsloth/Qwen3.8-27B-GGUF MTP",
    "revision": "0669b98607d47046c7c2b3f801011d54a08cfccf",
    "manifest": "repro/qwen38-27b-q8-tp1-b70/model-direct.json"
   },
   "runtime": {
    "kind": "native",
    "project": "mndodd/llama.cpp",
    "revision": "4302fb59969a5d8cf9f8e5f55fdd4506d0ed2126",
    "build": "repro/qwen38-27b-q8-q4mtp-mtp2-tp1-b70/restore-and-build.sh"
   },
   "known_limitations": [
    "The 37.062028 tok/s headline is a one-slot short-context result with a 1024-token configured context. No Q8+MTP2 long-context or concurrency value has been measured.",
    "All 24 MTP2 outputs matched the matched MTP0 oracle in the strict suite, but this does not prove universal parity for every content or context shape.",
    "The tested 16 GiB host used swap and a 13 GiB process memory cap. A 32 GiB or larger host is simpler.",
    "A clean-host Intel/oneAPI installation and source-build replay remain pending.",
    "The target and MTP draft come from two pinned repositories/revisions and both downloads are required."
   ],
   "project_patches": {
    "required": true,
    "items": [
     "patches/qwen36-27b-q8-tp2-asrock-b70/llama-cpp-mndodd-4302fb599-lab-tp2-dp4a2-20260815.diff.gz.b64",
     "patches/qwen38-27b-q4km-tp2-asrock-b70/llama-cpp-q4k-mmvq-swiglu-tp2-20260815.diff.gz.b64",
     "patches/qwen38-27b-q4km-tp1-b70s/llama-cpp-tp1-gdn-state-io-widen-20260821.diff.gz.b64",
     "patches/qwen38-27b-q4km-tp1-b70s/llama-cpp-tp1-conv-qk-widen-20260821.diff.gz.b64",
     "patches/qwen38-27b-q4km-tp1-b70s/llama-cpp-tp1-qk-norm-rope-src-widen-20260821.diff.gz.b64",
     "patches/qwen38-27b-q4km-tp1-b70s/llama-cpp-tp1-q8out-rejected-memo320-20260821.diff.gz.b64"
    ],
    "reason": "The measured Q8 target and MTP verifier path uses the complete lab TP1 source stack; the guide links and verifies every patch."
   },
   "commands": {
    "preflight": "TARGET_DIR=/path/to/target DRAFT_DIR=/path/to/draft BUILD_DIR=/path/to/build repro/qwen38-27b-q8-q4mtp-mtp2-tp1-b70/preflight.sh",
    "launch": "TARGET_DIR=/path/to/target DRAFT_DIR=/path/to/draft BUILD_DIR=/path/to/build repro/qwen38-27b-q8-q4mtp-mtp2-tp1-b70/run-server.sh",
    "health": "curl -fsS http://127.0.0.1:18141/health",
    "benchmark": "MTP_DEPTH=2 ATTEMPT=my-attempt TARGET_DIR=/path/to/target DRAFT_DIR=/path/to/draft BUILD_DIR=/path/to/build OUT_DIR=/path/to/new-result experiments/qwen38-27b-b70/scripts/run-20260827-qwen38-q8-q4mtp-tp1-screen-attempt.sh",
    "stop": "Ctrl-C in the foreground server terminal; then pgrep -x llama-server must return no process"
   },
   "dependencies": [
    "repro/qwen38-27b-q8-q4mtp-mtp2-tp1-b70/README.md",
    "experiments/qwen38-27b-b70/data/2026-08-27-qwen38-q4mtp-draft-direct.json",
    "experiments/qwen38-27b-b70/data/qwen38-q8-tp1-strict-reasoningoff-native-20260827-r1b/performance.json",
    "repro/gemma4-26b-a4b-q8-b70/realistic-suite-v1.json",
    "repro/qwen36-27b-autoround-int4-b70/realistic-suite-v1.json",
    "repro/qwen38-27b-autoround-int4-b70/scripts/verify-model-direct.py",
    "repro/qwen38-27b-q8-tp1-b70/verify-model-direct.sh",
    "repro/qwen38-27b-q8-q4mtp-mtp2-tp1-b70/verify-models.sh",
    "repro/qwen38-27b-q8-q4mtp-mtp2-tp1-b70/preflight.sh",
    "repro/qwen38-27b-q8-q4mtp-mtp2-tp1-b70/restore-and-build.sh",
    "repro/qwen38-27b-q8-q4mtp-mtp2-tp1-b70/run-server.sh",
    "repro/qwen38-27b-q8-tp1-b70/model-direct.json",
    "repro/qwen38-27b-q4km-mtp2-tp1-b70/draft-model-direct.json",
    "experiments/qwen38-27b-b70/data/2026-08-27-qwen38-q8-q4mtp-tp1-depth-screen-r1-prereg.json",
    "experiments/qwen38-27b-b70/data/2026-08-27-qwen38-q8-q4mtp-tp1-depth-screen-r1-control-amendment.json",
    "experiments/qwen38-27b-b70/data/2026-08-27-qwen38-q8-q4mtp-tp1-mtp2-strict-r1-result.json",
    "experiments/qwen38-27b-b70/notes/2026-08-27-qwen38-q8-q4mtp-tp1-mtp2-strict-r1-result.md",
    "experiments/qwen38-27b-b70/scripts/run-20260827-qwen38-q8-q4mtp-tp1-screen-attempt.sh",
    "experiments/qwen38-27b-b70/scripts/validate-20260827-qwen38-q8-q4mtp-tp1-mtp2-strict-r1.py",
    "scripts/bench-openai-realistic-suite.py",
    "scripts/neural-download-canaries.py",
    "community/mndodd-qwen36-27b-llamacpp-sycl/STATUS.md",
    "patches/qwen36-27b-q8-tp2-asrock-b70/llama-cpp-mndodd-4302fb599-lab-tp2-dp4a2-20260815.diff.gz.b64",
    "patches/qwen38-27b-q4km-tp2-asrock-b70/llama-cpp-q4k-mmvq-swiglu-tp2-20260815.diff.gz.b64",
    "patches/qwen38-27b-q4km-tp1-b70s/llama-cpp-tp1-gdn-state-io-widen-20260821.diff.gz.b64",
    "patches/qwen38-27b-q4km-tp1-b70s/llama-cpp-tp1-conv-qk-widen-20260821.diff.gz.b64",
    "patches/qwen38-27b-q4km-tp1-b70s/llama-cpp-tp1-qk-norm-rope-src-widen-20260821.diff.gz.b64",
    "patches/qwen38-27b-q4km-tp1-b70s/llama-cpp-tp1-q8out-rejected-memo320-20260821.diff.gz.b64"
   ],
   "missing": [
    "Q8 plus MTP2 realistic-content HTTP speed and TTFT from 2K through 32K",
    "Q8 plus MTP2 output-qualified concurrency",
    "tested clean-host Intel driver and oneAPI installation",
    "clean-host source build and endpoint replay",
    "beginner recovery flow"
   ]
  },
  {
   "manifest": "packages/qwen38-27b-q8-tp1-b70/package.json",
   "format": "b70-model-package-v1",
   "id": "qwen38-27b-q8-tp1-b70",
   "name": "Qwen3.8 27B Q8_0 on one Intel Arc Pro B70",
   "status": "candidate",
   "audience": "intermediate",
   "guide": "repro/qwen38-27b-q8-tp1-b70/README.md",
   "clean_host_tested": false,
   "library": {
    "model_family": "Qwen3.8",
    "publisher": "Alibaba / Qwen",
    "variant": "27B",
    "summary": "A quality-conservative one-card Qwen3.8 lane using Q8_0 weights, F16 KV, llama.cpp/SYCL, and no speculative decoding.",
    "quantization": "Q8_0",
    "runtime_label": "llama.cpp SYCL",
    "operating_systems": [
     "Linux"
    ],
    "delivery": [
     "native"
    ],
    "modalities": [
     "text"
    ],
    "use_cases": [
     "general",
     "coding",
     "long context"
    ],
    "tags": [
     "one card",
     "target only",
     "cache zero",
     "source build",
     "quality conservative"
    ],
    "published_at": "2026-08-27",
    "featured_metric": {
     "value": 19.619239834548473,
     "unit": "tok/s",
     "label": "strict varied-prompt decode",
     "scope": "Median of two fresh-server class-balanced medians over the full 12-prompt/six-class, 512-cap, cache-zero raw-completion suite; target-only TP1, MTP0, 12/12 complete token arrays exact within TP1 and all objective canaries passed.",
     "evidence": "experiments/qwen38-27b-b70/data/2026-08-27-qwen38-q8-tp1-strict-reasoningoff-native-r1-result.json"
    }
   },
   "contributors": [
    {
     "id": "b70-optimization-lab",
     "name": "neural.download lab",
     "kind": "lab",
     "profile": "https://github.com/steveseguin/b70-optimization-lab",
     "contribution": "Qwen3.8 Q8 one-card bring-up, source/patch integration, exact depth measurement, service-quality gate, and package.",
     "status": "integrated",
     "validated_effect": "The exact package tuple completed all 14 depth/metric rows through 32K and passed 7/7 canaries, 8/8 repeat stability, long-context needle recall, and 16/16 cache-zero responses.",
     "evidence": "experiments/qwen38-27b-b70/notes/2026-08-25-qwen38-q8weights-f16-tp1-service-quality-r1-result.md"
    },
    {
     "id": "mndodd",
     "name": "mndodd",
     "kind": "external",
     "profile": "https://github.com/mndodd",
     "contribution": "Optimized Intel SYCL llama.cpp fork used as the pinned runtime base beneath the lab patch stack.",
     "status": "integrated",
     "validated_effect": "A separate matched Qwen3.6 Q8 TP2 control measured +5.836% for the fork; that A/B does not assign this later Qwen3.8 one-card result to the fork.",
     "evidence": "community/mndodd-qwen36-27b-llamacpp-sycl/STATUS.md"
    }
   ],
   "performance_profiles": [
    {
     "id": "decode-f16-kv-vs-context-depth",
     "label": "Q8_0 raw decode with F16 KV",
     "metric": "decode",
     "unit": "tok/s",
     "x_label": "Existing context depth before tg128",
     "scope": "Direct llama-bench raw-engine tg128, exact Q8_0 TP1 tuple, one B70, F16 K/V, flash attention on, and five repetitions per displayed depth. Every marker is measured; no point is interpolated or extrapolated.",
     "evidence": "experiments/qwen38-27b-b70/data/qwen38-q8weights-f16-tp1-local-20260825-r2/result.json",
     "points": [
      {
       "context_tokens": 0,
       "value": 19.662501,
       "samples": 5
      },
      {
       "context_tokens": 2048,
       "value": 19.592508,
       "samples": 5
      },
      {
       "context_tokens": 4096,
       "value": 19.509481,
       "samples": 5
      },
      {
       "context_tokens": 8192,
       "value": 19.293993,
       "samples": 5
      },
      {
       "context_tokens": 16384,
       "value": 18.83887,
       "samples": 5
      },
      {
       "context_tokens": 24576,
       "value": 18.418098,
       "samples": 5
      },
      {
       "context_tokens": 32768,
       "value": 18.023689,
       "samples": 5
      }
     ]
    },
    {
     "id": "prefill-f16-kv-vs-context-depth",
     "label": "Q8_0 raw pp2048 with F16 KV",
     "metric": "prefill",
     "unit": "tok/s",
     "x_label": "Existing context depth before pp2048",
     "scope": "Direct llama-bench raw-engine pp2048, exact Q8_0 TP1 tuple, one B70, F16 K/V, flash attention on, and five repetitions per displayed depth. Every marker is measured; no point is interpolated or extrapolated.",
     "evidence": "experiments/qwen38-27b-b70/data/qwen38-q8weights-f16-tp1-local-20260825-r2/result.json",
     "points": [
      {
       "context_tokens": 0,
       "value": 996.89102,
       "samples": 5
      },
      {
       "context_tokens": 2048,
       "value": 987.388875,
       "samples": 5
      },
      {
       "context_tokens": 4096,
       "value": 959.560446,
       "samples": 5
      },
      {
       "context_tokens": 8192,
       "value": 914.062353,
       "samples": 5
      },
      {
       "context_tokens": 16384,
       "value": 837.512717,
       "samples": 5
      },
      {
       "context_tokens": 24576,
       "value": 773.146408,
       "samples": 5
      },
      {
       "context_tokens": 32768,
       "value": 719.144647,
       "samples": 5
      }
     ]
    },
    {
     "id": "http-output-audited-aggregate-vs-concurrent-users",
     "label": "Queued Q8_0 TP1 HTTP aggregate decode",
     "metric": "aggregate_decode",
     "unit": "tok/s",
     "x_metric": "concurrent_sequences",
     "x_label": "Simultaneous HTTP requests",
     "scope": "llama.cpp native HTTP /completion on one B70 with eight active slots, 4K total F16 KV context, and excess requests queued at 16/32/64 incoming concurrency. Prompt caching and slot similarity were disabled; every response returned 128 raw token IDs. Each marker is the median of two preregistered fresh-server attempts whose relative range was at most 0.96%. Output isolation passed with no cross-base oracle collision, but multi-user greedy token identity is batch-shape-dependent. This is aggregate batch-wall throughput, not queued TTFT or per-request latency. No point is interpolated or extrapolated.",
     "evidence": "experiments/qwen38-27b-b70/data/2026-08-25-qwen38-q8-tp1-http-p8-queue-concurrency-r5-result.json",
     "points": [
      {
       "concurrent_sequences": 1,
       "value": 18.07112270930881,
       "per_user_value": 18.07112270930881,
       "samples": 2
      },
      {
       "concurrent_sequences": 2,
       "value": 29.14791749677608,
       "per_user_value": 14.57395874838804,
       "samples": 2
      },
      {
       "concurrent_sequences": 4,
       "value": 47.517546622837,
       "per_user_value": 11.87938665570925,
       "samples": 2
      },
      {
       "concurrent_sequences": 8,
       "value": 67.07319480808911,
       "per_user_value": 8.38414935101114,
       "samples": 2
      },
      {
       "concurrent_sequences": 16,
       "value": 68.12752786452799,
       "per_user_value": 4.2579704915329994,
       "samples": 2
      },
      {
       "concurrent_sequences": 32,
       "value": 68.31104982515942,
       "per_user_value": 2.1347203070362317,
       "samples": 2
      },
      {
       "concurrent_sequences": 64,
       "value": 68.55554362988714,
       "per_user_value": 1.0711803692169866,
       "samples": 2
      }
     ]
    }
   ],
   "hardware": {
    "accelerator": "Intel Arc Pro B70 32 GiB",
    "cards": 1,
    "host_ram_min_bytes": 16106127360,
    "host_ram_plus_swap_min_bytes": 32212254720,
    "model_weight_bytes": 28595763552
   },
   "model": {
    "repository": "ggml-org/Qwen3.8-27B-GGUF",
    "revision": "0669b98607d47046c7c2b3f801011d54a08cfccf",
    "manifest": "repro/qwen38-27b-q8-tp1-b70/model-direct.json"
   },
   "runtime": {
    "kind": "native",
    "project": "mndodd/llama.cpp",
    "revision": "4302fb59969a5d8cf9f8e5f55fdd4506d0ed2126",
    "build": "repro/qwen38-27b-q8-tp1-b70/restore-and-build.sh"
   },
   "known_limitations": [
    "The 19.619240 tok/s headline is a short-context varied-prompt HTTP result. The separate depth curve remains raw llama-bench pp2048/tg128 and must not be substituted for that workload.",
    "TP1 is deterministic 12/12 across its two fresh servers but differs from the TP2 raw oracle 12/12, with first divergences at generated tokens 59-444. It is a separately quality-gated arithmetic identity, not an exact cross-card-count output claim.",
    "The low-latency launcher is one slot at 8K. The throughput launcher uses eight active slots at 4K total context and queues excess requests; its aggregate curve does not qualify queued TTFT or per-request latency.",
    "Exact 64-slot/32K and 32-slot/16K F16-KV profiles do not fit one B70. A 16-slot/8K profile fits but is slower than queued p8 above eight users.",
    "The tested host had 16 GB nominal RAM plus swap. Clean-host installation and replay remain pending."
   ],
   "project_patches": {
    "required": true,
    "items": [
     "patches/qwen36-27b-q8-tp2-asrock-b70/llama-cpp-mndodd-4302fb599-lab-tp2-dp4a2-20260815.diff.gz.b64",
     "patches/qwen38-27b-q4km-tp2-asrock-b70/llama-cpp-q4k-mmvq-swiglu-tp2-20260815.diff.gz.b64",
     "patches/qwen38-27b-q4km-tp1-b70s/llama-cpp-tp1-gdn-state-io-widen-20260821.diff.gz.b64",
     "patches/qwen38-27b-q4km-tp1-b70s/llama-cpp-tp1-conv-qk-widen-20260821.diff.gz.b64",
     "patches/qwen38-27b-q4km-tp1-b70s/llama-cpp-tp1-qk-norm-rope-src-widen-20260821.diff.gz.b64",
     "patches/qwen38-27b-q4km-tp1-b70s/llama-cpp-tp1-q8out-rejected-memo320-20260821.diff.gz.b64"
    ],
    "reason": "The exact measured binaries were reconstructed from the complete shared Qwen3.8 TP1 source stack; Q4-only paths are retained but inert for Q8 weights."
   },
   "commands": {
    "preflight": "MODEL_DIR=/path/to/qwen3.8-27b-q8 BUILD_DIR=/path/to/llama.cpp/build-sycl-aot-bmg-g31 repro/qwen38-27b-q8-tp1-b70/preflight.sh",
    "launch": "MODEL_DIR=/path/to/qwen3.8-27b-q8 BUILD_DIR=/path/to/llama.cpp/build-sycl-aot-bmg-g31 repro/qwen38-27b-q8-tp1-b70/run-server.sh",
    "health": "curl -fsS http://127.0.0.1:18088/health",
    "benchmark": "MODEL_DIR=/path/to/qwen3.8-27b-q8 BUILD_DIR=/path/to/llama.cpp/build-sycl-aot-bmg-g31 OUT=/path/to/depth.json repro/qwen38-27b-q8-tp1-b70/bench-depth.sh",
    "stop": "Ctrl-C in the foreground server terminal; then pgrep -x llama-server must return no process"
   },
   "dependencies": [
    "repro/qwen38-27b-q8-tp1-b70/README.md",
    "repro/qwen38-27b-autoround-int4-b70/scripts/verify-model-direct.py",
    "repro/qwen38-27b-q8-tp1-b70/model-direct.json",
    "repro/qwen38-27b-q8-tp1-b70/verify-model-direct.sh",
    "repro/qwen38-27b-q8-tp1-b70/restore-and-build.sh",
    "repro/qwen38-27b-q8-tp1-b70/preflight.sh",
    "repro/qwen38-27b-q8-tp1-b70/run-server.sh",
    "repro/qwen38-27b-q8-tp1-b70/run-throughput-server.sh",
    "repro/qwen38-27b-q8-tp1-b70/bench-depth.sh",
    "repro/qwen38-27b-q8-tp1-b70/quality.sh",
    "repro/qwen38-27b-q4km-tp1-b70/restore-and-build.sh",
    "patches/qwen38-27b-q4km-tp1-b70s/README.md",
    "experiments/qwen38-27b-b70/data/qwen38-q8weights-f16-tp1-local-20260825-r2/result.json",
    "experiments/qwen38-27b-b70/notes/2026-08-25-qwen38-q8weights-f16-tp1-local-r2-result.md",
    "experiments/qwen38-27b-b70/data/qwen38-q8weights-f16-tp1-service-quality-20260825-r1/qualification.json",
    "experiments/qwen38-27b-b70/notes/2026-08-25-qwen38-q8weights-f16-tp1-service-quality-r1-result.md",
    "experiments/qwen38-27b-b70/data/2026-08-25-qwen38-q8-tp1-http-p8-queue-concurrency-r5-result.json",
    "experiments/qwen38-27b-b70/data/2026-08-25-qwen38-q8-tp1-package-throughput-launch-smoke.json",
    "experiments/qwen38-27b-b70/data/2026-08-25-qwen38-q8-tp1-http-p8-queue-concurrency-r5-prereg.json",
    "experiments/qwen38-27b-b70/data/2026-08-25-qwen38-q8-tp1-http-p8-queue-oracle-digests.json",
    "experiments/qwen38-27b-b70/notes/2026-08-25-qwen38-q8-tp1-http-p16-qualified-and-p8-queue-prereg.md",
    "experiments/qwen38-27b-b70/scripts/run-qwen38-q8-tp1-http-p8-queue-concurrency-r5.sh",
    "experiments/qwen38-27b-b70/data/2026-08-27-qwen38-q8-tp1-strict-reasoningoff-native-r1-result.json",
    "experiments/qwen38-27b-b70/data/2026-08-27-qwen38-q8-tp1-strict-reasoningoff-native-r1-comparison.json",
    "experiments/qwen38-27b-b70/data/qwen38-q8-tp1-strict-reasoningoff-native-20260827-r1a/performance.json",
    "experiments/qwen38-27b-b70/data/qwen38-q8-tp1-strict-reasoningoff-native-20260827-r1a/canaries.json",
    "experiments/qwen38-27b-b70/data/qwen38-q8-tp1-strict-reasoningoff-native-20260827-r1b/performance.json",
    "experiments/qwen38-27b-b70/data/qwen38-q8-tp1-strict-reasoningoff-native-20260827-r1b/canaries.json",
    "experiments/qwen38-27b-b70/scripts/run-20260827-qwen38-q8-tp1-strict-attempt.sh",
    "community/mndodd-qwen36-27b-llamacpp-sycl/STATUS.md",
    "patches/qwen36-27b-q8-tp2-asrock-b70/llama-cpp-mndodd-4302fb599-lab-tp2-dp4a2-20260815.diff.gz.b64",
    "patches/qwen38-27b-q4km-tp2-asrock-b70/llama-cpp-q4k-mmvq-swiglu-tp2-20260815.diff.gz.b64",
    "patches/qwen38-27b-q4km-tp1-b70s/llama-cpp-tp1-gdn-state-io-widen-20260821.diff.gz.b64",
    "patches/qwen38-27b-q4km-tp1-b70s/llama-cpp-tp1-conv-qk-widen-20260821.diff.gz.b64",
    "patches/qwen38-27b-q4km-tp1-b70s/llama-cpp-tp1-qk-norm-rope-src-widen-20260821.diff.gz.b64",
    "patches/qwen38-27b-q4km-tp1-b70s/llama-cpp-tp1-q8out-rejected-memo320-20260821.diff.gz.b64"
   ],
   "missing": [
    "tested clean-host Intel driver and oneAPI installation",
    "clean-host source build and endpoint replay",
    "beginner recovery flow",
    "realistic-prompt HTTP speed and TTFT context sweep beyond the qualified short-context headline",
    "queued TTFT and per-request latency profile"
   ]
  },
  {
   "manifest": "packages/qwen38-27b-q8-tp2-b70/package.json",
   "format": "b70-model-package-v1",
   "id": "qwen38-27b-q8-tp2-asrock-b70",
   "name": "Qwen3.8 27B Q8_0 on two Intel Arc Pro B70 cards",
   "status": "candidate",
   "audience": "expert",
   "guide": "repro/qwen38-27b-q8-tp2-asrock-b70/README.md",
   "clean_host_tested": false,
   "library": {
    "model_family": "Qwen3.8",
    "publisher": "Alibaba / Qwen",
    "variant": "27B Q8_0 target-only",
    "summary": "Alibaba's Qwen3.8 27B on two Arc Pro B70 cards in 8-bit form with llama.cpp - the quality-conservative choice when 4-bit is not enough.",
    "quantization": "Q8_0 / F16 KV",
    "runtime_label": "llama.cpp SYCL",
    "operating_systems": [
     "Linux"
    ],
    "delivery": [
     "native"
    ],
    "modalities": [
     "text"
    ],
    "use_cases": [
     "general",
     "coding",
     "long context",
     "quality conservative"
    ],
    "tags": [
     "two cards",
     "TP2",
     "target only",
     "no speculation",
     "quality gated"
    ],
    "published_at": "2026-08-27",
    "featured_metric": {
     "value": 36.726447009509016,
     "unit": "tok/s",
     "label": "strict varied-prompt decode",
     "scope": "Median of two fresh-server class-balanced medians over the full 12-prompt/six-class, 512-cap, cache-zero raw-completion suite; target-only TP2, MTP0, packaged --reasoning off launcher, 12/12 complete token arrays exact and all objective canaries passed.",
     "evidence": "experiments/qwen38-27b-b70/data/2026-08-27-qwen38-q8-tp2-strict-reasoningoff-native-r2-result.json"
    }
   },
   "contributors": [
    {
     "id": "b70-optimization-lab",
     "name": "neural.download lab",
     "kind": "lab",
     "profile": "https://github.com/steveseguin/b70-optimization-lab",
     "contribution": "Qwen3.8 transfer, full B70 TP2 patch stack, DP4A2 by SG24 optimization, semantic and repeat gates, benchmarking, and packaging.",
     "status": "integrated",
     "validated_effect": "The accepted Qwen3.8 optimized lane measured 36.772932 versus 31.353431 tok/s for its matched runtime-base control (+17.285%), with 12/12 exact output hashes and cache-zero gates.",
     "evidence": "experiments/qwen38-27b-b70/data/2026-08-15-q8-tp2-transfer-summary.json"
    },
    {
     "id": "mndodd",
     "name": "mndodd",
     "kind": "external",
     "profile": "https://github.com/mndodd",
     "contribution": "Provided the optimized Intel SYCL llama.cpp fork used as the runtime base for the lab's Qwen Q8 TP2 work.",
     "status": "integrated",
     "validated_effect": "A separate matched Qwen3.6 Q8 TP2 A/B measured 31.338765 versus 29.610651 tok/s (+5.836%) for that base; this does not assign the later Qwen3.8 result to the fork.",
     "evidence": "community/mndodd-qwen36-27b-llamacpp-sycl/STATUS.md"
    }
   ],
   "performance_profiles": [
    {
     "id": "http-decode-vs-active-context",
     "label": "Qualified Q8_0 TP2 HTTP decode over exact active context",
     "metric": "decode",
     "unit": "tok/s",
     "x_label": "Exact active prompt tokens",
     "scope": "One-slot native HTTP completions on the exact Q8_0 TP2 stack, two B70s, equal tensor split, F16 KV, 33,024 configured context, cache disabled, no truncation or context shift, and 128 returned token IDs. The fixture is grade C repeated-token shape evidence, not natural prose. Every marker is measured; no point is interpolated or extrapolated.",
     "evidence": "experiments/qwen38-27b-b70/data/qwen38-q8-tp2-http-depth-prefill-20260825-r3-attempt1/summary.json",
     "points": [
      {
       "context_tokens": 2048,
       "value": 36.79674030798057,
       "samples": 1
      },
      {
       "context_tokens": 4096,
       "value": 36.553471815162666,
       "samples": 1
      },
      {
       "context_tokens": 8192,
       "value": 36.01395813311342,
       "samples": 1
      },
      {
       "context_tokens": 16384,
       "value": 35.085401486482745,
       "samples": 1
      },
      {
       "context_tokens": 24576,
       "value": 34.518837744162,
       "samples": 1
      },
      {
       "context_tokens": 32768,
       "value": 33.848820185540816,
       "samples": 1
      }
     ]
    },
    {
     "id": "http-prefill-vs-active-context",
     "label": "Q8_0 TP2 server prompt evaluation",
     "metric": "prefill",
     "unit": "tok/s",
     "x_label": "Exact submitted prompt tokens",
     "scope": "llama-server prompt-evaluation counters from the same fail-closed exact-depth replay. Exactly one timing row was required for every registered prompt depth. This is the repeated-token grade-C fixture, not natural prose or a derived TTFT estimate. Every marker is measured; no point is interpolated or extrapolated.",
     "evidence": "experiments/qwen38-27b-b70/data/qwen38-q8-tp2-http-depth-prefill-20260825-r3-attempt1/summary.json",
     "points": [
      {
       "context_tokens": 2048,
       "value": 1032.33,
       "samples": 1
      },
      {
       "context_tokens": 4096,
       "value": 1038.62,
       "samples": 1
      },
      {
       "context_tokens": 8192,
       "value": 1019.69,
       "samples": 1
      },
      {
       "context_tokens": 16384,
       "value": 983.52,
       "samples": 1
      },
      {
       "context_tokens": 24576,
       "value": 947.33,
       "samples": 1
      },
      {
       "context_tokens": 32768,
       "value": 915.09,
       "samples": 1
      }
     ]
    },
    {
     "id": "http-ttft-vs-active-context",
     "label": "Qualified Q8_0 TP2 HTTP TTFT over exact active context",
     "metric": "ttft",
     "unit": "ms",
     "x_label": "Exact active prompt tokens",
     "scope": "TTFT from the same one-slot exact-token HTTP receipts as the decode and prompt-evaluation curves. Cache was zero and prompt counts were exact at every marker. The fixture is grade C repeated-token shape evidence and is not a natural-prose latency claim. Every marker is measured; no point is interpolated or extrapolated.",
     "evidence": "experiments/qwen38-27b-b70/data/qwen38-q8-tp2-http-depth-prefill-20260825-r3-attempt1/summary.json",
     "points": [
      {
       "context_tokens": 2048,
       "value": 2003.0714339809492,
       "samples": 1
      },
      {
       "context_tokens": 4096,
       "value": 3957.1460890583694,
       "samples": 1
      },
      {
       "context_tokens": 8192,
       "value": 8047.487462987192,
       "samples": 1
      },
      {
       "context_tokens": 16384,
       "value": 16681.124167051166,
       "samples": 1
      },
      {
       "context_tokens": 24576,
       "value": 25957.882561022416,
       "samples": 1
      },
      {
       "context_tokens": 32768,
       "value": 35832.39820506424,
       "samples": 1
      }
     ]
    },
    {
     "id": "http-output-audited-aggregate-vs-concurrent-users",
     "label": "Output-audited Q8_0 TP2 HTTP aggregate decode",
     "metric": "aggregate_decode",
     "unit": "tok/s",
     "x_metric": "concurrent_sequences",
     "x_label": "Concurrent HTTP users",
     "scope": "llama.cpp native HTTP /completion on two B70s with 64 active slots, 32K total F16 KV context, prompt cache and slot similarity disabled, unique short prompts, and 128 returned raw token IDs per request. Each marker is the median of two preregistered fresh-server attempts whose relative range was at most 1.46%. Output isolation passed with no cross-base oracle collision, but multi-user greedy token identity is batch-shape-dependent. The directly reproduced c8-to-c16 throughput dip is retained rather than smoothed. No point is interpolated or extrapolated.",
     "evidence": "experiments/qwen38-27b-b70/data/2026-08-25-qwen38-q8-tp2-http-concurrency-r2-result.json",
     "points": [
      {
       "concurrent_sequences": 1,
       "value": 32.48673907107751,
       "per_user_value": 32.48673907107751,
       "samples": 2
      },
      {
       "concurrent_sequences": 2,
       "value": 51.60144040951495,
       "per_user_value": 25.800720204757475,
       "samples": 2
      },
      {
       "concurrent_sequences": 4,
       "value": 85.59522372041042,
       "per_user_value": 21.398805930102604,
       "samples": 2
      },
      {
       "concurrent_sequences": 8,
       "value": 125.41437907811792,
       "per_user_value": 15.67679738476474,
       "samples": 2
      },
      {
       "concurrent_sequences": 16,
       "value": 84.3286328753673,
       "per_user_value": 5.2705395547104565,
       "samples": 2
      },
      {
       "concurrent_sequences": 32,
       "value": 125.83227328244796,
       "per_user_value": 3.9322585400764987,
       "samples": 2
      },
      {
       "concurrent_sequences": 64,
       "value": 163.64420555060713,
       "per_user_value": 2.5569407117282363,
       "samples": 2
      }
     ]
    }
   ],
   "hardware": {
    "accelerator": "Intel Arc Pro B70 32 GiB",
    "cards": 2,
    "model_weight_bytes": 28595763552
   },
   "model": {
    "repository": "ggml-org/Qwen3.8-27B-GGUF",
    "revision": "0669b98607d47046c7c2b3f801011d54a08cfccf",
    "manifest": "experiments/qwen38-27b-b70/data/2026-08-15-q8-tp2-transfer-summary.json"
   },
   "runtime": {
    "kind": "native",
    "repository": "mndodd/llama.cpp",
    "revision": "4302fb59969a5d8cf9f8e5f55fdd4506d0ed2126",
    "build": "patches/qwen38-27b-q8-tp2-asrock-b70/README.md"
   },
   "project_patches": {
    "required": true,
    "items": [
     "patches/qwen36-27b-q8-tp2-asrock-b70/llama-cpp-mndodd-4302fb599-lab-tp2-dp4a2-20260815.diff.gz.b64",
     "patches/qwen38-27b-q8-tp2-asrock-b70/recurrent-quad-sg16-20260817.diff",
     "patches/qwen38-27b-q8-tp2-asrock-b70/recurrent-quad-sg24-20260817.diff"
    ],
    "reason": "The service and record depend on the lab's complete TP2/DP4A2 patch plus the Qwen3.8 recurrent-quad SG16 and SG24 increments."
   },
   "commands": {
    "preflight": "repro/qwen38-27b-q8-tp2-asrock-b70/verify-artifacts.sh && repro/qwen38-27b-q8-tp2-asrock-b70/verify-model-direct.sh /path/to/model-directory",
    "launch": "QWEN38_SOURCE_DIR=/path/to/source QWEN38_BUILD_DIR=/path/to/build QWEN38_MODEL=/path/to/Qwen3.8-27B-Q8_0.gguf repro/qwen38-27b-q8-tp2-asrock-b70/run-server.sh",
    "health": "curl -fsS http://127.0.0.1:18088/health",
    "benchmark": "OUT=/path/to/result.json repro/qwen38-27b-q8-tp2-asrock-b70/bench.sh",
    "stop": "Press Ctrl-C once in the foreground launcher and wait; then require pgrep -x llama-server to return no process. Do not signal both the systemd scope and child."
   },
   "dependencies": [
    "repro/qwen38-27b-q8-tp2-asrock-b70/README.md",
    "repro/gemma4-26b-a4b-q8-b70/realistic-suite-v1.json",
    "repro/qwen36-27b-autoround-int4-b70/realistic-suite-v1.json",
    "repro/qwen38-27b-autoround-int4-b70/scripts/verify-model-direct.py",
    "scripts/bench-openai-realistic-suite.py",
    "repro/qwen38-27b-q8-tp2-asrock-b70/config.env",
    "repro/qwen38-27b-q8-tp2-asrock-b70/runtime-common.sh",
    "repro/qwen38-27b-q8-tp2-asrock-b70/run-server.sh",
    "repro/qwen38-27b-q8-tp2-asrock-b70/run-throughput-server.sh",
    "repro/qwen38-27b-q8-tp2-asrock-b70/bench.sh",
    "repro/qwen38-27b-q8-tp2-asrock-b70/verify-artifacts.sh",
    "repro/qwen38-27b-q8-tp2-asrock-b70/verify-model-direct.sh",
    "repro/qwen38-27b-q8-tp1-b70/model-direct.json",
    "patches/qwen38-27b-q8-tp2-asrock-b70/README.md",
    "experiments/qwen38-27b-b70/data/2026-08-15-q8-tp2-transfer-summary.json",
    "experiments/qwen38-27b-b70/data/2026-08-17-q8-dp4a2-sg24-accepted.json",
    "experiments/qwen38-27b-b70/data/2026-08-25-qwen38-q8-tp2-http-concurrency-r2-result.json",
    "experiments/qwen38-27b-b70/data/2026-08-25-qwen38-q8-tp2-http-concurrency-r2-prereg.json",
    "experiments/qwen38-27b-b70/data/2026-08-25-qwen38-q8-tp2-http-concurrency-oracle-digests.json",
    "experiments/qwen38-27b-b70/scripts/run-qwen38-q8-tp2-http-concurrency-r2.sh",
    "experiments/qwen38-27b-b70/data/qwen38-q8-tp2-http-depth-prefill-20260825-r3-attempt1/summary.json",
    "experiments/qwen38-27b-b70/data/2026-08-25-qwen38-q8-tp2-http-depth-prefill-r3-prereg.json",
    "experiments/qwen38-27b-b70/scripts/run-qwen38-q8-tp2-http-depth-prefill-r3.sh",
    "experiments/qwen38-27b-b70/data/2026-08-25-qwen38-q8-tp2-package-throughput-launch-smoke.json",
    "experiments/qwen38-27b-b70/data/2026-08-27-qwen38-q8-tp2-strict-reasoningoff-native-r2-result.json",
    "experiments/qwen38-27b-b70/data/2026-08-27-qwen38-q8-tp2-strict-reasoningoff-native-r2-comparison.json",
    "experiments/qwen38-27b-b70/data/qwen38-q8-tp2-strict-reasoningoff-native-20260827-r2a/performance.json",
    "experiments/qwen38-27b-b70/data/qwen38-q8-tp2-strict-reasoningoff-native-20260827-r2a/canaries.json",
    "experiments/qwen38-27b-b70/data/qwen38-q8-tp2-strict-reasoningoff-native-20260827-r2b/performance.json",
    "experiments/qwen38-27b-b70/data/qwen38-q8-tp2-strict-reasoningoff-native-20260827-r2b/canaries.json",
    "community/mndodd-qwen36-27b-llamacpp-sycl/STATUS.md",
    "patches/qwen36-27b-q8-tp2-asrock-b70/llama-cpp-mndodd-4302fb599-lab-tp2-dp4a2-20260815.diff.gz.b64",
    "patches/qwen38-27b-q8-tp2-asrock-b70/recurrent-quad-sg16-20260817.diff",
    "patches/qwen38-27b-q8-tp2-asrock-b70/recurrent-quad-sg24-20260817.diff"
   ],
   "known_limitations": [
    "The 36.726447 tok/s headline uses raw untemplated completions on the packaged --reasoning off launcher. Those outputs match the historical raw-completions oracle 12/12, but the result must not be relabeled as chat-template service throughput.",
    "Multi-user greedy token identity is batch-shape-dependent. The qualified curve proves complete isolated responses with zero cross-base oracle collisions, not sequential byte identity.",
    "The c8-to-c16 aggregate drop reproduced on both fresh servers. It is retained as measured and is not interpolated away.",
    "The concurrency profile measures aggregate batch-wall throughput, not per-request TTFT or latency under queueing."
   ],
   "missing": [
    "tested platform installation",
    "clean-host replay",
    "beginner recovery flow",
    "natural-prompt HTTP speed and TTFT context sweep beyond the qualified short-context headline",
    "queued TTFT and per-request latency profile"
   ]
  },
  {
   "manifest": "packages/qwen38-flash-next-fp8-tp4-mtp0-w13n64-b70-34tps-20260908/package.json",
   "format": "b70-model-package-v1",
   "id": "qwen38-flash-next-fp8-tp4-mtp0-w13n64-b70-34tps-20260908",
   "name": "Qwen3.8 Flash-Next FP8 with no speculation, never-routed experts host-placed, both reference Triton kernels restored and the W13 MoE tile at the base width, on four Intel Arc Pro B70 cards",
   "status": "candidate",
   "audience": "expert",
   "guide": "repro/qwen38-flash-next-fp8-tp4-mtp0-w13n64-b70-34tps-20260908/README.md",
   "clean_host_tested": false,
   "library": {
    "model_family": "Qwen3.8 Flash-Next",
    "publisher": "Qwen",
    "variant": "official FP8 export, TP4/EP4, deterministic full-decode graph, no speculation, never-routed experts host-placed, hyper-connection glue and QSA pre-indexer restored to the reference kernels, W13 per-phase MoE tile set to the base width",
    "summary": "Qwen's 125B-A6B hybrid-attention MoE, served from its official FP8 weights across four Arc Pro B70 cards, with no speculative decoding. The n-gram table and embeddings live in host memory, every hot expert stays on the cards, the experts a routing census never selects are parked in host memory behind a per-expert table in the MoE kernel, and the two Triton kernels the XPU port had replaced are restored to the model's own reference implementations. The record over the previous one is a removal: a per-phase MoE tile that had been adopted bundled with an unrelated change, and never isolated, cost 2.1% once it was measured alone. One line of a tuned configuration file, no source change, outputs bit-identical across three servers.",
    "quantization": "FP8 block-128 weights / BF16 KV",
    "runtime_label": "vLLM XPU",
    "operating_systems": [
     "Linux"
    ],
    "delivery": [
     "native"
    ],
    "modalities": [
     "text"
    ],
    "use_cases": [
     "general",
     "coding",
     "deterministic serving research"
    ],
    "tags": [
     "four cards",
     "TP4",
     "EP4",
     "XPU graph",
     "UVA offload",
     "expert host placement",
     "lab replay",
     "originating host",
     "Triton HC glue",
     "new output authority",
     "fused QSA pre-indexer",
     "reference kernels restored",
     "MTP0",
     "no speculation",
     "W13-N64"
    ],
    "published_at": "2026-09-08",
    "featured_metric": {
     "value": 34.495291913070886,
     "unit": "tok/s",
     "label": "class-balanced decode median",
     "scope": "median of prompt-class medians, 99 inter-token intervals after TTFT, fixed realistic suite run once cold; A326 fresh-server repeat, 2026-09-08.",
     "evidence": "experiments/qwen38-flash-next-fp8-b70/data/20260908-tp4-mtp0-a326-w13n64-fresh-realistic-suite-v1-result.json"
    }
   },
   "contributors": [
    {
     "id": "b70-optimization-lab",
     "name": "neural.download lab",
     "kind": "lab",
     "profile": "https://github.com/steveseguin/b70-optimization-lab",
     "contribution": "Flash-Next XPU bring-up, deterministic full-decode graph, lossless MTP1 verification, the VRAM-headroom root cause, the per-expert host placement (offset-table Triton MoE kernel, load-time placement, tuned-map key on the logical expert count), the Triton hyper-connection glue and the fused QSA pre-indexer on XPU (the torch-fallback root cause of the 8.4 ms hyper-connection mixes), new-authority certification across five servers, frozen-packet certification, and record packaging.",
     "status": "integrated",
     "validated_effect": "34.495292 tok/s class-balanced on the fresh-server A326 repeat; A325 measured 34.510128 tok/s. No speculation; W13-N64 preserves the superseded MTP0 output stream across three servers at exact 2K and 4K.",
     "evidence": "experiments/qwen38-flash-next-fp8-b70/data/20260908-tp4-mtp0-a326-w13n64-promotion-attestation.json"
    }
   ],
   "hardware": {
    "accelerator": "Intel Arc Pro B70 32 GiB",
    "cards": 4
   },
   "model": {
    "repository": "Qwen/Qwen3.8-Flash-Next-FP8",
    "revision": "bcd9f01ddc9cff2316eb84281bebcd5b058bddce",
    "manifest": "repro/qwen38-flash-next-fp8-tp4-mtp3-b70/model-contract.json"
   },
   "runtime": {
    "kind": "native",
    "repository": "vllm-project/vllm",
    "revision": "2a372e860e273273357cb7437ac7de1694304f9f",
    "build": "repro/qwen38-flash-next-fp8-tp4-mtp0-w13n64-b70-34tps-20260908/README.md"
   },
   "project_patches": {
    "required": true,
    "items": [
     "patches/qwen38-flash-next-fp8-b70/vllm-placement-mtp1-005dc578/vllm-q38-placement-mtp1-005dc578-20260906.bundle",
     "patches/qwen38-flash-next-fp8-b70/vllm-placement-mtp1-005dc578/series.sha256",
     "patches/qwen38-flash-next-fp8-b70/vllm-lossless-mtp1-1b2a17c1/vllm-q38-lossless-mtp1-1b2a17c1-20260906.bundle",
     "patches/qwen38-flash-next-fp8-b70/vllm-lossless-mtp1-1b2a17c1/series.sha256",
     "patches/qwen38-flash-next-fp8-b70/oneccl-4ceafd1-b70-public/lib.sha256",
     "repro/qwen38-flash-next-fp8-tp4-mtp3-b70/runtime-stage.sha256",
     "patches/qwen38-flash-next-fp8-b70/vllm-hctriton-mtp1-62219122/vllm-q38-hctriton-mtp1-62219122-20260907.bundle",
     "patches/qwen38-flash-next-fp8-b70/vllm-hctriton-mtp1-62219122/series.sha256",
     "patches/qwen38-flash-next-fp8-b70/vllm-qsafused-mtp1-6d872457/vllm-q38-qsafused-mtp1-6d872457-20260907.bundle",
     "patches/qwen38-flash-next-fp8-b70/vllm-qsafused-mtp1-6d872457/series.sha256"
    ],
    "reason": "The record loads the lab's nine placement commits over its 55-commit lossless-MTP1 overlay on public vLLM 76cfe1cd, the hosted 2f829747 kernel stage, and the hosted public oneCCL 4ceafd1 build; all are pinned by bytes, and the never-routed expert placement file is pinned by hash."
   },
   "commands": {
    "preflight": "Follow the Run it section of repro/qwen38-flash-next-fp8-tp4-mtp0-w13n64-b70-34tps-20260908/README.md and verify identity.json, frozen-a325-a326-packets.sha256, and verifier-and-map-pin.txt before replay.",
    "launch": "Follow the native sibling restore-source procedure linked from the guide, with overlay 2a372e860e273273357cb7437ac7de1694304f9f, MTP=0, and configs/moe-m1-w13-n64. A standalone clean-host launcher is not certified.",
    "health": "curl -fsS http://127.0.0.1:19900/health",
    "benchmark": "Replay the frozen A325/A326 cold 12-prompt realistic suite once; compare against the package identity and A326 promotion attestation. Do not run the sibling MTP1 record gate for this package.",
    "stop": "Use the frozen MTP0 packet supervisor teardown and verify that its workers and listener are gone before another attempt."
   },
   "dependencies": [
    "data/localmaxxing-responses/qwen38-flash-next-fp8-tp4-mtp0-qsafused-realistic-20260907.json",
    "data/localmaxxing-responses/qwen38-flash-next-fp8-tp4-mtp1-qsafused-realistic-20260907.json",
    "experiments/qwen38-flash-next-fp8-b70/configs/moe-m1-w13-n32",
    "experiments/qwen38-flash-next-fp8-b70/data/20260906-q38-expert-host-placement-3p5gib-per-rank.json",
    "experiments/qwen38-flash-next-fp8-b70/data/20260907-tp4-mtp1-a305-fresh-repeat-deterministic-summary.json",
    "experiments/qwen38-flash-next-fp8-b70/data/20260907-tp4-mtp1-a306-promotion-attestation.json",
    "experiments/qwen38-flash-next-fp8-b70/data/20260907-tp4-mtp1-a306-qsafused-realistic-suite-v1-result.json",
    "experiments/qwen38-flash-next-fp8-b70/data/20260907-tp4-mtp1-a307-record-gate-replay-realistic-suite-v1-result.json",
    "experiments/qwen38-flash-next-fp8-b70/data/20260907-tp4-mtp1-qsafused-exact-2k-pair-summary.json",
    "experiments/qwen38-flash-next-fp8-b70/tools/launch-tp4-ep4-eager-mtp0-long-context-base.sh",
    "experiments/qwen38-flash-next-fp8-b70/tools/launch-tp4-mtp1-4352-ple-only-a306-fullgraphdet-w13n32.sh",
    "experiments/qwen38-flash-next-fp8-b70/tools/q38-launch-frozen-attempt.sh",
    "experiments/qwen38-flash-next-fp8-b70/tools/run-q38-a306-host-controlled.sh",
    "experiments/qwen38-flash-next-fp8-b70/tools/run-tp4-mtp1-4352-ple-only-a306-fullgraphdet-w13n32-client.sh",
    "experiments/qwen38-flash-next-fp8-b70/tools/supervise-tp4-mtp1-4352-ple-only-a306-fullgraphdet-w13n32.sh",
    "experiments/qwen38-flash-next-fp8-b70/tools/verify-moe-m1-w13-n32-selection.py",
    "experiments/qwen38-flash-next-fp8-b70/tools/verify-moe-m1-w13-n64-selection.py",
    "patches/qwen38-flash-next-fp8-b70/oneccl-4ceafd1-b70-public/README.md",
    "patches/qwen38-flash-next-fp8-b70/oneccl-4ceafd1-b70-public/lib.sha256",
    "patches/qwen38-flash-next-fp8-b70/vllm-hctriton-mtp1-62219122/README.md",
    "patches/qwen38-flash-next-fp8-b70/vllm-hctriton-mtp1-62219122/series.sha256",
    "patches/qwen38-flash-next-fp8-b70/vllm-hctriton-mtp1-62219122/verify-series.sh",
    "patches/qwen38-flash-next-fp8-b70/vllm-hctriton-mtp1-62219122/vllm-q38-hctriton-mtp1-62219122-20260907.bundle",
    "patches/qwen38-flash-next-fp8-b70/vllm-lossless-mtp1-1b2a17c1/README.md",
    "patches/qwen38-flash-next-fp8-b70/vllm-lossless-mtp1-1b2a17c1/series.sha256",
    "patches/qwen38-flash-next-fp8-b70/vllm-lossless-mtp1-1b2a17c1/verify-series.sh",
    "patches/qwen38-flash-next-fp8-b70/vllm-lossless-mtp1-1b2a17c1/vllm-q38-lossless-mtp1-1b2a17c1-20260906.bundle",
    "patches/qwen38-flash-next-fp8-b70/vllm-placement-mtp1-005dc578/README.md",
    "patches/qwen38-flash-next-fp8-b70/vllm-placement-mtp1-005dc578/series.sha256",
    "patches/qwen38-flash-next-fp8-b70/vllm-placement-mtp1-005dc578/verify-series.sh",
    "patches/qwen38-flash-next-fp8-b70/vllm-placement-mtp1-005dc578/vllm-q38-placement-mtp1-005dc578-20260906.bundle",
    "patches/qwen38-flash-next-fp8-b70/vllm-qsafused-mtp1-6d872457/README.md",
    "patches/qwen38-flash-next-fp8-b70/vllm-qsafused-mtp1-6d872457/series.sha256",
    "patches/qwen38-flash-next-fp8-b70/vllm-qsafused-mtp1-6d872457/verify-series.sh",
    "patches/qwen38-flash-next-fp8-b70/vllm-qsafused-mtp1-6d872457/vllm-q38-qsafused-mtp1-6d872457-20260907.bundle",
    "repro/qwen38-flash-next-fp8-tp4-mtp0-w13n64-b70-34tps-20260908/CONTAINER-STATUS.md",
    "repro/qwen38-flash-next-fp8-tp4-mtp0-w13n64-b70-34tps-20260908/README.md",
    "repro/qwen38-flash-next-fp8-tp4-mtp0-w13n64-b70-34tps-20260908/evidence/record-evidence.sha256",
    "repro/qwen38-flash-next-fp8-tp4-mtp0-w13n64-b70-34tps-20260908/frozen-a325-a326-packets.sha256",
    "repro/qwen38-flash-next-fp8-tp4-mtp0-w13n64-b70-34tps-20260908/identity.json",
    "repro/qwen38-flash-next-fp8-tp4-mtp0-w13n64-b70-34tps-20260908/verifier-and-map-pin.txt",
    "repro/qwen38-flash-next-fp8-tp4-mtp1-qsafused-b70-38tps-20260907/check-replay-result.py",
    "repro/qwen38-flash-next-fp8-tp4-mtp1-qsafused-b70-38tps-20260907/frozen-a306-packet.sha256",
    "repro/qwen38-flash-next-fp8-tp4-mtp1-qsafused-b70-38tps-20260907/make-replay-attempt.py",
    "repro/qwen38-flash-next-fp8-tp4-mtp1-qsafused-b70-38tps-20260907/run-record-gate.sh",
    "repro/qwen38-flash-next-fp8-tp4-mtp1-qsafused-b70-38tps-20260907/verifier-pin.txt",
    "repro/qwen38-flash-next-fp8-tp4-mtp1-qsafused-b70-38tps-20260907/verify-identity.sh",
    "repro/qwen38-flash-next-fp8-tp4-mtp1-qsafused-b70-38tps-20260907/wait-and-run-client.sh",
    "repro/qwen38-flash-next-fp8-tp4-mtp3-b70/RELEASE-NOTES.md",
    "repro/qwen38-flash-next-fp8-tp4-mtp3-b70/model-contract.json",
    "repro/qwen38-flash-next-fp8-tp4-mtp3-b70/pip-freeze-observed.txt",
    "repro/qwen38-flash-next-fp8-tp4-mtp3-b70/prepare-runtime.py",
    "repro/qwen38-flash-next-fp8-tp4-mtp3-b70/runtime-stage.sha256",
    "repro/qwen38-flash-next-fp8-tp4-mtp3-b70/verify-model.py",
    "repro/rapid-model-snapshots-b70/realistic-suite-v1.json",
    "results/qwen38-flash-next-fp8-b70/README.md",
    "scripts/bench-openai-concurrency.py",
    "scripts/bench-openai-realistic-suite.py",
    "scripts/bench-openai-token-depth-suite.py",
    "experiments/qwen38-flash-next-fp8-b70/data/20260908-tp4-mtp0-a326-w13n64-promotion-attestation.json",
    "experiments/qwen38-flash-next-fp8-b70/data/20260908-tp4-mtp0-a326-w13n64-fresh-realistic-suite-v1-result.json",
    "experiments/qwen38-flash-next-fp8-b70/configs/moe-m1-w13-n64"
   ],
   "missing": [
    "built and replayed container image",
    "clean-host beginner install path (the recovery flow now in the guide covers failures during a run, not a first install)",
    "container recipe (not published for this line; see CONTAINER-STATUS.md for the shared torch-ABI blocker)",
    "decode, prefill, and TTFT context sweep",
    "installable dependency hash lock",
    "non-originating-host replay",
    "record-gate replay scripts (this guide pins the packets and evidence; the MTP1 sibling carries the replay helpers)",
    "record-specific model acquisition helper",
    "tested platform installation (venv, oneAPI runtime, xe driver)"
   ]
  },
  {
   "manifest": "packages/qwen38-flash-next-fp8-tp4-mtp1-hctriton-b70-37tps-20260907/package.json",
   "format": "b70-model-package-v1",
   "id": "qwen38-flash-next-fp8-tp4-mtp1-hctriton-b70-37tps-20260907",
   "name": "Qwen3.8 Flash-Next FP8 with lossless MTP1, never-routed experts host-placed and the Triton hyper-connection glue on four Intel Arc Pro B70 cards",
   "status": "candidate",
   "audience": "expert",
   "guide": "repro/qwen38-flash-next-fp8-tp4-mtp1-hctriton-b70-37tps-20260907/README.md",
   "clean_host_tested": false,
   "library": {
    "model_family": "Qwen3.8 Flash-Next",
    "publisher": "Qwen",
    "variant": "official FP8 export, TP4/EP4, deterministic full-decode graph, lossless MTP1, never-routed experts host-placed, Triton HC glue on XPU",
    "summary": "Qwen's 125B-A6B hybrid-attention MoE, served from its official FP8 weights across four Arc Pro B70 cards. The n-gram table and embeddings live in host memory, every hot expert stays on the cards, the experts a routing census never selects are parked in host memory behind a per-expert table in the MoE kernel, and the model's own Triton hyper-connection glue kernels run on XPU instead of torch fallbacks. One speculative token per step with every output identical to the no-speculation line. The outputs are a new deterministic authority (the Triton glue rounds at the last bf16 bit), reproduced across five servers with the quality profile of the certified rows. A replay of the lab's certified run.",
    "quantization": "FP8 block-128 weights / BF16 KV",
    "runtime_label": "vLLM XPU",
    "operating_systems": [
     "Linux"
    ],
    "delivery": [
     "native"
    ],
    "modalities": [
     "text"
    ],
    "use_cases": [
     "general",
     "coding",
     "deterministic serving research"
    ],
    "tags": [
     "four cards",
     "TP4",
     "EP4",
     "MTP1",
     "XPU graph",
     "UVA offload",
     "expert host placement",
     "lab replay",
     "originating host",
     "lossless speculation",
     "Triton HC glue",
     "new output authority"
    ],
    "published_at": "2026-09-07",
    "featured_metric": {
     "value": 37.04584447577281,
     "unit": "tok/s",
     "label": "class-balanced decode median",
     "scope": "Median of prompt-class medians over 99 inter-token intervals after TTFT on the fixed cold 12-prompt realistic suite, sent once (A272, 2026-09-07).",
     "evidence": "experiments/qwen38-flash-next-fp8-b70/data/20260907-tp4-mtp1-a272-realistic-suite-v1-result.json"
    }
   },
   "contributors": [
    {
     "id": "b70-optimization-lab",
     "name": "neural.download lab",
     "kind": "lab",
     "profile": "https://github.com/steveseguin/b70-optimization-lab",
     "contribution": "Flash-Next XPU bring-up, deterministic full-decode graph, lossless MTP1 verification, the VRAM-headroom root cause, the per-expert host placement (offset-table Triton MoE kernel, load-time placement, tuned-map key on the logical expert count), the Triton hyper-connection glue on XPU (the torch-fallback root cause of the 8.4 ms hyper-connection mixes), new-authority certification across five servers, frozen-packet certification, and record packaging.",
     "status": "integrated",
     "validated_effect": "37.045844 tok/s class-balanced (A272) against 31.929484 for the placement line, with the lossless MTP1 pins equal to the Triton-HC MTP0 line's on every server; exact-2K 36.4 and exact-4K 36.4 tok/s on a separate server (A271); outputs are a new authority, not bit-identical to the torch-fallback rows.",
     "evidence": "experiments/qwen38-flash-next-fp8-b70/data/20260907-tp4-mtp1-a272-promotion-attestation.json"
    }
   ],
   "hardware": {
    "accelerator": "Intel Arc Pro B70 32 GiB",
    "cards": 4
   },
   "model": {
    "repository": "Qwen/Qwen3.8-Flash-Next-FP8",
    "revision": "bcd9f01ddc9cff2316eb84281bebcd5b058bddce",
    "manifest": "repro/qwen38-flash-next-fp8-tp4-mtp3-b70/model-contract.json"
   },
   "runtime": {
    "kind": "native",
    "repository": "vllm-project/vllm",
    "revision": "622191221475b53cc6f7f4d847860939f4c300ab",
    "build": "repro/qwen38-flash-next-fp8-tp4-mtp1-hctriton-b70-37tps-20260907/README.md"
   },
   "project_patches": {
    "required": true,
    "items": [
     "patches/qwen38-flash-next-fp8-b70/vllm-placement-mtp1-005dc578/vllm-q38-placement-mtp1-005dc578-20260906.bundle",
     "patches/qwen38-flash-next-fp8-b70/vllm-placement-mtp1-005dc578/series.sha256",
     "patches/qwen38-flash-next-fp8-b70/vllm-lossless-mtp1-1b2a17c1/vllm-q38-lossless-mtp1-1b2a17c1-20260906.bundle",
     "patches/qwen38-flash-next-fp8-b70/vllm-lossless-mtp1-1b2a17c1/series.sha256",
     "patches/qwen38-flash-next-fp8-b70/oneccl-4ceafd1-b70-public/lib.sha256",
     "repro/qwen38-flash-next-fp8-tp4-mtp3-b70/runtime-stage.sha256",
     "patches/qwen38-flash-next-fp8-b70/vllm-hctriton-mtp1-62219122/vllm-q38-hctriton-mtp1-62219122-20260907.bundle",
     "patches/qwen38-flash-next-fp8-b70/vllm-hctriton-mtp1-62219122/series.sha256"
    ],
    "reason": "The record loads the lab's nine placement commits over its 55-commit lossless-MTP1 overlay on public vLLM 76cfe1cd, the hosted 2f829747 kernel stage, and the hosted public oneCCL 4ceafd1 build; all are pinned by bytes, and the never-routed expert placement file is pinned by hash."
   },
   "commands": {
    "preflight": "repro/qwen38-flash-next-fp8-tp4-mtp1-hctriton-b70-37tps-20260907/verify-identity.sh",
    "launch": "REPRO_ATTEMPT=<unused number above 272> repro/qwen38-flash-next-fp8-tp4-mtp1-hctriton-b70-37tps-20260907/run-record-gate.sh",
    "health": "curl -fsS http://127.0.0.1:19900/health",
    "benchmark": "The record gate sends the fixed cold 12-prompt realistic suite exactly once through the frozen client and compares every output SHA-256 and gate with the record (check-replay-result.py).",
    "stop": "The packet supervisor tears the server down after the client run and writes /tmp/q38-mtp1-ple-only-a<N>.rc; verify no Worker_TP process or listener on the packet port remains."
   },
   "dependencies": [
    "repro/qwen38-flash-next-fp8-tp4-mtp1-hctriton-b70-37tps-20260907/verify-moe-selection-frozen.py",
    "data/localmaxxing-responses/qwen38-flash-next-fp8-tp4-mtp0-placement-hctriton-realistic-20260907.json",
    "data/localmaxxing-responses/qwen38-flash-next-fp8-tp4-mtp1-placement-hctriton-realistic-20260907.json",
    "experiments/qwen38-flash-next-fp8-b70/configs/moe-m1-w13-n32",
    "experiments/qwen38-flash-next-fp8-b70/data/20260906-q38-expert-host-placement-3p5gib-per-rank.json",
    "experiments/qwen38-flash-next-fp8-b70/data/20260907-tp4-mtp1-a266-a268-a271-exact-2k-pair-summary.json",
    "experiments/qwen38-flash-next-fp8-b70/data/20260907-tp4-mtp1-a271-fresh-repeat-deterministic-summary.json",
    "experiments/qwen38-flash-next-fp8-b70/data/20260907-tp4-mtp1-a272-promotion-attestation.json",
    "experiments/qwen38-flash-next-fp8-b70/data/20260907-tp4-mtp1-a272-realistic-suite-v1-result.json",
    "experiments/qwen38-flash-next-fp8-b70/data/20260907-tp4-mtp1-a273-record-gate-replay-realistic-suite-v1-result.json",
    "experiments/qwen38-flash-next-fp8-b70/tools/launch-tp4-ep4-eager-mtp0-long-context-base.sh",
    "experiments/qwen38-flash-next-fp8-b70/tools/launch-tp4-mtp1-4352-ple-only-a272-fullgraphdet-w13n32.sh",
    "experiments/qwen38-flash-next-fp8-b70/tools/q38-launch-frozen-attempt.sh",
    "experiments/qwen38-flash-next-fp8-b70/tools/run-q38-a272-host-controlled.sh",
    "experiments/qwen38-flash-next-fp8-b70/tools/run-tp4-mtp1-4352-ple-only-a272-fullgraphdet-w13n32-client.sh",
    "experiments/qwen38-flash-next-fp8-b70/tools/supervise-tp4-mtp1-4352-ple-only-a272-fullgraphdet-w13n32.sh",
    "experiments/qwen38-flash-next-fp8-b70/tools/verify-moe-m1-w13-n32-selection.py",
    "patches/qwen38-flash-next-fp8-b70/oneccl-4ceafd1-b70-public/README.md",
    "patches/qwen38-flash-next-fp8-b70/oneccl-4ceafd1-b70-public/lib.sha256",
    "patches/qwen38-flash-next-fp8-b70/vllm-hctriton-mtp1-62219122/README.md",
    "patches/qwen38-flash-next-fp8-b70/vllm-hctriton-mtp1-62219122/series.sha256",
    "patches/qwen38-flash-next-fp8-b70/vllm-hctriton-mtp1-62219122/verify-series.sh",
    "patches/qwen38-flash-next-fp8-b70/vllm-hctriton-mtp1-62219122/vllm-q38-hctriton-mtp1-62219122-20260907.bundle",
    "patches/qwen38-flash-next-fp8-b70/vllm-lossless-mtp1-1b2a17c1/README.md",
    "patches/qwen38-flash-next-fp8-b70/vllm-lossless-mtp1-1b2a17c1/series.sha256",
    "patches/qwen38-flash-next-fp8-b70/vllm-lossless-mtp1-1b2a17c1/verify-series.sh",
    "patches/qwen38-flash-next-fp8-b70/vllm-lossless-mtp1-1b2a17c1/vllm-q38-lossless-mtp1-1b2a17c1-20260906.bundle",
    "patches/qwen38-flash-next-fp8-b70/vllm-placement-mtp1-005dc578/README.md",
    "patches/qwen38-flash-next-fp8-b70/vllm-placement-mtp1-005dc578/series.sha256",
    "patches/qwen38-flash-next-fp8-b70/vllm-placement-mtp1-005dc578/verify-series.sh",
    "patches/qwen38-flash-next-fp8-b70/vllm-placement-mtp1-005dc578/vllm-q38-placement-mtp1-005dc578-20260906.bundle",
    "repro/qwen38-flash-next-fp8-tp4-mtp1-hctriton-b70-37tps-20260907/CONTAINER-STATUS.md",
    "repro/qwen38-flash-next-fp8-tp4-mtp1-hctriton-b70-37tps-20260907/Dockerfile",
    "repro/qwen38-flash-next-fp8-tp4-mtp1-hctriton-b70-37tps-20260907/README.md",
    "repro/qwen38-flash-next-fp8-tp4-mtp1-hctriton-b70-37tps-20260907/build-image.sh",
    "repro/qwen38-flash-next-fp8-tp4-mtp1-hctriton-b70-37tps-20260907/check-replay-result.py",
    "repro/qwen38-flash-next-fp8-tp4-mtp1-hctriton-b70-37tps-20260907/container-serve.sh",
    "repro/qwen38-flash-next-fp8-tp4-mtp1-hctriton-b70-37tps-20260907/evidence/a271-run.sha256",
    "repro/qwen38-flash-next-fp8-tp4-mtp1-hctriton-b70-37tps-20260907/evidence/a272-run.sha256",
    "repro/qwen38-flash-next-fp8-tp4-mtp1-hctriton-b70-37tps-20260907/evidence/a273-record-gate-replay.sha256",
    "repro/qwen38-flash-next-fp8-tp4-mtp1-hctriton-b70-37tps-20260907/evidence/a273-record-gate.log",
    "repro/qwen38-flash-next-fp8-tp4-mtp1-hctriton-b70-37tps-20260907/frozen-a272-packet.sha256",
    "repro/qwen38-flash-next-fp8-tp4-mtp1-hctriton-b70-37tps-20260907/identity.json",
    "repro/qwen38-flash-next-fp8-tp4-mtp1-hctriton-b70-37tps-20260907/make-replay-attempt.py",
    "repro/qwen38-flash-next-fp8-tp4-mtp1-hctriton-b70-37tps-20260907/run-record-gate.sh",
    "repro/qwen38-flash-next-fp8-tp4-mtp1-hctriton-b70-37tps-20260907/verifier-pin.txt",
    "repro/qwen38-flash-next-fp8-tp4-mtp1-hctriton-b70-37tps-20260907/verify-identity.sh",
    "repro/qwen38-flash-next-fp8-tp4-mtp1-hctriton-b70-37tps-20260907/wait-and-run-client.sh",
    "repro/qwen38-flash-next-fp8-tp4-mtp3-b70/RELEASE-NOTES.md",
    "repro/qwen38-flash-next-fp8-tp4-mtp3-b70/model-contract.json",
    "repro/qwen38-flash-next-fp8-tp4-mtp3-b70/pip-freeze-observed.txt",
    "repro/qwen38-flash-next-fp8-tp4-mtp3-b70/prepare-runtime.py",
    "repro/qwen38-flash-next-fp8-tp4-mtp3-b70/runtime-stage.sha256",
    "repro/qwen38-flash-next-fp8-tp4-mtp3-b70/verify-model.py",
    "repro/rapid-model-snapshots-b70/realistic-suite-v1.json",
    "results/qwen38-flash-next-fp8-b70/README.md",
    "scripts/bench-openai-concurrency.py",
    "scripts/bench-openai-realistic-suite.py",
    "scripts/bench-openai-token-depth-suite.py"
   ],
   "missing": [
    "tested platform installation (venv, oneAPI runtime, xe driver)",
    "installable dependency hash lock",
    "record-specific model acquisition helper",
    "non-originating-host replay",
    "built and replayed container image",
    "beginner recovery flow",
    "decode, prefill, and TTFT context sweep"
   ]
  },
  {
   "manifest": "packages/qwen38-flash-next-fp8-tp4-mtp1-lossless-b70-27tps-20260905/package.json",
   "format": "b70-model-package-v1",
   "id": "qwen38-flash-next-fp8-tp4-mtp1-lossless-b70-27tps-20260905",
   "name": "Qwen3.8 Flash-Next FP8 with lossless MTP1 on four Intel Arc Pro B70 cards",
   "status": "candidate",
   "audience": "expert",
   "guide": "repro/qwen38-flash-next-fp8-tp4-mtp1-lossless-b70-27tps-20260905/README.md",
   "clean_host_tested": false,
   "library": {
    "model_family": "Qwen3.8 Flash-Next",
    "publisher": "Qwen",
    "variant": "official FP8 export, TP4/EP4, deterministic full-decode graph, lossless MTP1",
    "summary": "Qwen's 125B-A6B hybrid-attention MoE, served from its official FP8 weights across four Arc Pro B70 cards with the expert and n-gram tables that do not fit placed in host memory. One speculative token per step with every output identical to the no-speculation line. A replay of the lab's certified run.",
    "quantization": "FP8 block-128 weights / BF16 KV",
    "runtime_label": "vLLM XPU",
    "operating_systems": [
     "Linux"
    ],
    "delivery": [
     "native"
    ],
    "modalities": [
     "text"
    ],
    "use_cases": [
     "general",
     "coding",
     "deterministic serving research"
    ],
    "tags": [
     "four cards",
     "TP4",
     "EP4",
     "MTP1",
     "lossless speculation",
     "XPU graph",
     "UVA offload",
     "lab replay",
     "originating host"
    ],
    "published_at": "2026-09-06",
    "featured_metric": {
     "value": 27.048435404121527,
     "unit": "tok/s",
     "label": "class-balanced decode median",
     "scope": "Median of prompt-class medians over 99 inter-token intervals after TTFT on the fixed cold 12-prompt realistic suite, sent once (A189, 2026-09-05).",
     "evidence": "experiments/qwen38-flash-next-fp8-b70/data/20260905-tp4-mtp1-a189-realistic-suite-v1-result.json"
    }
   },
   "contributors": [
    {
     "id": "b70-optimization-lab",
     "name": "neural.download lab",
     "kind": "lab",
     "profile": "https://github.com/steveseguin/b70-optimization-lab",
     "contribution": "Flash-Next XPU bring-up (PLE and embedding UVA placement, Qwen4Exp dispatch, QSA and GDN exactness), deterministic full-decode graph, lossless MTP1 verification, the VRAM-headroom root cause and fix, frozen-packet certification, and record packaging.",
     "status": "integrated",
     "validated_effect": "27.048435 tok/s class-balanced with 12/12 outputs identical to the MTP0 line, against 14.43 tok/s for the previously approved Flash-Next line; fresh-server repeat reproduced every output pin.",
     "evidence": "experiments/qwen38-flash-next-fp8-b70/data/20260905-tp4-mtp1-a189-promotion-attestation.json"
    }
   ],
   "hardware": {
    "accelerator": "Intel Arc Pro B70 32 GiB",
    "cards": 4
   },
   "model": {
    "repository": "Qwen/Qwen3.8-Flash-Next-FP8",
    "revision": "bcd9f01ddc9cff2316eb84281bebcd5b058bddce",
    "manifest": "repro/qwen38-flash-next-fp8-tp4-mtp3-b70/model-contract.json"
   },
   "runtime": {
    "kind": "native",
    "repository": "vllm-project/vllm",
    "revision": "1b2a17c1e7c41985d6a5e0eb324ada4775c25e60",
    "build": "repro/qwen38-flash-next-fp8-tp4-mtp1-lossless-b70-27tps-20260905/README.md"
   },
   "project_patches": {
    "required": true,
    "items": [
     "patches/qwen38-flash-next-fp8-b70/vllm-lossless-mtp1-1b2a17c1/vllm-q38-lossless-mtp1-1b2a17c1-20260906.bundle",
     "patches/qwen38-flash-next-fp8-b70/vllm-lossless-mtp1-1b2a17c1/series.sha256",
     "patches/qwen38-flash-next-fp8-b70/oneccl-4ceafd1-b70-public/lib.sha256",
     "repro/qwen38-flash-next-fp8-tp4-mtp3-b70/runtime-stage.sha256"
    ],
    "reason": "The record loads the lab's 55-commit XPU overlay over public vLLM 76cfe1cd, the hosted 2f829747 kernel stage, and the hosted public oneCCL 4ceafd1 build; all three are pinned by bytes."
   },
   "commands": {
    "preflight": "repro/qwen38-flash-next-fp8-tp4-mtp1-lossless-b70-27tps-20260905/verify-identity.sh",
    "launch": "REPRO_ATTEMPT=<unused number above 189> repro/qwen38-flash-next-fp8-tp4-mtp1-lossless-b70-27tps-20260905/run-record-gate.sh",
    "health": "curl -fsS http://127.0.0.1:19900/health",
    "benchmark": "The record gate sends the fixed cold 12-prompt realistic suite exactly once through the frozen client and compares every output SHA-256 and gate with the record (check-replay-result.py).",
    "stop": "The packet supervisor tears the server down after the client run and writes /tmp/q38-mtp1-ple-only-a<N>.rc; verify no Worker_TP process or listener on the packet port remains."
   },
   "dependencies": [
    "repro/qwen38-flash-next-fp8-tp4-mtp1-lossless-b70-27tps-20260905/verify-moe-selection-frozen.py",
    "data/localmaxxing-responses/qwen38-flash-next-fp8-tp4-mtp1-headroom-realistic-20260905.json",
    "experiments/qwen38-flash-next-fp8-b70/configs/moe-m1-w13-n32",
    "experiments/qwen38-flash-next-fp8-b70/data/20260905-tp4-mtp1-a189-promotion-attestation.json",
    "experiments/qwen38-flash-next-fp8-b70/data/20260905-tp4-mtp1-a189-realistic-suite-v1-result.json",
    "experiments/qwen38-flash-next-fp8-b70/data/20260905-tp4-mtp1-a190-fresh-repeat-deterministic-summary.json",
    "experiments/qwen38-flash-next-fp8-b70/tools/launch-tp4-ep4-eager-mtp0-long-context-base.sh",
    "experiments/qwen38-flash-next-fp8-b70/tools/launch-tp4-mtp1-4352-ple-only-a189-fullgraphdet-w13n32.sh",
    "experiments/qwen38-flash-next-fp8-b70/tools/q38-launch-frozen-attempt.sh",
    "experiments/qwen38-flash-next-fp8-b70/tools/run-q38-a189-host-controlled.sh",
    "experiments/qwen38-flash-next-fp8-b70/tools/run-tp4-mtp1-4352-ple-only-a189-fullgraphdet-w13n32-client.sh",
    "experiments/qwen38-flash-next-fp8-b70/tools/supervise-tp4-mtp1-4352-ple-only-a189-fullgraphdet-w13n32.sh",
    "experiments/qwen38-flash-next-fp8-b70/tools/verify-moe-m1-w13-n32-selection.py",
    "patches/qwen38-flash-next-fp8-b70/oneccl-4ceafd1-b70-public/README.md",
    "patches/qwen38-flash-next-fp8-b70/oneccl-4ceafd1-b70-public/lib.sha256",
    "patches/qwen38-flash-next-fp8-b70/vllm-lossless-mtp1-1b2a17c1/README.md",
    "patches/qwen38-flash-next-fp8-b70/vllm-lossless-mtp1-1b2a17c1/series.sha256",
    "patches/qwen38-flash-next-fp8-b70/vllm-lossless-mtp1-1b2a17c1/verify-series.sh",
    "patches/qwen38-flash-next-fp8-b70/vllm-lossless-mtp1-1b2a17c1/vllm-q38-lossless-mtp1-1b2a17c1-20260906.bundle",
    "repro/qwen38-flash-next-fp8-tp4-mtp1-lossless-b70-27tps-20260905/CONTAINER-STATUS.md",
    "repro/qwen38-flash-next-fp8-tp4-mtp1-lossless-b70-27tps-20260905/Dockerfile",
    "repro/qwen38-flash-next-fp8-tp4-mtp1-lossless-b70-27tps-20260905/README.md",
    "repro/qwen38-flash-next-fp8-tp4-mtp1-lossless-b70-27tps-20260905/build-image.sh",
    "repro/qwen38-flash-next-fp8-tp4-mtp1-lossless-b70-27tps-20260905/check-replay-result.py",
    "repro/qwen38-flash-next-fp8-tp4-mtp1-lossless-b70-27tps-20260905/container-serve.sh",
    "repro/qwen38-flash-next-fp8-tp4-mtp1-lossless-b70-27tps-20260905/evidence/a189-run.sha256",
    "repro/qwen38-flash-next-fp8-tp4-mtp1-lossless-b70-27tps-20260905/evidence/a190-run.sha256",
    "repro/qwen38-flash-next-fp8-tp4-mtp1-lossless-b70-27tps-20260905/frozen-a189-packet.sha256",
    "repro/qwen38-flash-next-fp8-tp4-mtp1-lossless-b70-27tps-20260905/identity.json",
    "repro/qwen38-flash-next-fp8-tp4-mtp1-lossless-b70-27tps-20260905/make-replay-attempt.py",
    "repro/qwen38-flash-next-fp8-tp4-mtp1-lossless-b70-27tps-20260905/run-record-gate.sh",
    "repro/qwen38-flash-next-fp8-tp4-mtp1-lossless-b70-27tps-20260905/verifier-pin.txt",
    "repro/qwen38-flash-next-fp8-tp4-mtp1-lossless-b70-27tps-20260905/verify-identity.sh",
    "repro/qwen38-flash-next-fp8-tp4-mtp1-lossless-b70-27tps-20260905/wait-and-run-client.sh",
    "repro/qwen38-flash-next-fp8-tp4-mtp3-b70/RELEASE-NOTES.md",
    "repro/qwen38-flash-next-fp8-tp4-mtp3-b70/model-contract.json",
    "repro/qwen38-flash-next-fp8-tp4-mtp3-b70/pip-freeze-observed.txt",
    "repro/qwen38-flash-next-fp8-tp4-mtp3-b70/prepare-runtime.py",
    "repro/qwen38-flash-next-fp8-tp4-mtp3-b70/runtime-stage.sha256",
    "repro/qwen38-flash-next-fp8-tp4-mtp3-b70/verify-model.py",
    "repro/rapid-model-snapshots-b70/realistic-suite-v1.json",
    "results/qwen38-flash-next-fp8-b70/README.md",
    "scripts/bench-openai-concurrency.py",
    "scripts/bench-openai-realistic-suite.py",
    "scripts/bench-openai-token-depth-suite.py"
   ],
   "missing": [
    "tested platform installation (venv, oneAPI runtime, xe driver)",
    "installable dependency hash lock",
    "record-specific model acquisition helper",
    "non-originating-host replay",
    "built and replayed container image",
    "beginner recovery flow",
    "decode, prefill, and TTFT context sweep"
   ]
  },
  {
   "manifest": "packages/qwen38-flash-next-fp8-tp4-mtp1-placement-b70-32tps-20260906/package.json",
   "format": "b70-model-package-v1",
   "id": "qwen38-flash-next-fp8-tp4-mtp1-placement-b70-32tps-20260906",
   "name": "Qwen3.8 Flash-Next FP8 with lossless MTP1 and never-routed experts host-placed on four Intel Arc Pro B70 cards",
   "status": "candidate",
   "audience": "expert",
   "guide": "repro/qwen38-flash-next-fp8-tp4-mtp1-placement-b70-32tps-20260906/README.md",
   "clean_host_tested": false,
   "library": {
    "model_family": "Qwen3.8 Flash-Next",
    "publisher": "Qwen",
    "variant": "official FP8 export, TP4/EP4, deterministic full-decode graph, lossless MTP1, never-routed experts host-placed",
    "summary": "Qwen's 125B-A6B hybrid-attention MoE, served from its official FP8 weights across four Arc Pro B70 cards. The n-gram table and embeddings live in host memory, every hot expert stays on the cards, and the experts a routing census never selects are parked in host memory behind a per-expert table in the MoE kernel. One speculative token per step with every output identical to the no-speculation line. A replay of the lab's certified run.",
    "quantization": "FP8 block-128 weights / BF16 KV",
    "runtime_label": "vLLM XPU",
    "operating_systems": [
     "Linux"
    ],
    "delivery": [
     "native"
    ],
    "modalities": [
     "text"
    ],
    "use_cases": [
     "general",
     "coding",
     "deterministic serving research"
    ],
    "tags": [
     "four cards",
     "TP4",
     "EP4",
     "MTP1",
     "lossless speculation",
     "XPU graph",
     "UVA offload",
     "expert host placement",
     "lab replay",
     "originating host"
    ],
    "published_at": "2026-09-06",
    "featured_metric": {
     "value": 31.929483735220813,
     "unit": "tok/s",
     "label": "class-balanced decode median",
     "scope": "Median of prompt-class medians over 99 inter-token intervals after TTFT on the fixed cold 12-prompt realistic suite, sent once (A226, 2026-09-06).",
     "evidence": "experiments/qwen38-flash-next-fp8-b70/data/20260906-tp4-mtp1-a226-realistic-suite-v1-result.json"
    }
   },
   "contributors": [
    {
     "id": "b70-optimization-lab",
     "name": "neural.download lab",
     "kind": "lab",
     "profile": "https://github.com/steveseguin/b70-optimization-lab",
     "contribution": "Flash-Next XPU bring-up, deterministic full-decode graph, lossless MTP1 verification, the VRAM-headroom root cause, the per-expert host placement (offset-table Triton MoE kernel, load-time placement, tuned-map key on the logical expert count), frozen-packet certification, and record packaging.",
     "status": "integrated",
     "validated_effect": "31.929484 tok/s class-balanced with 12/12 outputs identical to the approved MTP0 and MTP1 rows, against 27.048435 tok/s for the lossless MTP1 headroom line; exact-2K 33.2 and exact-4K 32.5 tok/s on a separate server.",
     "evidence": "experiments/qwen38-flash-next-fp8-b70/data/20260906-tp4-mtp1-a226-promotion-attestation.json"
    }
   ],
   "hardware": {
    "accelerator": "Intel Arc Pro B70 32 GiB",
    "cards": 4
   },
   "model": {
    "repository": "Qwen/Qwen3.8-Flash-Next-FP8",
    "revision": "bcd9f01ddc9cff2316eb84281bebcd5b058bddce",
    "manifest": "repro/qwen38-flash-next-fp8-tp4-mtp3-b70/model-contract.json"
   },
   "runtime": {
    "kind": "native",
    "repository": "vllm-project/vllm",
    "revision": "005dc57895896f770157ea94f68e473e7447139e",
    "build": "repro/qwen38-flash-next-fp8-tp4-mtp1-placement-b70-32tps-20260906/README.md"
   },
   "project_patches": {
    "required": true,
    "items": [
     "patches/qwen38-flash-next-fp8-b70/vllm-placement-mtp1-005dc578/vllm-q38-placement-mtp1-005dc578-20260906.bundle",
     "patches/qwen38-flash-next-fp8-b70/vllm-placement-mtp1-005dc578/series.sha256",
     "patches/qwen38-flash-next-fp8-b70/vllm-lossless-mtp1-1b2a17c1/vllm-q38-lossless-mtp1-1b2a17c1-20260906.bundle",
     "patches/qwen38-flash-next-fp8-b70/vllm-lossless-mtp1-1b2a17c1/series.sha256",
     "patches/qwen38-flash-next-fp8-b70/oneccl-4ceafd1-b70-public/lib.sha256",
     "repro/qwen38-flash-next-fp8-tp4-mtp3-b70/runtime-stage.sha256"
    ],
    "reason": "The record loads the lab's nine placement commits over its 55-commit lossless-MTP1 overlay on public vLLM 76cfe1cd, the hosted 2f829747 kernel stage, and the hosted public oneCCL 4ceafd1 build; all are pinned by bytes, and the never-routed expert placement file is pinned by hash."
   },
   "commands": {
    "preflight": "repro/qwen38-flash-next-fp8-tp4-mtp1-placement-b70-32tps-20260906/verify-identity.sh",
    "launch": "REPRO_ATTEMPT=<unused number above 226> repro/qwen38-flash-next-fp8-tp4-mtp1-placement-b70-32tps-20260906/run-record-gate.sh",
    "health": "curl -fsS http://127.0.0.1:19900/health",
    "benchmark": "The record gate sends the fixed cold 12-prompt realistic suite exactly once through the frozen client and compares every output SHA-256 and gate with the record (check-replay-result.py).",
    "stop": "The packet supervisor tears the server down after the client run and writes /tmp/q38-mtp1-ple-only-a<N>.rc; verify no Worker_TP process or listener on the packet port remains."
   },
   "dependencies": [
    "repro/qwen38-flash-next-fp8-tp4-mtp1-placement-b70-32tps-20260906/verify-moe-selection-frozen.py",
    "data/localmaxxing-responses/qwen38-flash-next-fp8-tp4-mtp1-placement-realistic-20260906.json",
    "experiments/qwen38-flash-next-fp8-b70/configs/moe-m1-w13-n32",
    "experiments/qwen38-flash-next-fp8-b70/data/20260906-q38-expert-host-placement-3p5gib-per-rank.json",
    "experiments/qwen38-flash-next-fp8-b70/data/20260906-tp4-mtp1-a190-a225-exact-2k-pair-summary.json",
    "experiments/qwen38-flash-next-fp8-b70/data/20260906-tp4-mtp1-a225-fresh-repeat-deterministic-summary.json",
    "experiments/qwen38-flash-next-fp8-b70/data/20260906-tp4-mtp1-a226-promotion-attestation.json",
    "experiments/qwen38-flash-next-fp8-b70/data/20260906-tp4-mtp1-a226-realistic-suite-v1-result.json",
    "experiments/qwen38-flash-next-fp8-b70/data/20260906-tp4-mtp1-a229-record-gate-replay-realistic-suite-v1-result.json",
    "experiments/qwen38-flash-next-fp8-b70/tools/launch-tp4-ep4-eager-mtp0-long-context-base.sh",
    "experiments/qwen38-flash-next-fp8-b70/tools/launch-tp4-mtp1-4352-ple-only-a226-fullgraphdet-w13n32.sh",
    "experiments/qwen38-flash-next-fp8-b70/tools/q38-launch-frozen-attempt.sh",
    "experiments/qwen38-flash-next-fp8-b70/tools/run-q38-a226-host-controlled.sh",
    "experiments/qwen38-flash-next-fp8-b70/tools/run-tp4-mtp1-4352-ple-only-a226-fullgraphdet-w13n32-client.sh",
    "experiments/qwen38-flash-next-fp8-b70/tools/supervise-tp4-mtp1-4352-ple-only-a226-fullgraphdet-w13n32.sh",
    "experiments/qwen38-flash-next-fp8-b70/tools/verify-moe-m1-w13-n32-selection.py",
    "patches/qwen38-flash-next-fp8-b70/oneccl-4ceafd1-b70-public/README.md",
    "patches/qwen38-flash-next-fp8-b70/oneccl-4ceafd1-b70-public/lib.sha256",
    "patches/qwen38-flash-next-fp8-b70/vllm-lossless-mtp1-1b2a17c1/README.md",
    "patches/qwen38-flash-next-fp8-b70/vllm-lossless-mtp1-1b2a17c1/series.sha256",
    "patches/qwen38-flash-next-fp8-b70/vllm-lossless-mtp1-1b2a17c1/verify-series.sh",
    "patches/qwen38-flash-next-fp8-b70/vllm-lossless-mtp1-1b2a17c1/vllm-q38-lossless-mtp1-1b2a17c1-20260906.bundle",
    "patches/qwen38-flash-next-fp8-b70/vllm-placement-mtp1-005dc578/README.md",
    "patches/qwen38-flash-next-fp8-b70/vllm-placement-mtp1-005dc578/series.sha256",
    "patches/qwen38-flash-next-fp8-b70/vllm-placement-mtp1-005dc578/verify-series.sh",
    "patches/qwen38-flash-next-fp8-b70/vllm-placement-mtp1-005dc578/vllm-q38-placement-mtp1-005dc578-20260906.bundle",
    "repro/qwen38-flash-next-fp8-tp4-mtp1-placement-b70-32tps-20260906/CONTAINER-STATUS.md",
    "repro/qwen38-flash-next-fp8-tp4-mtp1-placement-b70-32tps-20260906/Dockerfile",
    "repro/qwen38-flash-next-fp8-tp4-mtp1-placement-b70-32tps-20260906/README.md",
    "repro/qwen38-flash-next-fp8-tp4-mtp1-placement-b70-32tps-20260906/build-image.sh",
    "repro/qwen38-flash-next-fp8-tp4-mtp1-placement-b70-32tps-20260906/check-replay-result.py",
    "repro/qwen38-flash-next-fp8-tp4-mtp1-placement-b70-32tps-20260906/container-serve.sh",
    "repro/qwen38-flash-next-fp8-tp4-mtp1-placement-b70-32tps-20260906/evidence/a225-run.sha256",
    "repro/qwen38-flash-next-fp8-tp4-mtp1-placement-b70-32tps-20260906/evidence/a226-run.sha256",
    "repro/qwen38-flash-next-fp8-tp4-mtp1-placement-b70-32tps-20260906/evidence/a229-record-gate-replay.sha256",
    "repro/qwen38-flash-next-fp8-tp4-mtp1-placement-b70-32tps-20260906/evidence/a229-record-gate.log",
    "repro/qwen38-flash-next-fp8-tp4-mtp1-placement-b70-32tps-20260906/frozen-a226-packet.sha256",
    "repro/qwen38-flash-next-fp8-tp4-mtp1-placement-b70-32tps-20260906/identity.json",
    "repro/qwen38-flash-next-fp8-tp4-mtp1-placement-b70-32tps-20260906/make-replay-attempt.py",
    "repro/qwen38-flash-next-fp8-tp4-mtp1-placement-b70-32tps-20260906/run-record-gate.sh",
    "repro/qwen38-flash-next-fp8-tp4-mtp1-placement-b70-32tps-20260906/verifier-pin.txt",
    "repro/qwen38-flash-next-fp8-tp4-mtp1-placement-b70-32tps-20260906/verify-identity.sh",
    "repro/qwen38-flash-next-fp8-tp4-mtp1-placement-b70-32tps-20260906/wait-and-run-client.sh",
    "repro/qwen38-flash-next-fp8-tp4-mtp3-b70/RELEASE-NOTES.md",
    "repro/qwen38-flash-next-fp8-tp4-mtp3-b70/model-contract.json",
    "repro/qwen38-flash-next-fp8-tp4-mtp3-b70/pip-freeze-observed.txt",
    "repro/qwen38-flash-next-fp8-tp4-mtp3-b70/prepare-runtime.py",
    "repro/qwen38-flash-next-fp8-tp4-mtp3-b70/runtime-stage.sha256",
    "repro/qwen38-flash-next-fp8-tp4-mtp3-b70/verify-model.py",
    "repro/rapid-model-snapshots-b70/realistic-suite-v1.json",
    "results/qwen38-flash-next-fp8-b70/README.md",
    "scripts/bench-openai-concurrency.py",
    "scripts/bench-openai-realistic-suite.py",
    "scripts/bench-openai-token-depth-suite.py"
   ],
   "missing": [
    "tested platform installation (venv, oneAPI runtime, xe driver)",
    "installable dependency hash lock",
    "record-specific model acquisition helper",
    "non-originating-host replay",
    "built and replayed container image",
    "beginner recovery flow",
    "decode, prefill, and TTFT context sweep"
   ]
  },
  {
   "manifest": "packages/qwen38-flash-next-fp8-tp4-mtp1-qsafused-b70-38tps-20260907/package.json",
   "format": "b70-model-package-v1",
   "id": "qwen38-flash-next-fp8-tp4-mtp1-qsafused-b70-38tps-20260907",
   "name": "Qwen3.8 Flash-Next FP8 with lossless MTP1, never-routed experts host-placed and both reference Triton kernels restored, on four Intel Arc Pro B70 cards",
   "status": "candidate",
   "audience": "expert",
   "guide": "repro/qwen38-flash-next-fp8-tp4-mtp1-qsafused-b70-38tps-20260907/README.md",
   "clean_host_tested": false,
   "library": {
    "model_family": "Qwen3.8 Flash-Next",
    "publisher": "Qwen",
    "variant": "official FP8 export, TP4/EP4, deterministic full-decode graph, lossless MTP1, never-routed experts host-placed, hyper-connection glue and QSA pre-indexer restored to the reference kernels",
    "summary": "Qwen's 125B-A6B hybrid-attention MoE, served from its official FP8 weights across four Arc Pro B70 cards. The n-gram table and embeddings live in host memory, every hot expert stays on the cards, the experts a routing census never selects are parked in host memory behind a per-expert table in the MoE kernel, and the two Triton kernels the XPU port had replaced, the hyper-connection glue and the QSA pre-indexer, are restored to the model's own reference implementations. One speculative token per step, lossless within the lineage. The outputs are a new authority whose difference is at depth: exact-2K coincides with the certified stream and exact-4K does not, while the quality profile is preserved byte for byte. A replay of the lab's certified run.",
    "quantization": "FP8 block-128 weights / BF16 KV",
    "runtime_label": "vLLM XPU",
    "operating_systems": [
     "Linux"
    ],
    "delivery": [
     "native"
    ],
    "modalities": [
     "text"
    ],
    "use_cases": [
     "general",
     "coding",
     "deterministic serving research"
    ],
    "tags": [
     "four cards",
     "TP4",
     "EP4",
     "MTP1",
     "XPU graph",
     "UVA offload",
     "expert host placement",
     "lab replay",
     "originating host",
     "lossless speculation",
     "Triton HC glue",
     "new output authority",
     "fused QSA pre-indexer",
     "reference kernels restored"
    ],
    "published_at": "2026-09-07",
    "featured_metric": {
     "value": 37.82565425736173,
     "unit": "tok/s",
     "label": "class-balanced decode median",
     "scope": "Median of prompt-class medians over 99 inter-token intervals after TTFT on the fixed cold 12-prompt realistic suite, sent once (A306, 2026-09-07).",
     "evidence": "experiments/qwen38-flash-next-fp8-b70/data/20260907-tp4-mtp1-a306-qsafused-realistic-suite-v1-result.json"
    }
   },
   "contributors": [
    {
     "id": "b70-optimization-lab",
     "name": "neural.download lab",
     "kind": "lab",
     "profile": "https://github.com/steveseguin/b70-optimization-lab",
     "contribution": "Flash-Next XPU bring-up, deterministic full-decode graph, lossless MTP1 verification, the VRAM-headroom root cause, the per-expert host placement (offset-table Triton MoE kernel, load-time placement, tuned-map key on the logical expert count), the Triton hyper-connection glue and the fused QSA pre-indexer on XPU (the torch-fallback root cause of the 8.4 ms hyper-connection mixes), new-authority certification across five servers, frozen-packet certification, and record packaging.",
     "status": "integrated",
     "validated_effect": "37.825654 tok/s class-balanced (A306) against 37.045844 for the Triton-HC line and 31.929484 for the placement line, with MTP1 lossless within the lineage; exact-2K 38.98 and exact-4K 39.30 tok/s on the certification battery (A305); the outputs are a new authority whose difference is at depth, with the quality profile preserved byte for byte.",
     "evidence": "experiments/qwen38-flash-next-fp8-b70/data/20260907-tp4-mtp1-a306-promotion-attestation.json"
    }
   ],
   "hardware": {
    "accelerator": "Intel Arc Pro B70 32 GiB",
    "cards": 4
   },
   "model": {
    "repository": "Qwen/Qwen3.8-Flash-Next-FP8",
    "revision": "bcd9f01ddc9cff2316eb84281bebcd5b058bddce",
    "manifest": "repro/qwen38-flash-next-fp8-tp4-mtp3-b70/model-contract.json"
   },
   "runtime": {
    "kind": "native",
    "repository": "vllm-project/vllm",
    "revision": "6d8724577dabbee5fa0bbc70c4d927c6174c8d8a",
    "build": "repro/qwen38-flash-next-fp8-tp4-mtp1-qsafused-b70-38tps-20260907/README.md"
   },
   "project_patches": {
    "required": true,
    "items": [
     "patches/qwen38-flash-next-fp8-b70/vllm-placement-mtp1-005dc578/vllm-q38-placement-mtp1-005dc578-20260906.bundle",
     "patches/qwen38-flash-next-fp8-b70/vllm-placement-mtp1-005dc578/series.sha256",
     "patches/qwen38-flash-next-fp8-b70/vllm-lossless-mtp1-1b2a17c1/vllm-q38-lossless-mtp1-1b2a17c1-20260906.bundle",
     "patches/qwen38-flash-next-fp8-b70/vllm-lossless-mtp1-1b2a17c1/series.sha256",
     "patches/qwen38-flash-next-fp8-b70/oneccl-4ceafd1-b70-public/lib.sha256",
     "repro/qwen38-flash-next-fp8-tp4-mtp3-b70/runtime-stage.sha256",
     "patches/qwen38-flash-next-fp8-b70/vllm-hctriton-mtp1-62219122/vllm-q38-hctriton-mtp1-62219122-20260907.bundle",
     "patches/qwen38-flash-next-fp8-b70/vllm-hctriton-mtp1-62219122/series.sha256",
     "patches/qwen38-flash-next-fp8-b70/vllm-qsafused-mtp1-6d872457/vllm-q38-qsafused-mtp1-6d872457-20260907.bundle",
     "patches/qwen38-flash-next-fp8-b70/vllm-qsafused-mtp1-6d872457/series.sha256"
    ],
    "reason": "The record loads the lab's nine placement commits over its 55-commit lossless-MTP1 overlay on public vLLM 76cfe1cd, the hosted 2f829747 kernel stage, and the hosted public oneCCL 4ceafd1 build; all are pinned by bytes, and the never-routed expert placement file is pinned by hash."
   },
   "commands": {
    "preflight": "repro/qwen38-flash-next-fp8-tp4-mtp1-qsafused-b70-38tps-20260907/verify-identity.sh",
    "launch": "REPRO_ATTEMPT=<unused number above 272> repro/qwen38-flash-next-fp8-tp4-mtp1-qsafused-b70-38tps-20260907/run-record-gate.sh",
    "health": "curl -fsS http://127.0.0.1:19900/health",
    "benchmark": "The record gate sends the fixed cold 12-prompt realistic suite exactly once through the frozen client and compares every output SHA-256 and gate with the record (check-replay-result.py).",
    "stop": "The packet supervisor tears the server down after the client run and writes /tmp/q38-mtp1-ple-only-a<N>.rc; verify no Worker_TP process or listener on the packet port remains."
   },
   "dependencies": [
    "data/localmaxxing-responses/qwen38-flash-next-fp8-tp4-mtp0-qsafused-realistic-20260907.json",
    "data/localmaxxing-responses/qwen38-flash-next-fp8-tp4-mtp1-qsafused-realistic-20260907.json",
    "experiments/qwen38-flash-next-fp8-b70/configs/moe-m1-w13-n32",
    "experiments/qwen38-flash-next-fp8-b70/data/20260906-q38-expert-host-placement-3p5gib-per-rank.json",
    "experiments/qwen38-flash-next-fp8-b70/data/20260907-tp4-mtp1-a305-fresh-repeat-deterministic-summary.json",
    "experiments/qwen38-flash-next-fp8-b70/data/20260907-tp4-mtp1-a306-promotion-attestation.json",
    "experiments/qwen38-flash-next-fp8-b70/data/20260907-tp4-mtp1-a306-qsafused-realistic-suite-v1-result.json",
    "experiments/qwen38-flash-next-fp8-b70/data/20260907-tp4-mtp1-a307-record-gate-replay-realistic-suite-v1-result.json",
    "experiments/qwen38-flash-next-fp8-b70/data/20260907-tp4-mtp1-qsafused-exact-2k-pair-summary.json",
    "experiments/qwen38-flash-next-fp8-b70/tools/launch-tp4-ep4-eager-mtp0-long-context-base.sh",
    "experiments/qwen38-flash-next-fp8-b70/tools/launch-tp4-mtp1-4352-ple-only-a306-fullgraphdet-w13n32.sh",
    "experiments/qwen38-flash-next-fp8-b70/tools/q38-launch-frozen-attempt.sh",
    "experiments/qwen38-flash-next-fp8-b70/tools/run-q38-a306-host-controlled.sh",
    "experiments/qwen38-flash-next-fp8-b70/tools/run-tp4-mtp1-4352-ple-only-a306-fullgraphdet-w13n32-client.sh",
    "experiments/qwen38-flash-next-fp8-b70/tools/supervise-tp4-mtp1-4352-ple-only-a306-fullgraphdet-w13n32.sh",
    "experiments/qwen38-flash-next-fp8-b70/tools/verify-moe-m1-w13-n32-selection.py",
    "patches/qwen38-flash-next-fp8-b70/oneccl-4ceafd1-b70-public/README.md",
    "patches/qwen38-flash-next-fp8-b70/oneccl-4ceafd1-b70-public/lib.sha256",
    "patches/qwen38-flash-next-fp8-b70/vllm-hctriton-mtp1-62219122/README.md",
    "patches/qwen38-flash-next-fp8-b70/vllm-hctriton-mtp1-62219122/series.sha256",
    "patches/qwen38-flash-next-fp8-b70/vllm-hctriton-mtp1-62219122/verify-series.sh",
    "patches/qwen38-flash-next-fp8-b70/vllm-hctriton-mtp1-62219122/vllm-q38-hctriton-mtp1-62219122-20260907.bundle",
    "patches/qwen38-flash-next-fp8-b70/vllm-lossless-mtp1-1b2a17c1/README.md",
    "patches/qwen38-flash-next-fp8-b70/vllm-lossless-mtp1-1b2a17c1/series.sha256",
    "patches/qwen38-flash-next-fp8-b70/vllm-lossless-mtp1-1b2a17c1/verify-series.sh",
    "patches/qwen38-flash-next-fp8-b70/vllm-lossless-mtp1-1b2a17c1/vllm-q38-lossless-mtp1-1b2a17c1-20260906.bundle",
    "patches/qwen38-flash-next-fp8-b70/vllm-placement-mtp1-005dc578/README.md",
    "patches/qwen38-flash-next-fp8-b70/vllm-placement-mtp1-005dc578/series.sha256",
    "patches/qwen38-flash-next-fp8-b70/vllm-placement-mtp1-005dc578/verify-series.sh",
    "patches/qwen38-flash-next-fp8-b70/vllm-placement-mtp1-005dc578/vllm-q38-placement-mtp1-005dc578-20260906.bundle",
    "patches/qwen38-flash-next-fp8-b70/vllm-qsafused-mtp1-6d872457/README.md",
    "patches/qwen38-flash-next-fp8-b70/vllm-qsafused-mtp1-6d872457/series.sha256",
    "patches/qwen38-flash-next-fp8-b70/vllm-qsafused-mtp1-6d872457/verify-series.sh",
    "patches/qwen38-flash-next-fp8-b70/vllm-qsafused-mtp1-6d872457/vllm-q38-qsafused-mtp1-6d872457-20260907.bundle",
    "repro/qwen38-flash-next-fp8-tp4-mtp1-qsafused-b70-38tps-20260907/CONTAINER-STATUS.md",
    "repro/qwen38-flash-next-fp8-tp4-mtp1-qsafused-b70-38tps-20260907/Dockerfile",
    "repro/qwen38-flash-next-fp8-tp4-mtp1-qsafused-b70-38tps-20260907/README.md",
    "repro/qwen38-flash-next-fp8-tp4-mtp1-qsafused-b70-38tps-20260907/build-image.sh",
    "repro/qwen38-flash-next-fp8-tp4-mtp1-qsafused-b70-38tps-20260907/check-replay-result.py",
    "repro/qwen38-flash-next-fp8-tp4-mtp1-qsafused-b70-38tps-20260907/container-serve.sh",
    "repro/qwen38-flash-next-fp8-tp4-mtp1-qsafused-b70-38tps-20260907/evidence/a305-run.sha256",
    "repro/qwen38-flash-next-fp8-tp4-mtp1-qsafused-b70-38tps-20260907/evidence/a306-run.sha256",
    "repro/qwen38-flash-next-fp8-tp4-mtp1-qsafused-b70-38tps-20260907/evidence/a307-record-gate-replay.sha256",
    "repro/qwen38-flash-next-fp8-tp4-mtp1-qsafused-b70-38tps-20260907/evidence/a307-record-gate.log",
    "repro/qwen38-flash-next-fp8-tp4-mtp1-qsafused-b70-38tps-20260907/frozen-a306-packet.sha256",
    "repro/qwen38-flash-next-fp8-tp4-mtp1-qsafused-b70-38tps-20260907/identity.json",
    "repro/qwen38-flash-next-fp8-tp4-mtp1-qsafused-b70-38tps-20260907/make-replay-attempt.py",
    "repro/qwen38-flash-next-fp8-tp4-mtp1-qsafused-b70-38tps-20260907/run-record-gate.sh",
    "repro/qwen38-flash-next-fp8-tp4-mtp1-qsafused-b70-38tps-20260907/verifier-pin.txt",
    "repro/qwen38-flash-next-fp8-tp4-mtp1-qsafused-b70-38tps-20260907/verify-identity.sh",
    "repro/qwen38-flash-next-fp8-tp4-mtp1-qsafused-b70-38tps-20260907/wait-and-run-client.sh",
    "repro/qwen38-flash-next-fp8-tp4-mtp3-b70/RELEASE-NOTES.md",
    "repro/qwen38-flash-next-fp8-tp4-mtp3-b70/model-contract.json",
    "repro/qwen38-flash-next-fp8-tp4-mtp3-b70/pip-freeze-observed.txt",
    "repro/qwen38-flash-next-fp8-tp4-mtp3-b70/prepare-runtime.py",
    "repro/qwen38-flash-next-fp8-tp4-mtp3-b70/runtime-stage.sha256",
    "repro/qwen38-flash-next-fp8-tp4-mtp3-b70/verify-model.py",
    "repro/rapid-model-snapshots-b70/realistic-suite-v1.json",
    "results/qwen38-flash-next-fp8-b70/README.md",
    "scripts/bench-openai-concurrency.py",
    "scripts/bench-openai-realistic-suite.py",
    "scripts/bench-openai-token-depth-suite.py"
   ],
   "missing": [
    "tested platform installation (venv, oneAPI runtime, xe driver)",
    "installable dependency hash lock",
    "record-specific model acquisition helper",
    "non-originating-host replay",
    "built and replayed container image",
    "beginner recovery flow",
    "decode, prefill, and TTFT context sweep"
   ]
  }
 ]
}