{
 "format": "neural-download-model-family-v1",
 "id": "qwen-4b",
 "primary_packet_id": "qwen35-4b-w4a16-b70",
 "name": "Qwen 4B",
 "display_name": "Qwen3.5 4B",
 "publisher": "Qwen / Alibaba",
 "updated_at": "2026-09-11",
 "summary": "Alibaba's smallest dense-hybrid Qwen3.5 with a shipped MTP head, served on one Arc Pro B70. Only the INT4 W4A16 build is packaged: the FP8 build of the same model is not repeat-exact on this stack, while the INT4 route passes every identity gate.",
 "architecture": {
  "class": "Qwen3_5ForConditionalGeneration",
  "model_type": "qwen3_5_text",
  "mtp_hidden_layers": 1,
  "evidence": "repro/qwen35-4b-w4a16-b70/README.md"
 },
 "dimensions": {
  "weight_revision": [
   "qwen3.5-4b-w4a16-7a613872",
   "qwen3.5-4b-fp8-397b7ba4"
  ],
  "weight_quantization": [
   "W4A16",
   "FP8-dynamic"
  ],
  "runtime": [
   "vLLM XPU 0.27.2rc1.dev77 (R276 image 521eb277)",
   "vLLM XPU 0.27.2rc1.dev77 (R293 image 40d46730 = R276 521eb277 + class-consistent FP16 linears behind CLASSPAD)"
  ],
  "tp": [
   1,
   2
  ],
  "mtp": [
   0,
   3
  ],
  "configured_max_context_tokens": [
   1024,
   8192,
   32768
  ],
  "kv": [
   "f16"
  ]
 },
 "weight_revisions": [
  {
   "id": "qwen3.5-4b-w4a16-7a613872",
   "label": "Qwen3.5 4B W4A16 (RedHatAI)",
   "role": "measured INT4 weights with the publisher MTP head",
   "repository": "RedHatAI/Qwen3.5-4B-quantized.w4a16",
   "revision": "7a613872f394578b0b52b683ff4ac47516b4bcaf",
   "model_manifest": "repro/qwen35-4b-w4a16-b70/manifests/model-direct-redhatai-qwen35-4b-w4a16-7a613872.json"
  },
  {
   "id": "qwen3.5-4b-fp8-397b7ba4",
   "label": "Qwen3.5 4B FP8-dynamic (RedHatAI)",
   "role": "measured but not packaged: fails the repeat-exactness gate",
   "repository": "RedHatAI/Qwen3.5-4B-FP8-dynamic",
   "revision": "397b7ba47a99b3221ebfc0cfc5a279118cb733ad",
   "model_manifest": "experiments/qwen35-4b-b70/manifests/model-direct-redhatai-qwen35-4b-fp8-dynamic-397b7ba4.json"
  }
 ],
 "model_variants": [],
 "transfer_scope": {
  "status": "one packaged INT4 revision on one card; the FP8 revision is measured and rejected",
  "transfers": [
   "the Qwen3.5-family vLLM XPU registration, the lab wNa16 INT4 kernel, the draft-only INT4 lm_head"
  ],
  "does_not_transfer": [
   "the FP8 route, which is not repeat-exact here",
   "other Qwen3.5 sizes without their own gates"
  ],
  "evidence": "repro/qwen35-4b-w4a16-b70/README.md"
 },
 "model_signals": {
  "b70_fit": {
   "band": "one-card measured",
   "scope": "W4A16, 5.54 GB weights, one B70",
   "basis": "Strict pairs completed on one B70 on 2026-09-07.",
   "reviewed_at": "2026-09-07"
  },
  "quality_evidence": {
   "band": "deployment-identity evidence",
   "scope": "12-prompt strict suite with complete token-id identity gates against a same-configuration no-speculation oracle; canaries; no broad model-quality evaluation",
   "evidence": [
    "experiments/qwen35-4b-b70/data/2026-09-07-qwen35-4b-w4a16-matrix-result.json",
    "experiments/qwen35-4b-b70/data/2026-09-07-qwen35-4b-fp8-repeat-exactness.json"
   ]
  },
  "popularity": {
   "state": "not-scored",
   "reason": "No dated popularity snapshot is stored."
  }
 },
 "run_measurements": [
  {
   "id": "q35-4b-w4a16-tp1-mtp0-graph-v1",
   "state": "lab-measured",
   "revision": "qwen3.5-4b-w4a16-7a613872",
   "variant": "W4A16",
   "quantization": "W4A16",
   "runtime": "vLLM XPU 0.27.2rc1.dev77 (R276 image 521eb277)",
   "config": {
    "tp": 1,
    "mtp": 0,
    "graph": "on",
    "kv": "f16",
    "configured_max_context_tokens": 1024,
    "draft_head": "n/a"
   },
   "profile_id": "qwen35-fp8-strict-completions-v1",
   "measurement_class": "strict fresh-server pair",
   "promotion_status": "promoted MTP0 oracle (v1)",
   "quality_scope": "G1 12/12",
   "workload": "strict 12-prompt six-class suite over the completions API, 512-token cap, cache zero, class-balanced median tok/s over tokens 1-100 after TTFT, two fresh servers",
   "metrics": {
    "decode_tok_s": [
     102.625,
     102.376
    ],
    "ttft_ms": [
     30.21,
     30.32
    ]
   },
   "evidence": "experiments/qwen35-4b-b70/data/2026-09-07-qwen35-4b-w4a16-matrix-result.json"
  },
  {
   "id": "q35-4b-w4a16-tp1-mtp3-graph-int4head-v1",
   "state": "lab-measured",
   "revision": "qwen3.5-4b-w4a16-7a613872",
   "variant": "W4A16",
   "quantization": "W4A16",
   "runtime": "vLLM XPU 0.27.2rc1.dev77 (R276 image 521eb277)",
   "config": {
    "tp": 1,
    "mtp": 3,
    "graph": "on",
    "kv": "f16",
    "configured_max_context_tokens": 1024,
    "draft_head": "draft INT4"
   },
   "profile_id": "qwen35-fp8-strict-completions-v1",
   "measurement_class": "strict fresh-server pair",
   "promotion_status": "promoted headline (v1); LocalMaxxing cmtrj2tp3000hps01n3fadg9d",
   "quality_scope": "G1/G2/G3 12/12",
   "workload": "strict 12-prompt six-class suite over the completions API, 512-token cap, cache zero, class-balanced median tok/s over tokens 1-100 after TTFT, two fresh servers",
   "metrics": {
    "decode_tok_s": [
     177.406,
     177.168
    ],
    "ttft_ms": [
     44.91,
     45.05
    ]
   },
   "evidence": "experiments/qwen35-4b-b70/data/2026-09-07-qwen35-4b-w4a16-matrix-result.json"
  }
 ],
 "series_measurements": [],
 "estimates": [],
 "packets": [
  {
   "id": "qwen35-4b-w4a16-b70",
   "label": "Qwen3.5 4B W4A16 \u00b7 one or two B70 \u00b7 MTP depth 3 with the draft INT4 head",
   "revision": "qwen3.5-4b-w4a16-7a613872",
   "quantization": "W4A16",
   "runtime": "vLLM XPU 0.27.2rc1.dev77 (R293 image 40d46730 = R276 521eb277 + class-consistent FP16 linears behind CLASSPAD)",
   "topologies": [
    1,
    2
   ],
   "status": "candidate",
   "evidence_level": "B70-measured strict pairs",
   "coverage": [
    "decode",
    "TTFT",
    "TP1",
    "TP2",
    "MTP0/3",
    "cache-zero strict suite",
    "stagger-exact concurrency",
    "CLASSPAD many-user mode"
   ],
   "grades": {
    "evidence": {
     "grade": "B",
     "scope": "one- and two-card strict pairs with identity gates and ladders",
     "basis": "two fresh-server pairs per card count, G1-G3 exact on both; the second card adds 35% without speculation and 32% with it, while two-card concurrency identity holds to 32 users without speculation and 16 with it; no clean-host replay yet. 2026-09-11 (R293, CLASSPAD=1): strict gates 12/12 on both card counts at 168/96 (one card) and 226/128 (two cards) tok/s; no-speculation ladders exact to c20 on one card and c16 on two at 25-40% higher aggregate throughput (2520 at c128 on one card, 4015 on two, both 512/512); with a 5 ms admission stagger 64 users are byte-identical to the sequential oracle over twenty passes on one card (2104 tok/s) and two (3164). TP2 is measured lossless at 32K context (18/18, 222 tok/s depth 3).",
     "reviewed_at": "2026-09-11",
     "evidence": [
      "experiments/qwen35-4b-b70/data/2026-09-07-qwen35-4b-w4a16-matrix-result.json",
      "repro/qwen35-4b-w4a16-b70/README.md",
      "experiments/qwen35-4b-b70/data/qwen35-4b-w4a16-tp2-mtp3-graph1-dhint4-20260907-t1-strict-result.json",
      "experiments/qwen35-4b-b70/data/2026-09-11-qwen35-4b-r290-classpad-on-the-server.json",
      "experiments/qwen35-4b-b70/data/2026-09-09-qwen35-4b-tp2-long-context.json",
      "experiments/qwen35-4b-b70/data/2026-09-09-qwen35-4b-staggered-admission-is-exact.json"
     ]
    }
   },
   "projection": {
    "model": "qwen3.5_4b",
    "quant": "int4",
    "runtime": "vllm",
    "spec": "mtp:3",
    "prompt_tokens": 128,
    "output_tokens": 100
   },
   "manifest": "packages/qwen35-4b-w4a16-b70/package.json"
  }
 ],
 "views": [
  {
   "id": "qwen-4b-mtp-depth",
   "title": "Speculative depth on one B70",
   "subtitle": "W4A16 \u00b7 strict completions suite \u00b7 graph capture on \u00b7 the draft-only INT4 head at depth 3",
   "x_label": "MTP depth",
   "discrete": true,
   "metrics": [
    "decode_tok_s",
    "ttft_ms"
   ],
   "series": [
    {
     "label": "W4A16",
     "measurement_ids": [
      "q35-4b-w4a16-tp1-mtp0-graph-v1",
      "q35-4b-w4a16-tp1-mtp3-graph-int4head-v1"
     ],
     "x_from": "config.mtp"
    }
   ]
  }
 ],
 "coverage_views": [
  {
   "id": "qwen-4b-quant-by-mtp",
   "label": "quantization \u00d7 MTP",
   "fixed": "One B70, R276 vLLM XPU, graph capture on. The FP8 row is measured but rejected: it fails the repeat-exactness gate.",
   "row_axis": {
    "key": "variant",
    "label": "Quantization",
    "prefix": ""
   },
   "column_axis": {
    "key": "mtp",
    "label": "MTP",
    "prefix": "MTP"
   },
   "fixed_selectors": {
    "runtime": "vLLM XPU 0.27.2rc1.dev77 (R276 image 521eb277)",
    "tp": 1
   },
   "rows": [
    "W4A16",
    "FP8-dynamic"
   ],
   "columns": [
    0,
    3
   ],
   "cells": {
    "W4A16:0": {
     "state": "lab-measured",
     "label": "D102.62",
     "evidence_id": "q35-4b-w4a16-tp1-mtp0-graph-v1",
     "packet_id": "qwen35-4b-w4a16-b70"
    },
    "W4A16:3": {
     "state": "lab-measured",
     "label": "D177.41",
     "evidence_id": "q35-4b-w4a16-tp1-mtp3-graph-int4head-v1",
     "packet_id": "qwen35-4b-w4a16-b70"
    },
    "FP8-dynamic:0": {
     "state": "missing",
     "label": "rejected: not repeat-exact (11/12, 9/12, 11/12)"
    },
    "FP8-dynamic:3": {
     "state": "missing",
     "label": "not run: base gate failed"
    }
   }
  }
 ],
 "family_closures": []
}