{"slug": "fp8-mtp-draft-stack-for-vllm-6-6-decode-on-modelopt-nvfp4-qwen3-5-family", "title": "FP8 MTP draft stack for vLLM — +6.6% decode on modelopt NVFP4 Qwen3.5-family checkpoints", "summary": "A vLLM contributor published an FP8 multi-token-prediction (MTP) draft stack for vLLM that yields a 6.6% decode throughput gain on modelopt NVFP4 Qwen3.5-family checkpoints. The change quantizes the previously BF16 draft stack (mtp.fc plus the MTP decoder layer's attention and MLP linears) to FP8 e4m3 with per-tensor scales and dynamic activation quantization, and adds an env-gated sliding window (MTP_DRAFT_WINDOW=8192) that bounds the draft's attention reads to a constant. Because the full-attention target still verifies every accepted token, the approach is lossless by construction and can only shift acceptance rate, not the output distribution.", "body_md": "|  | # SPDX-License-Identifier: Apache-2.0 | \n|  | # SPDX-FileCopyrightText: Copyright contributors to the vLLM project | \n|  | \"\"\"Inference-only Qwen3_5 MTP model.\"\"\" | \n|  |  | \n|  | from collections.abc import Iterable | \n|  |  | \n|  | import torch | \n|  | from torch import nn | \n|  |  | \n|  | from vllm._aiter_ops import rocm_aiter_ops | \n|  | from vllm.compilation.decorators import support_torch_compile | \n|  | from vllm.config import VllmConfig, get_current_vllm_config | \n|  | from vllm.distributed import get_pp_group, tensor_model_parallel_all_gather | \n|  | from vllm.logger import init_logger | \n|  | from vllm.model_executor.layers.linear import ColumnParallelLinear | \n|  | from vllm.model_executor.layers.logits_processor import LogitsProcessor | \n|  | from vllm.model_executor.layers.vocab_parallel_embedding import ( | \n|  | ParallelLMHead, | \n|  | VocabParallelEmbedding, | \n|  | ) | \n|  | from vllm.model_executor.models.interfaces import LocalArgmaxMixin | \n|  | from vllm.model_executor.models.qwen3_5 import ( | \n|  | Qwen3_5DecoderLayer, | \n|  | Qwen3_5Model, | \n|  | Qwen3_5RMSNorm, | \n|  | ) | \n|  | from vllm.model_executor.models.qwen3_next import ( | \n|  | QwenNextMixtureOfExperts, | \n|  | _is_shared_expert_fse_compatible, | \n|  | ) | \n|  | from vllm.model_executor.models.utils import sequence_parallel_chunk | \n|  | from vllm.sequence import IntermediateTensors | \n|  | from vllm.transformers_utils.configs.qwen3_5 import Qwen3_5TextConfig | \n|  | from vllm.transformers_utils.configs.qwen3_5_moe import Qwen3_5MoeTextConfig | \n|  |  | \n|  | from .interfaces import ( | \n|  | MultiModalEmbeddings, | \n|  | SupportsMultiModal, | \n|  | _require_is_multimodal, | \n|  | ) | \n|  | from .utils import ( | \n|  | AutoWeightsLoader, | \n|  | PPMissingLayer, | \n|  | _merge_multimodal_embeddings, | \n|  | make_empty_intermediate_tensors_factory, | \n|  | maybe_fuse_shared_experts, | \n|  | maybe_prefix, | \n|  | ) | \n|  |  | \n|  | logger = init_logger(__name__) | \n|  |  | \n|  | # Windowed-MTP (arXiv 2607.21535): sliding window on the DRAFT's attention | \n|  | # only. The MTP draft otherwise runs full attention over the whole KV cache | \n|  | # at every draft step, so its read grows linearly with context; windowing | \n|  | # bounds it to a constant. Lossless by construction — the full-attention | \n|  | # target still verifies every accepted token, so only the proposals (and | \n|  | # hence acceptance rate) can change, never the output distribution. | \n|  | # Env-gated: MTP_DRAFT_WINDOW=8192 (absent/0 = off). | \n|  | import os as _os | \n|  |  | \n|  | _draft_window = int(_os.environ.get(\"MTP_DRAFT_WINDOW\", \"0\")) | \n|  | if _draft_window > 0: | \n|  | from vllm.model_executor.layers.attention import Attention as _Attention | \n|  |  | \n|  | _orig_attention_init = _Attention.__init__ | \n|  |  | \n|  | def _mtp_windowed_attention_init(self, *args, **kwargs): | \n|  | if \"mtp.layers.\" in kwargs.get(\"prefix\", \"\"): | \n|  | kwargs.setdefault(\"per_layer_sliding_window\", _draft_window) | \n|  | _orig_attention_init(self, *args, **kwargs) | \n|  |  | \n|  | _Attention.__init__ = _mtp_windowed_attention_init | \n|  | logger.info( | \n|  | \"Windowed-MTP: draft attention sliding window = %d tokens\", | \n|  | _draft_window, | \n|  | ) | \n|  |  | \n|  | # house: FP8 the BF16 draft stack (mtp.fc + the MTP decoder layer's | \n|  | # attention/MLP linears). modelopt_fp4 checkpoints exclude mtp*, so the whole | \n|  | # draft stack is BF16 passthrough (~850 MiB/layer read per draft forward). | \n|  | # With MTP_FP8=1 and an -mtpfp8 variant checkpoint (docker/quant/fp8_mtp.py) | \n|  | # those linears load as FP8 e4m3 with per-tensor scales + dynamic activation | \n|  | # quant. Draft proposals only — the target still verifies every token, so | \n|  | # outputs are unchanged; only acceptance rate can shift. The 40960-row draft | \n|  | # head stays BF16 (ParallelLMHead has no FP8 method). | \n|  | _mtp_fp8 = _os.environ.get(\"MTP_FP8\", \"0\") == \"1\" | \n|  |  | \n|  |  | \n|  | def _mtp_fp8_config(): | \n|  | from vllm.model_executor.layers.quantization.fp8 import Fp8Config | \n|  |  | \n|  | return Fp8Config( | \n|  | is_checkpoint_fp8_serialized=True, | \n|  | activation_scheme=\"dynamic\", | \n|  | ) | \n|  |  | \n|  |  | \n|  | class _Fp8DraftHeadMethod: | \n|  | \"\"\"quant_method for the vocab-truncated draft head under MTP_FP8. | \n|  | ParallelLMHead has no FP8 path, so the BF16 head is converted in place | \n|  | after load (see Qwen3_5MTP.load_weights): weight stored column-major | \n|  | [K, N] e4m3 + per-tensor scale, dynamic per-token activation quant, | \n|  | CUTLASS scaled_mm — same recipe as Fp8LinearMethod. Draft proposals | \n|  | only; the target still verifies every token.\"\"\" | \n|  |  | \n|  | def apply(self, layer, x, bias=None): | \n|  | from vllm import _custom_ops as ops | \n|  |  | \n|  | flat = x.reshape(-1, x.shape[-1]) | \n|  | qx, sx = ops.scaled_fp8_quant(flat, use_per_token_if_dynamic=True) | \n|  | return ops.cutlass_scaled_mm( | \n|  | qx, layer.weight, sx, layer.weight_scale, x.dtype, bias | \n|  | ) | \n|  |  | \n|  |  | \n|  | def _fp8_convert_draft_head(head: nn.Module) -> None: | \n|  | w = head.weight  # [rows_per_rank, hidden] BF16 | \n|  | assert w.dtype == torch.bfloat16 and w.dim() == 2 | \n|  | fp8_max = torch.finfo(torch.float8_e4m3fn).max | \n|  | amax = w.float().abs().amax().clamp_min(1e-12) | \n|  | scale = (amax / fp8_max).reshape(1) | \n|  | qw = (w.float() / scale).clamp(-fp8_max, fp8_max).to(torch.float8_e4m3fn) | \n|  | head.weight = nn.Parameter(qw.t(), requires_grad=False)  # col-major [K, N] | \n|  | head.weight_scale = scale | \n|  | head.quant_method = _Fp8DraftHeadMethod() | \n|  | logger.info( | \n|  | \"MTP_FP8: draft head (%d rows/rank) converted to FP8 e4m3\", | \n|  | qw.shape[0], | \n|  | ) | \n|  |  | \n|  |  | \n|  | @support_torch_compile( | \n|  | dynamic_arg_dims={ | \n|  | \"input_ids\": 0, | \n|  | # positions is of shape (3, seq_len) if mrope is enabled for qwen2-vl, | \n|  | # otherwise (seq_len, ). | \n|  | \"positions\": -1, | \n|  | \"intermediate_tensors\": 0, | \n|  | \"inputs_embeds\": 0, | \n|  | \"hidden_states\": 0, | \n|  | } | \n|  | ) | \n|  | class Qwen3_5MultiTokenPredictor(nn.Module): | \n|  | hf_to_vllm_mapper = Qwen3_5Model.hf_to_vllm_mapper | \n|  |  | \n|  | def __init__(self, *, vllm_config: VllmConfig, prefix: str = \"\"): | \n|  | super().__init__() | \n|  |  | \n|  | model_config = vllm_config.model_config | \n|  | quant_config = vllm_config.quant_config | \n|  |  | \n|  | config: Qwen3_5TextConfig \\| Qwen3_5MoeTextConfig = model_config.hf_text_config | \n|  |  | \n|  | self.config = config | \n|  |  | \n|  | self.vocab_size = config.vocab_size | \n|  |  | \n|  | self.mtp_start_layer_idx = config.num_hidden_layers | \n|  | self.num_mtp_layers = getattr(config, \"mtp_num_hidden_layers\", 1) | \n|  |  | \n|  | self.embed_tokens = VocabParallelEmbedding( | \n|  | self.vocab_size, | \n|  | config.hidden_size, | \n|  | ) | \n|  |  | \n|  | # syv patch: vocab-truncated draft head. If the checkpoint ships | \n|  | # mtp_draft_vocab_ids.pt (built by build_draft_vocab.py) the drafter | \n|  | # scores only those rows (mtp.draft_lm_head.*) instead of the full | \n|  | # 248k-row lm_head; logits for all other ids are -inf. Speculative | \n|  | # decoding stays exact, only the acceptance rate can change. | \n|  | import os as _os | \n|  | self.draft_lm_head = None | \n|  | self.draft_vocab_ids = None | \n|  | # house: model_config.model is an HF repo id here, not a local dir — | \n|  | # resolve it to the cached snapshot so the ids file is found. | \n|  | # snapshot_download refuses partially-fetched snapshots (vLLM never | \n|  | # pulls README/LICENSE); try_to_load_from_cache just resolves the dir. | \n|  | _model_dir = model_config.model | \n|  | if not _os.path.isdir(_model_dir): | \n|  | try: | \n|  | from huggingface_hub import try_to_load_from_cache | \n|  |  | \n|  | _cfg = try_to_load_from_cache(_model_dir, \"config.json\") | \n|  | if isinstance(_cfg, str): | \n|  | _model_dir = _os.path.dirname(_cfg) | \n|  | except Exception: | \n|  | pass | \n|  | _ids_path = _os.path.join(_model_dir, \"mtp_draft_vocab_ids.pt\") | \n|  | if _os.path.exists(_ids_path) and _os.environ.get(\"MTP_DRAFT_VOCAB\", \"1\") != \"0\": | \n|  | _ids = torch.load(_ids_path, map_location=\"cpu\") | \n|  | self.draft_vocab_ids = _ids | \n|  | # house: our checkpoint stores the draft head as plain BF16 rows | \n|  | # sliced from lm_head; modelopt_fp4 would otherwise try to load it | \n|  | # as NVFP4 (same mtp.fc exclusion gap as below). Force unquantized. | \n|  | _draft_quant = ( | \n|  | None | \n|  | if (quant_config and quant_config.get_name() == \"modelopt_fp4\") | \n|  | else quant_config | \n|  | ) | \n|  | self.draft_lm_head = ParallelLMHead( | \n|  | int(_ids.numel()), | \n|  | config.hidden_size, | \n|  | quant_config=_draft_quant, | \n|  | prefix=maybe_prefix(prefix, \"draft_lm_head\"), | \n|  | ) | \n|  | logger.info(\"MTP drafter uses a %d-token draft head\", int(_ids.numel())) | \n|  |  | \n|  | # Workaround: mtp.fc is stored as BF16 in NVFP4 checkpoints but is | \n|  | # missing from hf_quant_config.json exclude_modules. Force unquantized. | \n|  | # Ref: https://github.com/vllm-project/vllm/pull/38650 | \n|  | # Ref: https://github.com/NVIDIA/Model-Optimizer/pull/1124 | \n|  | if quant_config and quant_config.get_name() == \"modelopt_fp4\": | \n|  | fc_quant = _mtp_fp8_config() if _mtp_fp8 else None | \n|  | else: | \n|  | fc_quant = quant_config | \n|  | self.fc = ColumnParallelLinear( | \n|  | self.config.hidden_size * 2, | \n|  | self.config.hidden_size, | \n|  | gather_output=True, | \n|  | bias=False, | \n|  | return_bias=False, | \n|  | quant_config=fc_quant, | \n|  | prefix=f\"{prefix}.fc\", | \n|  | ) | \n|  |  | \n|  | # GPTQ: quantized checkpoints may exclude MTP from quantization via | \n|  | # quantization_config.dynamic with \"-:pattern\" entries. When detected, | \n|  | # disable quantization for MTP layers so they use unquantized params. | \n|  | original_quant = vllm_config.quant_config | \n|  | if ( | \n|  | _mtp_fp8 | \n|  | and quant_config | \n|  | and quant_config.get_name() == \"modelopt_fp4\" | \n|  | ): | \n|  | # house: modelopt excludes mtp* (BF16 passthrough); run the MTP | \n|  | # decoder layer's linears as FP8 on the -mtpfp8 variant instead. | \n|  | vllm_config.quant_config = _mtp_fp8_config() | \n|  | elif quant_config and quant_config.get_name() not in (\"modelopt_fp4\",): | \n|  | hf_qc = getattr(model_config.hf_config, \"quantization_config\", None) | \n|  | if isinstance(hf_qc, dict): | \n|  | dynamic = hf_qc.get(\"dynamic\", {}) | \n|  | if any(k.startswith(\"-:\") and \"mtp\" in k for k in dynamic): | \n|  | vllm_config.quant_config = None | \n|  | self.layers = torch.nn.ModuleList( | \n|  | Qwen3_5DecoderLayer( | \n|  | vllm_config, | \n|  | layer_type=\"full_attention\", | \n|  | prefix=f\"{prefix}.layers.{idx}\", | \n|  | ) | \n|  | for idx in range(self.num_mtp_layers) | \n|  | ) | \n|  | vllm_config.quant_config = original_quant | \n|  | self.make_empty_intermediate_tensors = make_empty_intermediate_tensors_factory( | \n|  | [\"hidden_states\", \"residual\"], config.hidden_size | \n|  | ) | \n|  | self.norm = Qwen3_5RMSNorm(config.hidden_size, eps=config.rms_norm_eps) | \n|  | self.pre_fc_norm_hidden = Qwen3_5RMSNorm( | \n|  | config.hidden_size, eps=config.rms_norm_eps | \n|  | ) | \n|  | self.pre_fc_norm_embedding = Qwen3_5RMSNorm( | \n|  | config.hidden_size, eps=config.rms_norm_eps | \n|  | ) | \n|  |  | \n|  | def embed_input_ids(self, input_ids: torch.Tensor) -> torch.Tensor: | \n|  | return self.embed_tokens(input_ids) | \n|  |  | \n|  | def forward( | \n|  | self, | \n|  | input_ids: torch.Tensor, | \n|  | positions: torch.Tensor, | \n|  | hidden_states: torch.Tensor, | \n|  | intermediate_tensors: IntermediateTensors \\| None = None, | \n|  | inputs_embeds: torch.Tensor \\| None = None, | \n|  | spec_step_idx: int = 0, | \n|  | ) -> torch.Tensor: | \n|  | if get_pp_group().is_first_rank: | \n|  | if inputs_embeds is None: | \n|  | inputs_embeds = self.embed_input_ids(input_ids) | \n|  | assert hidden_states.shape[-1] == inputs_embeds.shape[-1] | \n|  | inputs_embeds = self.pre_fc_norm_embedding(inputs_embeds) | \n|  | hidden_states = self.pre_fc_norm_hidden(hidden_states) | \n|  | hidden_states = torch.cat([inputs_embeds, hidden_states], dim=-1) | \n|  | hidden_states = self.fc(hidden_states) | \n|  | residual = None | \n|  | else: | \n|  | assert intermediate_tensors is not None | \n|  | hidden_states = intermediate_tensors[\"hidden_states\"] | \n|  | residual = intermediate_tensors[\"residual\"] | \n|  |  | \n|  | current_step_idx = spec_step_idx % self.num_mtp_layers | \n|  | mtp_layer = self.layers[current_step_idx] | \n|  | if mtp_layer.use_attn_reduce_scatter_for_moe: | \n|  | assert hidden_states.shape[0] == positions.shape[-1] | \n|  | hidden_states = sequence_parallel_chunk(hidden_states) | \n|  | assert residual is None | \n|  | hidden_states, residual = mtp_layer( | \n|  | positions=positions, | \n|  | hidden_states=hidden_states, | \n|  | residual=residual, | \n|  | ) | \n|  |  | \n|  | if not get_pp_group().is_last_rank: | \n|  | return IntermediateTensors( | \n|  | {\"hidden_states\": hidden_states, \"residual\": residual} | \n|  | ) | \n|  |  | \n|  | hidden_states, _ = self.norm(hidden_states, residual) | \n|  | if mtp_layer.use_attn_reduce_scatter_for_moe: | \n|  | hidden_states = tensor_model_parallel_all_gather(hidden_states, 0) | \n|  | hidden_states = hidden_states[: positions.shape[-1]] | \n|  | return hidden_states | \n|  |  | \n|  | def load_weights(self, weights: Iterable[tuple[str, torch.Tensor]]) -> set[str]: | \n|  | weights = maybe_fuse_shared_experts( | \n|  | weights, | \n|  | enabled=rocm_aiter_ops.is_fusion_moe_shared_experts_enabled() | \n|  | and _is_shared_expert_fse_compatible( | \n|  | get_current_vllm_config().quant_config | \n|  | ), | \n|  | n_routed_experts=getattr(self.config, \"num_experts\", 0), | \n|  | n_shared_experts=1, | \n|  | ckpt_prefix=\"mlp.shared_expert\", | \n|  | ) | \n|  | loader = AutoWeightsLoader(self) | \n|  | return loader.load_weights(weights, mapper=self.hf_to_vllm_mapper) | \n|  |  | \n|  |  | \n|  | @support_torch_compile( | \n|  | dynamic_arg_dims={ | \n|  | \"input_ids\": 0, | \n|  | # positions is of shape (3, seq_len) if mrope is enabled for qwen2-vl, | \n|  | # otherwise (seq_len, ). | \n|  | \"positions\": -1, | \n|  | \"intermediate_tensors\": 0, | \n|  | \"inputs_embeds\": 0, | \n|  | \"hidden_states\": 0, | \n|  | } | \n|  | ) | \n|  | class Qwen3_5MTP(LocalArgmaxMixin, nn.Module, SupportsMultiModal): | \n|  | packed_modules_mapping = { | \n|  | \"qkv_proj\": [ | \n|  | \"q_proj\", | \n|  | \"k_proj\", | \n|  | \"v_proj\", | \n|  | ], | \n|  | \"gate_up_proj\": [\"gate_proj\", \"up_proj\"], | \n|  | } | \n|  |  | \n|  | def __init__(self, *, vllm_config: VllmConfig, prefix: str = \"\"): | \n|  | config = vllm_config.model_config.hf_text_config | \n|  | self.vllm_config = vllm_config | \n|  | cache_config = vllm_config.cache_config | \n|  | if cache_config.mamba_cache_mode == \"all\": | \n|  | raise NotImplementedError( | \n|  | \"Qwen3_5MTP currently does not support 'all' prefix caching, \" | \n|  | \"please use '--mamba-cache-mode=align' instead\" | \n|  | ) | \n|  |  | \n|  | self.quant_config = vllm_config.quant_config | \n|  |  | \n|  | super().__init__() | \n|  | self.config = config | \n|  | self.model = Qwen3_5MultiTokenPredictor( | \n|  | vllm_config=vllm_config, prefix=maybe_prefix(prefix, \"mtp\") | \n|  | ) | \n|  |  | \n|  | if get_pp_group().is_last_rank: | \n|  | self.lm_head = ParallelLMHead( | \n|  | config.vocab_size, | \n|  | config.hidden_size, | \n|  | quant_config=self.quant_config, | \n|  | prefix=maybe_prefix(prefix, \"lm_head\"), | \n|  | ) | \n|  | if config.tie_word_embeddings: | \n|  | self.lm_head = self.lm_head.tie_weights(self.model.embed_tokens) | \n|  | else: | \n|  | self.lm_head = PPMissingLayer() | \n|  |  | \n|  | self.logits_processor = LogitsProcessor(config.vocab_size) | \n|  | # syv patch: vocab-truncated draft head | \n|  | self.draft_logits_processor = ( | \n|  | LogitsProcessor(int(self.model.draft_vocab_ids.numel())) | \n|  | if getattr(self.model, \"draft_lm_head\", None) is not None | \n|  | else None | \n|  | ) | \n|  |  | \n|  | def embed_input_ids( | \n|  | self, | \n|  | input_ids: torch.Tensor, | \n|  | multimodal_embeddings: MultiModalEmbeddings \\| None = None, | \n|  | *, | \n|  | is_multimodal: torch.Tensor \\| None = None, | \n|  | ) -> torch.Tensor: | \n|  | inputs_embeds = self._embed_text_input_ids( | \n|  | input_ids, | \n|  | self.model.embed_input_ids, | \n|  | is_multimodal=is_multimodal, | \n|  | ) | \n|  |  | \n|  | if multimodal_embeddings is None or len(multimodal_embeddings) == 0: | \n|  | return inputs_embeds | \n|  |  | \n|  | is_multimodal = _require_is_multimodal(is_multimodal) | \n|  |  | \n|  | inputs_embeds = _merge_multimodal_embeddings( | \n|  | inputs_embeds=inputs_embeds, | \n|  | multimodal_embeddings=multimodal_embeddings, | \n|  | is_multimodal=is_multimodal, | \n|  | ) | \n|  |  | \n|  | return inputs_embeds | \n|  |  | \n|  | def forward( | \n|  | self, | \n|  | input_ids: torch.Tensor, | \n|  | positions: torch.Tensor, | \n|  | hidden_states: torch.Tensor, | \n|  | intermediate_tensors: IntermediateTensors \\| None = None, | \n|  | inputs_embeds: torch.Tensor \\| None = None, | \n|  | **kwargs: object, | \n|  | ): | \n|  | hidden_states = self.model( | \n|  | input_ids, positions, hidden_states, intermediate_tensors, inputs_embeds | \n|  | ) | \n|  | return hidden_states | \n|  |  | \n|  | def compute_logits( | \n|  | self, | \n|  | hidden_states: torch.Tensor, | \n|  | spec_step_idx: int = 0, | \n|  | ) -> torch.Tensor \\| None: | \n|  | # syv patch: vocab-truncated draft head | \n|  | if self.draft_logits_processor is not None: | \n|  | sub = self.draft_logits_processor(self.model.draft_lm_head, hidden_states) | \n|  | if sub is None: | \n|  | return None | \n|  | ids = self.model.draft_vocab_ids | \n|  | if ids.device != sub.device: | \n|  | ids = ids.to(sub.device) | \n|  | self.model.draft_vocab_ids = ids | \n|  | full = sub.new_full((sub.shape[0], self.config.vocab_size), float(\"-inf\")) | \n|  | full.index_copy_(1, ids, sub) | \n|  | return full | \n|  | return self.logits_processor(self.lm_head, hidden_states) | \n|  |  | \n|  | def load_weights(self, weights: Iterable[tuple[str, torch.Tensor]]) -> set[str]: | \n|  | def remap_weight_names(weights): | \n|  | for name, weight in weights: | \n|  | # syv patch: skip the truncated draft head when it is disabled | \n|  | if \"draft_lm_head\" in name and self.model.draft_lm_head is None: | \n|  | continue | \n|  | if name.startswith(\"mtp.\"): | \n|  | name = name.replace(\"mtp.\", \"model.\") | \n|  | elif any(key in name for key in [\"embed_tokens\", \"lm_head\"]): | \n|  | if \"embed_tokens\" in name: | \n|  | name = name.replace(\"language_model.\", \"\") | \n|  | else: | \n|  | continue | \n|  | yield name, weight | \n|  |  | \n|  | loader = AutoWeightsLoader(self) | \n|  | loaded = loader.load_weights(remap_weight_names(weights)) | \n|  | # house: MTP_FP8 also FP8s the BF16 draft head in place (no FP8 | \n|  | # ParallelLMHead path in vLLM). Runs after load, before CUDA graphs. | \n|  | if _mtp_fp8 and getattr(self.model, \"draft_lm_head\", None) is not None: | \n|  | _fp8_convert_draft_head(self.model.draft_lm_head) | \n|  | return loaded | \n|  |  | \n|  |  | \n|  | class Qwen3_5MoeMTP(Qwen3_5MTP, QwenNextMixtureOfExperts): | \n|  | def __init__(self, *, vllm_config: VllmConfig, prefix: str = \"\"): | \n|  | super().__init__(vllm_config=vllm_config, prefix=prefix) | \n|  | self.set_moe_parameters() |", "url": "https://wpnews.pro/news/fp8-mtp-draft-stack-for-vllm-6-6-decode-on-modelopt-nvfp4-qwen3-5-family", "canonical_source": "https://gist.github.com/jaderinoo/0d1c0487073cef9d30ef4fd03499fa54", "published_at": "2026-09-15 17:52:06+00:00", "updated_at": "2026-09-28 15:19:37.410977+00:00", "lang": "en", "topics": ["large-language-models", "ai-infrastructure", "mlops", "developer-tools"], "entities": ["vLLM", "Qwen3.5", "modelopt", "NVIDIA"], "also_reported_by": [], "alternates": {"html": "https://wpnews.pro/news/fp8-mtp-draft-stack-for-vllm-6-6-decode-on-modelopt-nvfp4-qwen3-5-family", "markdown": "https://wpnews.pro/news/fp8-mtp-draft-stack-for-vllm-6-6-decode-on-modelopt-nvfp4-qwen3-5-family.md", "text": "https://wpnews.pro/news/fp8-mtp-draft-stack-for-vllm-6-6-decode-on-modelopt-nvfp4-qwen3-5-family.txt", "jsonld": "https://wpnews.pro/news/fp8-mtp-draft-stack-for-vllm-6-6-decode-on-modelopt-nvfp4-qwen3-5-family.jsonld"}}