{"slug": "minimax-h3-ref2va-aligned-video-guides-ic-lora-style-v2v-for-ai-toolkit", "title": "MiniMax-H3 ref2va: aligned video guides (IC-LoRA style v2v) for ai-toolkit", "summary": "A developer contributed a patch to ai-toolkit's MiniMax-H3 diffusion model extension that fixes reference-to-video-alignment (ref2va) handling of control images and videos in IC-LoRA-style video-to-video workflows. The change refactors keyframe conversion into a helper and preserves the nested per-batch-item structure of control_tensor_list, which previously collapsed multiple references into a single <Picture> block and caused an \"image features and image tokens do not match\" error in the Qwen3-VL processor.", "body_md": "|  | diff --git a/extensions_built_in/diffusion_models/minimax_h3/minimax_h3.py b/extensions_built_in/diffusion_models/minimax_h3/minimax_h3.py | \n|  | index 4c21927..2509789 100644 | \n|  | --- a/extensions_built_in/diffusion_models/minimax_h3/minimax_h3.py | \n|  | +++ b/extensions_built_in/diffusion_models/minimax_h3/minimax_h3.py | \n|  | @@ -537,44 +537,56 @@ class MinimaxH3Model(BaseModel): | \n|  | # control tensors arrive in [0, 1]; the Qwen3-VL processor wants PIL | \n|  | keyframes_per_prompt = [None] * len(prompt) | \n|  | if control_images is not None: | \n|  | -            if isinstance(control_images, torch.Tensor): | \n|  | -                images = [control_images[i] for i in range(control_images.shape[0])] | \n|  | -            elif isinstance(control_images, list): | \n|  | -                images = [ | \n|  | -                    c[0] if isinstance(c, torch.Tensor) and c.ndim == 4 else c | \n|  | -                    for c in control_images | \n|  | -                ] | \n|  | -            else: | \n|  | -                images = [control_images] | \n|  | -            pil_images = [] | \n|  | -            for img in images: | \n|  | + | \n|  | +            def _to_keyframe(img): | \n|  | if isinstance(img, torch.Tensor): | \n|  | if img.ndim == 4: | \n|  | img = img[0] | \n|  | arr = (img.float().clamp(0, 1) * 255).round().to(torch.uint8) | \n|  | -                    pil_images.append( | \n|  | -                        self._present_image_control( | \n|  | -                            Image.fromarray(arr.permute(1, 2, 0).cpu().numpy()) | \n|  | -                        ) | \n|  | +                    return self._present_image_control( | \n|  | +                        Image.fromarray(arr.permute(1, 2, 0).cpu().numpy()) | \n|  | ) | \n|  | -                elif isinstance(img, str): | \n|  | +                if isinstance(img, str): | \n|  | # a control VIDEO path: 2 fps timestamped presentation over | \n|  | # the SAME frames the latent rows use (dataset treatment | \n|  | # when caching training embeds, sample-length at sampling) | \n|  | ds_cfg = getattr(self, \"_ref_video_dataset_config\", None) | \n|  | -                    pil_images.append( | \n|  | -                        load_video_ref_for_te( | \n|  | -                            self, img, ds_cfg, max_frames=self._sample_ref_max_frames | \n|  | -                        ) | \n|  | +                    return load_video_ref_for_te( | \n|  | +                        self, img, ds_cfg, max_frames=self._sample_ref_max_frames | \n|  | ) | \n|  | -                else: | \n|  | -                    pil_images.append(img) | \n|  | -            if len(pil_images) == 1: | \n|  | -                keyframes_per_prompt = [pil_images] * len(prompt) | \n|  | -            elif len(pil_images) == len(prompt): | \n|  | -                keyframes_per_prompt = [[img] for img in pil_images] | \n|  | +                return img | \n|  | + | \n|  | +            # batch.control_tensor_list is [item][ref]: several references per | \n|  | +            # batch item. It must stay nested per item -- collapsing it emits a | \n|  | +            # single <Picture> block while the image processor flattens the | \n|  | +            # inner list into N images (\"image features and image tokens do not | \n|  | +            # match\", tokens counted from image_grid_thw[0] only). | \n|  | +            if ( | \n|  | +                isinstance(control_images, list) | \n|  | +                and len(control_images) > 0 | \n|  | +                and all(isinstance(c, (list, tuple)) for c in control_images) | \n|  | +            ): | \n|  | +                per_item = [[_to_keyframe(r) for r in refs] for refs in control_images] | \n|  | +                if len(per_item) == 1 and len(prompt) > 1: | \n|  | +                    per_item = per_item * len(prompt) | \n|  | +                keyframes_per_prompt = per_item | \n|  | else: | \n|  | -                keyframes_per_prompt = [pil_images] * len(prompt) | \n|  | +                if isinstance(control_images, torch.Tensor): | \n|  | +                    images = [control_images[i] for i in range(control_images.shape[0])] | \n|  | +                elif isinstance(control_images, list): | \n|  | +                    images = [ | \n|  | +                        c[0] if isinstance(c, torch.Tensor) and c.ndim == 4 else c | \n|  | +                        for c in control_images | \n|  | +                    ] | \n|  | +                else: | \n|  | +                    images = [control_images] | \n|  | +                pil_images = [_to_keyframe(img) for img in images] | \n|  | +                if len(pil_images) == 1: | \n|  | +                    keyframes_per_prompt = [pil_images] * len(prompt) | \n|  | +                elif len(pil_images) == len(prompt): | \n|  | +                    keyframes_per_prompt = [[img] for img in pil_images] | \n|  | +                else: | \n|  | +                    keyframes_per_prompt = [pil_images] * len(prompt) | \n|  |  | \n|  | embeds_list, tags_list = [], [] | \n|  | for p, keyframes in zip(prompt, keyframes_per_prompt): | \n|  | @@ -721,6 +733,23 @@ class MinimaxH3Model(BaseModel): | \n|  | # ------------------------------------------------------------------ | \n|  | # Training forward | \n|  | # ------------------------------------------------------------------ | \n|  | +    def _aligned_ref_flags(self, ref_blocks): | \n|  | +        \"\"\"Which reference blocks share the target's coordinates. | \n|  | + | \n|  | +        With ``align_video_refs`` a *video* reference stops being a loose reference and | \n|  | +        becomes a v2v guide: same rotary clock, same spatial grid as the target, so guide | \n|  | +        frame ``i`` drives output frame `` i`` without the model having to search for the | \n|  | +        correspondence. Image references are never aligned — an identity photo has no | \n|  | +        spatial correspondence with the output and should not claim one. | \n|  | + | \n|  | +        A head-swap pack is exactly this pair: aligned driving video, unaligned identity. | \n|  | +        \"\"\" | \n|  | +        if not ref_blocks or not self.model_config.model_kwargs.get( | \n|  | +            \"align_video_refs\", False | \n|  | +        ): | \n|  | +            return () | \n|  | +        return tuple(block[0] > 1 for block in ref_blocks) | \n|  | + | \n|  | def _build_condition( | \n|  | self, batch: \"DataLoaderBatchDTO\", latent_shape, device, dtype | \n|  | ): | \n|  | @@ -890,6 +919,7 @@ class MinimaxH3Model(BaseModel): | \n|  | text_tag_list = [t for _, t in trimmed] | \n|  |  | \n|  | # --- packed layout (per item: text lengths differ) -------------- | \n|  | +            aligned_refs = self._aligned_ref_flags(ref_blocks) | \n|  | layouts = [] | \n|  | for i in range(batch_size): | \n|  | layouts.append( | \n|  | @@ -901,6 +931,7 @@ class MinimaxH3Model(BaseModel): | \n|  | num_audio_latents=a_lat, | \n|  | keyframe_anchors=keyframe_anchors, | \n|  | ref_blocks=ref_blocks, | \n|  | +                        aligned_refs=aligned_refs, | \n|  | ) | \n|  | ) | \n|  | ( | \n|  | @@ -1128,9 +1159,11 @@ class MinimaxH3Ref2VAModel(MinimaxH3Model): | \n|  | def text_embedding_space_version(self): | \n|  | # the presentation of image references changes the embeds -> new cache key | \n|  | n = self._image_ref_video_frames() | \n|  | -        if n: | \n|  | -            return f\"{self.arch}:img_as_vid{n}\" | \n|  | -        return self.arch | \n|  | +        base = f\"{self.arch}:img_as_vid{n}\" if n else self.arch | \n|  | +        if getattr(self, \"control_latent_only\", False): | \n|  | +            # embeds drop the control media entirely -> different space | \n|  | +            return f\"{base}:ctrl_latent_only\" | \n|  | +        return base | \n|  |  | \n|  | def _present_image_control(self, image: Image.Image): | \n|  | n = self._image_ref_video_frames() | \n|  | @@ -1150,6 +1183,22 @@ class MinimaxH3Ref2VAModel(MinimaxH3Model): | \n|  | # D-OPSD: a no-grad teacher pass with the target as its own reference | \n|  | # becomes the training target for the reference-free student pass | \n|  | self.dopsd = bool(self.model_config.model_kwargs.get(\"dopsd\", False)) | \n|  | +        # VLM-gap variant of D-OPSD: teacher and student share the SAME condition | \n|  | +        # (the aligned guide); the information gap is the VLM's reading of that | \n|  | +        # guide, which only the teacher's text embeds carry. The student's | \n|  | +        # sequence loses the vision tokens -- here ~8.2k of ~14.6k -- so it is | \n|  | +        # cheaper to train and to sample, and it matches ComfyUI's AddGuide path, | \n|  | +        # where the guide reaches the DiT as latents but never reaches the VLM. | \n|  | +        # Control media reaches the DiT as latents only -- it is never presented | \n|  | +        # to the VLM. For a v2v guide that is usually what you want: the guide | \n|  | +        # already lands on the target's rotary grid via align_video_refs, the | \n|  | +        # vision tokens are pure cost (~8.2k of a ~14.6k sequence on a 73-frame | \n|  | +        # 512x288 clip), and inference paths that inject the guide as a keyframe | \n|  | +        # (ComfyUI's Add Guide) never show it to the VLM either -- so training | \n|  | +        # with the presentation creates a train/inference gap. | \n|  | +        self.control_latent_only = bool( | \n|  | +            self.model_config.model_kwargs.get(\"control_latent_only\", False) | \n|  | +        ) | \n|  | if self.dopsd: | \n|  | self.dopsd_self_ref = True | \n|  | self.require_pixel_tensor_cache = True | \n|  | @@ -1336,10 +1385,16 @@ class MinimaxH3Ref2VAModel(MinimaxH3Model): | \n|  | ] | \n|  | cap.release() | \n|  | h0, w0 = frames[0].shape[:2] | \n|  | -        # match the sample canvas's pixel area, own aspect kept | \n|  | -        ph, pw = packing.reference_video_pixel_size( | \n|  | -            w0, h0, gen_config.height, gen_config.width | \n|  | -        ) | \n|  | +        if bool(self.model_config.model_kwargs.get(\"align_video_refs\", False)): | \n|  | +            # aligned guide: the sample canvas exactly, same rule the training path uses. | \n|  | +            # Area-matching the guide's own aspect here would hand build_packed_sequence a | \n|  | +            # grid the target does not have and fail the sample, not the step. | \n|  | +            ph, pw = int(gen_config.height), int(gen_config.width) | \n|  | +        else: | \n|  | +            # match the sample canvas's pixel area, own aspect kept | \n|  | +            ph, pw = packing.reference_video_pixel_size( | \n|  | +                w0, h0, gen_config.height, gen_config.width | \n|  | +            ) | \n|  | pixels = torch.from_numpy(np.stack(frames)).float() / 255.0 * 2.0 - 1.0 | \n|  | pixels = pixels.permute(3, 0, 1, 2)[None]  # (1, 3, T, H, W) | \n|  | pixels = ( | \n|  | @@ -1422,12 +1477,18 @@ class MinimaxH3Ref2VAModel(MinimaxH3Model): | \n|  | \"ref2va: every item in a batch must have the same number of \" | \n|  | \"reference videos\" | \n|  | ) | \n|  | +        # with align_video_refs a control VIDEO is a v2v guide, not a loose reference: | \n|  | +        # it has to sit on the target's exact latent grid (build_packed_sequence enforces | \n|  | +        # it), so it is encoded at the target's size instead of area-matched to its own | \n|  | +        # aspect. Identity images never take this path. | \n|  | +        align = bool(self.model_config.model_kwargs.get(\"align_video_refs\", False)) | \n|  | for ref_idx in range(vid_count): | \n|  | lats = [] | \n|  | auds = [] | \n|  | for per_item in paths_per_item: | \n|  | entry = load_ref_video_latent( | \n|  | -                    self, per_item[ref_idx], batch.dataset_config, target_h, target_w | \n|  | +                    self, per_item[ref_idx], batch.dataset_config, target_h, target_w, | \n|  | +                    align=align, | \n|  | ) | \n|  | lats.append(entry[\"latent\"].to(device, torch.float32)) | \n|  | auds.append(entry.get(\"audio_rows\")) | \n|  | diff --git a/extensions_built_in/diffusion_models/minimax_h3/src/packing.py b/extensions_built_in/diffusion_models/minimax_h3/src/packing.py | \n|  | index 6e77429..39c1c72 100644 | \n|  | --- a/extensions_built_in/diffusion_models/minimax_h3/src/packing.py | \n|  | +++ b/extensions_built_in/diffusion_models/minimax_h3/src/packing.py | \n|  | @@ -297,6 +297,7 @@ def build_packed_sequence( | \n|  | patch_size=(1, 2, 2), | \n|  | keyframe_anchors: Tuple[str, ...] = (), | \n|  | ref_blocks: Tuple[Tuple[int, int, int], ...] = (), | \n|  | +    aligned_refs: Tuple[bool, ...] = (), | \n|  | ) -> PackedLayout: | \n|  | \"\"\"Build the [text \\| conditions \\| target audio \\| target video] layout. | \n|  |  | \n|  | @@ -307,7 +308,22 @@ def build_packed_sequence( | \n|  | for images. References keep their OWN aspect on their own | \n|  | aspect-normalized grid; an image block advances the shared media clock by | \n|  | 1.0, a video block by its temporal span, and the target streams start | \n|  | -    after the cumulative advance).\"\"\" | \n|  | +    after the cumulative advance). | \n|  | + | \n|  | +    ``aligned_refs`` marks blocks that should be *aligned* to the target rather | \n|  | +    than placed beside it: an aligned block reuses the target's rotary clock and | \n|  | +    the target's spatial grid, so its latent frame ``i`` and pixel ``(h, w)`` | \n|  | +    land on exactly the coordinates of the target's, and it does not advance the | \n|  | +    media clock. That is the v2v/IC-LoRA arrangement — attention between a guide | \n|  | +    row and the target row at the same position costs nothing positionally, which | \n|  | +    is what makes frame-accurate control (pose, depth, a driving performance) | \n|  | +    learnable. Ordinary references stay unaligned: an identity reference has no | \n|  | +    spatial correspondence with the output and should not claim one. | \n|  | + | \n|  | +    A head-swap pack therefore uses both — an aligned guide video plus an | \n|  | +    unaligned identity reference. Aligned blocks must match the target's latent | \n|  | +    height and width; their frame count may be shorter (they align from the | \n|  | +    target's first frame).\"\"\" | \n|  | if keyframe_anchors and ref_blocks: | \n|  | raise ValueError(\"keyframe_anchors and ref_blocks are mutually exclusive\") | \n|  | _, ph, pw = patch_size | \n|  | @@ -317,6 +333,17 @@ def build_packed_sequence( | \n|  | # reference's soundtrack packs as clean audio rows immediately BEFORE its | \n|  | # own video rows | \n|  | ref_blocks = tuple(tuple(b) + (0,) * (4 - len(b)) for b in ref_blocks) | \n|  | +    aligned_flags = tuple(aligned_refs) + (False,) * (len(ref_blocks) - len(aligned_refs)) | \n|  | +    if len(aligned_flags) != len(ref_blocks): | \n|  | +        raise ValueError( | \n|  | +            f\"aligned_refs has {len(aligned_refs)} entries for {len(ref_blocks)} ref blocks\" | \n|  | +        ) | \n|  | +    for (_, b_h, b_w, _), aligned in zip(ref_blocks, aligned_flags): | \n|  | +        if aligned and (b_h, b_w) != (latent_height, latent_width): | \n|  | +            raise ValueError( | \n|  | +                \"an aligned reference must share the target's latent resolution \" | \n|  | +                f\"({latent_height}x{latent_width}), got {b_h}x{b_w}\" | \n|  | +            ) | \n|  | ref_vid_rows = [t * (h // ph) * (w // pw) for t, h, w, _ in ref_blocks] | \n|  | ref_aud_rows = [a * AUDIO_CHANNELS for _, _, _, a in ref_blocks] | \n|  | num_cond = ( | \n|  | @@ -337,7 +364,12 @@ def build_packed_sequence( | \n|  | # text rows sit on the time axis at their row index; the media clock | \n|  | # continues from there past the reference blocks, so prompt length (and | \n|  | # reference count/length) shifts the whole media clock | \n|  | -    media_advance = sum(_block_advance(t, a) for t, _, _, a in ref_blocks) | \n|  | +    # Aligned blocks sit ON the target's clock, so they must not push it forward. | \n|  | +    media_advance = sum( | \n|  | +        _block_advance(t, a) | \n|  | +        for (t, _, _, a), aligned in zip(ref_blocks, aligned_flags) | \n|  | +        if not aligned | \n|  | +    ) | \n|  | media_origin = float(num_text) + media_advance | \n|  | position_ids = torch.zeros(seq_len, 3, dtype=torch.float64) | \n|  | position_ids[:num_text, 0] = torch.arange(num_text, dtype=torch.float64) | \n|  | @@ -377,25 +409,34 @@ def build_packed_sequence( | \n|  | cond_video_idx.append(torch.arange(cond_start, audio_start)) | \n|  | ref_cursor = audio_start | \n|  | for i, (ref_t, ref_h, ref_w, ref_a) in enumerate(ref_blocks): | \n|  | -        # each reference on its own aspect-normalized grid (area-matched to | \n|  | -        # the target, so the grids span comparable ranges) | \n|  | -        ref_sqrt_area = math.sqrt(ref_h * ref_w) | \n|  | -        w_grid = _spatial_position_grid(ref_w, pw, ref_sqrt_area) | \n|  | -        ref_grid = torch.stack( | \n|  | -            [ | \n|  | -                g.reshape(-1) | \n|  | -                for g in torch.meshgrid( | \n|  | -                    _spatial_position_grid(ref_h, ph, ref_sqrt_area), | \n|  | -                    w_grid, | \n|  | -                    indexing=\"ij\", | \n|  | -                ) | \n|  | -            ], | \n|  | -            dim=-1, | \n|  | -        ) | \n|  | +        aligned = aligned_flags[i] | \n|  | +        if aligned: | \n|  | +            # the target's own grid and clock: row (i, h, w) of the guide lands on | \n|  | +            # the coordinate of row (i, h, w) of the output | \n|  | +            w_grid = width_grid | \n|  | +            ref_grid = frame_grid | \n|  | +            block_clock = media_origin | \n|  | +        else: | \n|  | +            # each reference on its own aspect-normalized grid (area-matched to | \n|  | +            # the target, so the grids span comparable ranges) | \n|  | +            ref_sqrt_area = math.sqrt(ref_h * ref_w) | \n|  | +            w_grid = _spatial_position_grid(ref_w, pw, ref_sqrt_area) | \n|  | +            ref_grid = torch.stack( | \n|  | +                [ | \n|  | +                    g.reshape(-1) | \n|  | +                    for g in torch.meshgrid( | \n|  | +                        _spatial_position_grid(ref_h, ph, ref_sqrt_area), | \n|  | +                        w_grid, | \n|  | +                        indexing=\"ij\", | \n|  | +                    ) | \n|  | +                ], | \n|  | +                dim=-1, | \n|  | +            ) | \n|  | +            block_clock = ref_clock | \n|  | if ref_a: | \n|  | # soundtrack rows first: channel-major, shared 40/s clock from the | \n|  | # block origin, width pinned to the ref grid's extremes | \n|  | -            a_time = ref_clock + torch.arange(ref_a, dtype=torch.float64) | \n|  | +            a_time = block_clock + torch.arange(ref_a, dtype=torch.float64) | \n|  | rows = slice(ref_cursor, ref_cursor + ref_aud_rows[i]) | \n|  | position_ids[rows, 0] = a_time.repeat(AUDIO_CHANNELS) | \n|  | position_ids[rows, 2] = torch.cat( | \n|  | @@ -408,12 +449,13 @@ def build_packed_sequence( | \n|  | ref_cursor += ref_aud_rows[i] | \n|  | rows_per_ref_frame = ref_grid.shape[0] | \n|  | block = torch.empty(ref_t, rows_per_ref_frame, 3, dtype=torch.float64) | \n|  | -        block[:, :, 0] = _temporal_position_grid(ref_t, ref_clock)[:, None] | \n|  | +        block[:, :, 0] = _temporal_position_grid(ref_t, block_clock)[:, None] | \n|  | block[:, :, 1:] = ref_grid[None] | \n|  | position_ids[ref_cursor : ref_cursor + ref_vid_rows[i]] = block.reshape(-1, 3) | \n|  | cond_video_idx.append(torch.arange(ref_cursor, ref_cursor + ref_vid_rows[i])) | \n|  | ref_cursor += ref_vid_rows[i] | \n|  | -        ref_clock += _block_advance(ref_t, ref_a) | \n|  | +        if not aligned: | \n|  | +            ref_clock += _block_advance(ref_t, ref_a) | \n|  |  | \n|  | # audio rows: channel-major, one rotary unit per latent (40/s = 24fps*5/3), | \n|  | # no height coordinate, width pinned to the grid extremes per channel | \n|  | diff --git a/extensions_built_in/diffusion_models/minimax_h3/src/pipeline.py b/extensions_built_in/diffusion_models/minimax_h3/src/pipeline.py | \n|  | index 6e08cd3..7a65747 100644 | \n|  | --- a/extensions_built_in/diffusion_models/minimax_h3/src/pipeline.py | \n|  | +++ b/extensions_built_in/diffusion_models/minimax_h3/src/pipeline.py | \n|  | @@ -136,6 +136,9 @@ class MiniMaxH3Pipeline: | \n|  | num_audio_latents=a_lat, | \n|  | keyframe_anchors=anchors, | \n|  | ref_blocks=ref_blocks, | \n|  | +            # same rule as training, from the same helper: a layout mismatch between the | \n|  | +            # two is silent — sampling just quietly stops matching what was trained | \n|  | +            aligned_refs=model._aligned_ref_flags(ref_blocks), | \n|  | ) | \n|  | num_cond = layout.num_condition_video_rows | \n|  |  | \n|  | diff --git a/extensions_built_in/diffusion_models/minimax_h3/src/ref_video_cache.py b/extensions_built_in/diffusion_models/minimax_h3/src/ref_video_cache.py | \n|  | index c54c263..a320cbc 100644 | \n|  | --- a/extensions_built_in/diffusion_models/minimax_h3/src/ref_video_cache.py | \n|  | +++ b/extensions_built_in/diffusion_models/minimax_h3/src/ref_video_cache.py | \n|  | @@ -164,7 +164,8 @@ def _cache_path(path: str, hash_dict: dict) -> str: | \n|  |  | \n|  | @torch.no_grad() | \n|  | def load_ref_video_latent( | \n|  | -    model, path: str, dataset_config, target_height: int, target_width: int | \n|  | +    model, path: str, dataset_config, target_height: int, target_width: int, | \n|  | +    align: bool = False, | \n|  | ) -> dict: | \n|  | \"\"\"Returns {\"latent\": (C, T, h, w) cpu tensor, \"num_frames\": int}, | \n|  | encoding + disk-caching on first use. ``model`` is the MinimaxH3 model | \n|  | @@ -173,8 +174,14 @@ def load_ref_video_latent( | \n|  | if mem_cache is None: | \n|  | mem_cache = {} | \n|  | model._ref_video_cache = mem_cache | \n|  | -    if path in mem_cache: | \n|  | -        return mem_cache[path] | \n|  | +    # The entry depends on the TARGET the ref was sized against, not just the file: with | \n|  | +    # multi-resolution buckets the same guide is requested at several target sizes in one | \n|  | +    # run, and a path-only key handed back whatever size happened to be encoded first. | \n|  | +    # Unaligned that passed silently (the guide just sat on a wrong grid, and cost whatever | \n|  | +    # its own size cost); aligned it raises \"must share the target's latent resolution\". | \n|  | +    mem_key = (path, int(target_height), int(target_width), bool(align)) | \n|  | +    if mem_key in mem_cache: | \n|  | +        return mem_cache[mem_key] | \n|  |  | \n|  | cap = cv2.VideoCapture(path) | \n|  | if not cap.isOpened(): | \n|  | @@ -198,10 +205,14 @@ def load_ref_video_latent( | \n|  | ) | \n|  | hash_dict = { | \n|  | \"signature\": get_quick_signature_string(path), | \n|  | -        \"ref_sizing\": \"match_target_area\", | \n|  | +        # an aligned guide is resized to the target's EXACT grid, so the cache key | \n|  | +        # needs the dimensions, not just the area -- two buckets with the same area | \n|  | +        # and different aspects are different tensors here | \n|  | +        \"ref_sizing\": \"match_target_exact\" if align else \"match_target_area\", | \n|  | \"target_area\": int(target_height * target_width) | \n|  | if target_height and target_width | \n|  | else 0, | \n|  | +        \"target_hw\": (int(target_height), int(target_width)) if align else 0, | \n|  | \"num_frames\": num_frames, | \n|  | \"fps\": dataset_config.fps, | \n|  | \"trim_tail\": trim_tail, | \n|  | @@ -217,13 +228,22 @@ def load_ref_video_latent( | \n|  | \"num_frames\": int(sd[\"num_frames\"].item()), | \n|  | \"audio_rows\": sd.get(\"audio_latent\"), | \n|  | } | \n|  | -        mem_cache[path] = entry | \n|  | +        mem_cache[mem_key] = entry | \n|  | return entry | \n|  |  | \n|  | -    # match the target's pixel area (the dataset bucket the target trains at) | \n|  | -    # with the ref's own aspect: same aspect -> identical size; aspect- | \n|  | -    # preserving resize, no crop | \n|  | -    out_h, out_w = reference_video_pixel_size(src_w, src_h, target_height, target_width) | \n|  | +    if align: | \n|  | +        # An aligned guide shares the target's rotary clock AND its spatial grid, so it | \n|  | +        # must land on the target's exact latent resolution -- build_packed_sequence | \n|  | +        # rejects anything else. Area-matching the guide's own aspect would leave 25% of | \n|  | +        # a real head-swap set a few percent off and crash the pack, so the guide is | \n|  | +        # resized to the target outright; the small anisotropy is the price of frame- | \n|  | +        # accurate correspondence, and it only applies to guides, never to identity refs. | \n|  | +        out_h, out_w = int(target_height), int(target_width) | \n|  | +    else: | \n|  | +        # match the target's pixel area (the dataset bucket the target trains at) | \n|  | +        # with the ref's own aspect: same aspect -> identical size; aspect- | \n|  | +        # preserving resize, no crop | \n|  | +        out_h, out_w = reference_video_pixel_size(src_w, src_h, target_height, target_width) | \n|  |  | \n|  | indices = ref_frame_indices( | \n|  | total, src_fps, num_frames, dataset_config.fps, trim_tail | \n|  | @@ -274,5 +294,5 @@ def load_ref_video_latent( | \n|  | os.makedirs(os.path.dirname(cache_file), exist_ok=True) | \n|  | save_file(state_dict, cache_file) | \n|  | entry = {\"latent\": latent, \"num_frames\": num_frames, \"audio_rows\": audio_rows} | \n|  | -    mem_cache[path] = entry | \n|  | +    mem_cache[mem_key] = entry | \n|  | return entry | \n|  | diff --git a/extensions_built_in/diffusion_models/minimax_h3/src/text_encoder.py b/extensions_built_in/diffusion_models/minimax_h3/src/text_encoder.py | \n|  | index f58863d..a9680e4 100644 | \n|  | --- a/extensions_built_in/diffusion_models/minimax_h3/src/text_encoder.py | \n|  | +++ b/extensions_built_in/diffusion_models/minimax_h3/src/text_encoder.py | \n|  | @@ -120,6 +120,8 @@ def encode_minimax_h3_prompt( | \n|  | processor,  # Qwen3VLProcessor (needed only when keyframes are present) | \n|  | prompt: str, | \n|  | keyframes: Optional[List] = None,  # PIL images already on the target canvas | \n|  | +    reference_videos: Optional[List] = None,  # videos sampled to 2 fps, each TCHW/THWC | \n|  | +    reference_video_timestamps: Optional[List[List[float]]] = None, | \n|  | device: Optional[torch.device] = None, | \n|  | dtype: Optional[torch.dtype] = None, | \n|  | max_length: Optional[ | \n|  | @@ -220,6 +222,41 @@ def encode_minimax_h3_prompt( | \n|  | ) | \n|  | pic_idx += 1 | \n|  |  | \n|  | +    if reference_videos: | \n|  | +        vision = processor.video_processor( | \n|  | +            videos=reference_videos, do_sample_frames=False, return_tensors=\"pt\" | \n|  | +        ) | \n|  | +        pixel_values_videos = vision[\"pixel_values_videos\"] | \n|  | +        video_grid_thw = vision[\"video_grid_thw\"] | \n|  | +        merge = processor.video_processor.merge_size**2 | \n|  | +        vision_start = tokenizer.convert_tokens_to_ids(\"<\\|vision_start\\|>\") | \n|  | +        vision_end = tokenizer.convert_tokens_to_ids(\"<\\|vision_end\\|>\") | \n|  | +        video_pad = tokenizer.convert_tokens_to_ids(\"<\\|video_pad\\|>\") | \n|  | +        for i in range(len(reference_videos)): | \n|  | +            label_ids = tokenizer(f\"<Video {i + 1}>: \", add_special_tokens=False)[ | \n|  | +                \"input_ids\" | \n|  | +            ] | \n|  | +            token_ids += label_ids | \n|  | +            token_tags += [TEXT_TAG] * len(label_ids) | \n|  | +            grid_t, grid_h, grid_w = (int(x) for x in video_grid_thw[i]) | \n|  | +            tokens_per_block = (grid_h * grid_w) // merge | \n|  | +            timestamps = ( | \n|  | +                reference_video_timestamps[i] | \n|  | +                if reference_video_timestamps is not None | \n|  | +                else [j / 2.0 for j in range(grid_t * 2)] | \n|  | +            ) | \n|  | +            if len(timestamps) % 2: | \n|  | +                timestamps = list(timestamps) + [timestamps[-1]] | \n|  | +            for block in range(grid_t): | \n|  | +                ts_idx = min(block * 2, len(timestamps) - 2) | \n|  | +                block_ts = (timestamps[ts_idx] + timestamps[ts_idx + 1]) / 2.0 | \n|  | +                time_ids = tokenizer( | \n|  | +                    f\"<{block_ts:.1f} seconds>\", add_special_tokens=False | \n|  | +                )[\"input_ids\"] | \n|  | +                vision_ids = [vision_start] + [video_pad] * tokens_per_block + [vision_end] | \n|  | +                token_ids += time_ids + vision_ids | \n|  | +                token_tags += [TEXT_TAG] * len(time_ids) + [VIDEO_TAG] * len(vision_ids) | \n|  | + | \n|  | prompt_ids = tokenizer(prompt, add_special_tokens=False)[\"input_ids\"] | \n|  | if max_length is not None and max_length > 0: | \n|  | # the cap applies to the caption only; a keyframe's vision block is | \n|  | diff --git a/tests/test_minimax_h3_aligned_refs.py b/tests/test_minimax_h3_aligned_refs.py | \n|  | new file mode 100644 | \n|  | index 0000000..8c565d3 | \n|  | --- /dev/null | \n|  | +++ b/tests/test_minimax_h3_aligned_refs.py | \n|  | @@ -0,0 +1,129 @@ | \n|  | +\"\"\"An aligned reference must land on the target's coordinates exactly. | \n|  | + | \n|  | +That is the whole point of the flag: a guide row and the target row it drives share a rotary | \n|  | +position, so attention between them costs nothing positionally. If the clock or the spatial grid | \n|  | +drifts by even one unit the guide stops being frame-accurate, and the failure is invisible — | \n|  | +training still runs, the model just never learns tight control. | \n|  | +\"\"\" | \n|  | + | \n|  | +import pytest | \n|  | +import torch | \n|  | + | \n|  | +from extensions_built_in.diffusion_models.minimax_h3.src.packing import build_packed_sequence | \n|  | + | \n|  | +TEXT = torch.ones(4, dtype=torch.long) | \n|  | +FRAMES, H, W = 7, 8, 8 | \n|  | + | \n|  | + | \n|  | +def _layout(ref_blocks=(), aligned_refs=()): | \n|  | +    return build_packed_sequence( | \n|  | +        text_token_tags=TEXT, | \n|  | +        num_latent_frames=FRAMES, | \n|  | +        latent_height=H, | \n|  | +        latent_width=W, | \n|  | +        num_audio_latents=0, | \n|  | +        ref_blocks=ref_blocks, | \n|  | +        aligned_refs=aligned_refs, | \n|  | +    ) | \n|  | + | \n|  | + | \n|  | +def _rows(layout, start, count): | \n|  | +    return layout.position_ids[start : start + count] | \n|  | + | \n|  | + | \n|  | +def test_aligned_reference_matches_the_target_coordinates_row_for_row() -> None: | \n|  | +    layout = _layout(ref_blocks=((FRAMES, H, W),), aligned_refs=(True,)) | \n|  | + | \n|  | +    rows_per_ref = FRAMES * (H // 2) * (W // 2) | \n|  | +    guide = _rows(layout, layout.video_indices[0].item(), rows_per_ref) | \n|  | +    target = layout.position_ids[-rows_per_ref:] | \n|  | + | \n|  | +    assert torch.equal(guide, target) | \n|  | + | \n|  | + | \n|  | +def test_aligned_reference_does_not_shift_the_target_clock() -> None: | \n|  | +    \"\"\"The target must sit where it would with no reference at all.\"\"\" | \n|  | +    bare = _layout() | \n|  | +    aligned = _layout(ref_blocks=((FRAMES, H, W),), aligned_refs=(True,)) | \n|  | + | \n|  | +    rows = FRAMES * (H // 2) * (W // 2) | \n|  | +    assert torch.equal(bare.position_ids[-rows:], aligned.position_ids[-rows:]) | \n|  | + | \n|  | + | \n|  | +def test_unaligned_reference_still_sits_beside_the_target() -> None: | \n|  | +    \"\"\"The default path is untouched: an ordinary reference keeps its own clock.\"\"\" | \n|  | +    layout = _layout(ref_blocks=((1, H, W),), aligned_refs=(False,)) | \n|  | + | \n|  | +    rows_per_ref = (H // 2) * (W // 2) | \n|  | +    ref = _rows(layout, layout.video_indices[0].item(), rows_per_ref) | \n|  | +    target = layout.position_ids[-FRAMES * rows_per_ref :] | \n|  | + | \n|  | +    assert ref[:, 0].max() < target[:, 0].min() | \n|  | + | \n|  | + | \n|  | +def test_omitting_the_flag_keeps_upstream_behaviour() -> None: | \n|  | +    explicit = _layout(ref_blocks=((1, H, W),), aligned_refs=(False,)) | \n|  | +    implicit = _layout(ref_blocks=((1, H, W),)) | \n|  | + | \n|  | +    assert torch.equal(explicit.position_ids, implicit.position_ids) | \n|  | + | \n|  | + | \n|  | +def test_head_swap_pack_aligns_only_the_guide() -> None: | \n|  | +    \"\"\"Guide aligned to the target, identity reference parked beside it.\"\"\" | \n|  | +    layout = _layout(ref_blocks=((FRAMES, H, W), (1, H, W)), aligned_refs=(True, False)) | \n|  | + | \n|  | +    rows_per_frame = (H // 2) * (W // 2) | \n|  | +    guide_rows = FRAMES * rows_per_frame | \n|  | +    start = layout.video_indices[0].item() | \n|  | + | \n|  | +    guide = _rows(layout, start, guide_rows) | \n|  | +    identity = _rows(layout, start + guide_rows, rows_per_frame) | \n|  | +    target = layout.position_ids[-guide_rows:] | \n|  | + | \n|  | +    assert torch.equal(guide, target) | \n|  | +    assert identity[:, 0].max() < target[:, 0].min() | \n|  | + | \n|  | + | \n|  | +def test_shorter_aligned_guide_starts_at_the_targets_first_frame() -> None: | \n|  | +    short = 3 | \n|  | +    layout = _layout(ref_blocks=((short, H, W),), aligned_refs=(True,)) | \n|  | + | \n|  | +    rows_per_frame = (H // 2) * (W // 2) | \n|  | +    guide = _rows(layout, layout.video_indices[0].item(), short * rows_per_frame) | \n|  | +    target = layout.position_ids[-FRAMES * rows_per_frame :] | \n|  | + | \n|  | +    assert torch.equal(guide, target[: short * rows_per_frame]) | \n|  | + | \n|  | + | \n|  | +def test_aligned_reference_must_match_the_target_resolution() -> None: | \n|  | +    \"\"\"A different grid cannot be aligned, and silently misaligning is the worst outcome.\"\"\" | \n|  | +    with pytest.raises(ValueError, match=\"aligned reference\"): | \n|  | +        _layout(ref_blocks=((FRAMES, H // 2, W),), aligned_refs=(True,)) | \n|  | + | \n|  | + | \n|  | +class _FakeModel: | \n|  | +    \"\"\"Just enough of the model to exercise the flag's plumbing.\"\"\" | \n|  | + | \n|  | +    def __init__(self, enabled): | \n|  | +        self.model_config = type(\"C\", (), {\"model_kwargs\": {\"align_video_refs\": enabled}})() | \n|  | + | \n|  | +    _aligned_ref_flags = None  # bound below | \n|  | + | \n|  | + | \n|  | +def _flags(enabled, ref_blocks): | \n|  | +    from extensions_built_in.diffusion_models.minimax_h3.minimax_h3 import MinimaxH3Model | \n|  | + | \n|  | +    return MinimaxH3Model._aligned_ref_flags(_FakeModel(enabled), ref_blocks) | \n|  | + | \n|  | + | \n|  | +def test_flag_off_aligns_nothing() -> None: | \n|  | +    assert _flags(False, ((7, 8, 8, 0), (1, 8, 8, 0))) == () | \n|  | + | \n|  | + | \n|  | +def test_flag_on_aligns_video_blocks_only() -> None: | \n|  | +    \"\"\"Video reference becomes the guide; the image reference stays a reference.\"\"\" | \n|  | +    assert _flags(True, ((7, 8, 8, 0), (1, 8, 8, 0))) == (True, False) | \n|  | + | \n|  | + | \n|  | +def test_no_references_needs_no_flags() -> None: | \n|  | +    assert _flags(True, ()) == () | \n|  | diff --git a/toolkit/data_loader.py b/toolkit/data_loader.py | \n|  | index 38b7ad4..a94e369 100644 | \n|  | --- a/toolkit/data_loader.py | \n|  | +++ b/toolkit/data_loader.py | \n|  | @@ -540,6 +540,7 @@ class AiToolkitDataset(LatentCachingMixin, ControlCachingMixin, CLIPCachingMixin | \n|  | encode_control_in_text_embeddings=self.sd.encode_control_in_text_embeddings if self.sd else False, | \n|  | encode_first_frame_in_text_embeddings=getattr(self.sd, 'encode_first_frame_in_text_embeddings', False) if self.sd else False, | \n|  | dopsd_self_ref=getattr(self.sd, 'dopsd_self_ref', False) if self.sd else False, | \n|  | +                    control_latent_only=getattr(self.sd, 'control_latent_only', False) if self.sd else False, | \n|  | text_embedding_space_version=self.sd.text_embedding_space_version if self.sd else \"sd1\", | \n|  | te_padding_side=self.sd.te_padding_side if self.sd else \"right\", | \n|  | latent_space_version=latent_space_version, | \n|  | diff --git a/toolkit/data_transfer_object/data_loader.py b/toolkit/data_transfer_object/data_loader.py | \n|  | index 8c7dfc6..2c9b615 100644 | \n|  | --- a/toolkit/data_transfer_object/data_loader.py | \n|  | +++ b/toolkit/data_transfer_object/data_loader.py | \n|  | @@ -85,6 +85,8 @@ class FileItemDTO( | \n|  | ) | \n|  | # D-OPSD: also cache teacher embeds with the item's own media as reference 1 | \n|  | self.dopsd_self_ref = kwargs.get(\"dopsd_self_ref\", False) | \n|  | +        # control media rides as latents only, never into the text encoder | \n|  | +        self.control_latent_only = kwargs.get(\"control_latent_only\", False) | \n|  | self.te_padding_side = kwargs.get(\"te_padding_side\", \"right\") | \n|  | self.latent_space_version = kwargs.get(\"latent_space_version\", \"sd1\") | \n|  | self.text_embedding_space_version = kwargs.get(\"text_embedding_space_version\", \"sd1\") | \n|  | diff --git a/toolkit/dataloader_mixins.py b/toolkit/dataloader_mixins.py | \n|  | index 60e0a9e..ae6a1e5 100644 | \n|  | --- a/toolkit/dataloader_mixins.py | \n|  | +++ b/toolkit/dataloader_mixins.py | \n|  | @@ -2414,8 +2414,15 @@ class TextEmbeddingCachingMixin: | \n|  | ctrl_img = ctrl_img_list[0] | \n|  | else: | \n|  | ctrl_img = ctrl_img_list | \n|  | +                        if getattr(file_item, 'control_latent_only', False): | \n|  | +                            # caption-only embeds. The control | \n|  | +                            # media still reaches the DiT as aligned latents; what the | \n|  | +                            # student loses is the VLM's reading of it, which is exactly | \n|  | +                            # the information the teacher distills down. Also matches | \n|  | +                            # ComfyUI's AddGuide path, where the guide never hits the VLM. | \n|  | +                            ctrl_img = None | \n|  | for path, caption in encode_targets: | \n|  | -                            if path in dropout_target_paths: | \n|  | +                            if path in dropout_target_paths or ctrl_img is None: | \n|  | # dropout embeds are plain text. Only fall back to the | \n|  | # control images if the model cannot encode without them | \n|  | try: |", "url": "https://wpnews.pro/news/minimax-h3-ref2va-aligned-video-guides-ic-lora-style-v2v-for-ai-toolkit", "canonical_source": "https://gist.github.com/alisson-anjos/b300f2b90e65cf85846519d78b660cd3", "published_at": "2026-08-27 15:13:22+00:00", "updated_at": "2026-09-20 01:54:11.489293+00:00", "lang": "en", "topics": ["generative-ai", "ai-tools", "computer-vision", "developer-tools"], "entities": ["MiniMax-H3", "ai-toolkit", "Qwen3-VL", "IC-LoRA"], "also_reported_by": [], "alternates": {"html": "https://wpnews.pro/news/minimax-h3-ref2va-aligned-video-guides-ic-lora-style-v2v-for-ai-toolkit", "markdown": "https://wpnews.pro/news/minimax-h3-ref2va-aligned-video-guides-ic-lora-style-v2v-for-ai-toolkit.md", "text": "https://wpnews.pro/news/minimax-h3-ref2va-aligned-video-guides-ic-lora-style-v2v-for-ai-toolkit.txt", "jsonld": "https://wpnews.pro/news/minimax-h3-ref2va-aligned-video-guides-ic-lora-style-v2v-for-ai-toolkit.jsonld"}}