Skip to content

minimax_h3

Pipeline configuration for MiniMax H3 joint video/audio generation.

Classes

fastvideo.configs.pipelines.minimax_h3.MiniMaxH3PipelineConfig dataclass

MiniMaxH3PipelineConfig(model_path: str = '', pipeline_config_path: str | None = None, embedded_cfg_scale: float | None = None, flow_shift: float | None = None, flow_shift_sr: float | None = None, disable_autocast: bool = False, scheduler_step_in_fp32: bool = False, is_causal: bool = False, dit_config: MiniMaxH3Config = MiniMaxH3Config(), dit_precision: str = 'bf16', upsampler_config: UpsamplerConfig = UpsamplerConfig(), upsampler_precision: str = 'fp32', vae_config: VAEConfig = MiniMaxH3VideoVAEConfig(), vae_precision: str = 'fp32', vae_decode_precision: str | None = None, vae_tiling: bool = True, vae_sp: bool = False, image_encoder_config: EncoderConfig = EncoderConfig(), image_encoder_precision: str = 'fp32', image_encoder_configs: tuple[EncoderConfig, ...] | None = None, image_encoder_precisions: tuple[str, ...] | None = None, text_encoder_configs: tuple[EncoderConfig, ...] = (lambda: (MiniMaxH3Qwen3VLConfig(),))(), text_encoder_precisions: tuple[str, ...] = (lambda: ('bf16',))(), preprocess_text_funcs: tuple[Callable[[str], str], ...] = (lambda: (preprocess_text,))(), postprocess_text_funcs: tuple[Callable[[BaseEncoderOutput], tensor], ...] = (lambda: (postprocess_text,))(), dmd_denoising_steps: list[int] | None = None, ti2v_task: bool = False, lucy_edit_task: bool = False, boundary_ratio: float | None = None, audio_vae_config: VAEConfig = MiniMaxH3AudioVAEConfig(), pdd_step_indices: tuple[int, ...] | None = None, vsa_ref_keep_rate: float | None = None)

Bases: PipelineConfig

Component and precision policy shared by T2VA, FL2VA, and Ref2VA.

Methods:

fastvideo.configs.pipelines.minimax_h3.MiniMaxH3PipelineConfig.apply_request_constraints
apply_request_constraints(request: GenerationRequest, sampling_param: SamplingParam) -> SamplingParam

Fix num_inference_steps to the block count of a PDD checkpoint.

A PDD checkpoint runs exactly one transformer forward per trained fused block: a request that leaves num_inference_steps unset gets the block count, and a request that sets another count raises. Checkpoints without a PDD partition return sampling_param unchanged.

Source code in fastvideo/configs/pipelines/minimax_h3.py
def apply_request_constraints(self, request: GenerationRequest, sampling_param: SamplingParam) -> SamplingParam:
    """Fix ``num_inference_steps`` to the block count of a PDD checkpoint.

    A PDD checkpoint runs exactly one transformer forward per trained fused
    block: a request that leaves ``num_inference_steps`` unset gets the block
    count, and a request that sets another count raises. Checkpoints without
    a PDD partition return ``sampling_param`` unchanged.
    """
    if self.pdd_step_indices is None:
        return sampling_param
    from fastvideo.api.compat import explicit_request_updates

    num_blocks = len(self.pdd_step_indices) - 1
    requested_steps = explicit_request_updates(request).get("num_inference_steps", num_blocks)
    if requested_steps != num_blocks:
        raise ValueError(f"This FastH3 PDD checkpoint runs exactly {num_blocks} transformer forwards; the request "
                         f"sets num_inference_steps={requested_steps}. Pass num_inference_steps={num_blocks}. Only "
                         "a request parsed from a mapping or a config file can leave it unset; a "
                         "GenerationRequest built in Python counts every field as set.")
    sampling_param.num_inference_steps = num_blocks
    return sampling_param
fastvideo.configs.pipelines.minimax_h3.MiniMaxH3PipelineConfig.resolve_checkpoint_settings
resolve_checkpoint_settings(fastvideo_args: FastVideoArgs) -> None

Apply a FastH3 Ref2VA PDD checkpoint's fastvideo_inference.json to this run.

The file is the one source of the PDD partition and the trained VSA settings, and a setting the run leaves unset takes the file's value. The partition, attention backend and tile size are fixed by training, so a different run value raises; VSA sparsity and the reference keep rate may differ, with a warning. For checkpoints without PDD fields (base MiniMax-H3, DMD exports) this method sets nothing.

Source code in fastvideo/configs/pipelines/minimax_h3.py
def resolve_checkpoint_settings(self, fastvideo_args: FastVideoArgs) -> None:
    """Apply a FastH3 Ref2VA PDD checkpoint's fastvideo_inference.json to this run.

    The file is the one source of the PDD partition and the trained VSA
    settings, and a setting the run leaves unset takes the file's value.
    The partition, attention backend and tile size are fixed by training,
    so a different run value raises; VSA sparsity and the reference keep
    rate may differ, with a warning. For checkpoints without PDD fields
    (base MiniMax-H3, DMD exports) this method sets nothing.
    """
    inference_file = _read_pdd_inference_file(fastvideo_args.model_path, fastvideo_args.revision)
    if inference_file is None:
        for name in ("pdd_step_indices", "vsa_ref_keep_rate"):
            if getattr(self, name) is not None:
                raise ValueError(f"{name} applies only to FastH3 PDD checkpoints, whose {FASTH3_INFERENCE_FILE} "
                                 f"sets it; {fastvideo_args.model_path} has no PDD fields.")
        return

    # Training-fixed settings: an unset run value takes the file's value; a different one raises.
    from fastvideo.attention.selector import coerce_attn_backend
    trained_backend = inference_file["attention_backend"]
    if fastvideo_args.attention_backend is None:
        fastvideo_args.attention_backend = trained_backend
    elif coerce_attn_backend(fastvideo_args.attention_backend) != coerce_attn_backend(trained_backend):
        raise ValueError(f"This FastH3 PDD checkpoint was trained with attention_backend={trained_backend}; "
                         f"this run requests {fastvideo_args.attention_backend} (argument or "
                         "FASTVIDEO_ATTENTION_BACKEND). Leave it unset or pass the trained value.")
    trained_tile = inference_file["vsa_tile_size"]
    if fastvideo_args.VSA_tile_size is None:
        fastvideo_args.VSA_tile_size = trained_tile
    elif fastvideo_args.VSA_tile_size != trained_tile:
        raise ValueError(f"This FastH3 PDD checkpoint was trained with VSA_tile_size={trained_tile}; this run "
                         f"requests {fastvideo_args.VSA_tile_size}. Leave it unset or pass the trained value.")
    indices = tuple(inference_file["pdd_step_indices"])
    if self.pdd_step_indices is not None and tuple(self.pdd_step_indices) != indices:
        raise ValueError(f"pdd_step_indices comes from {FASTH3_INFERENCE_FILE} ({list(indices)}); this run sets "
                         f"{list(self.pdd_step_indices)}.")
    if self.dmd_denoising_steps is not None:
        raise ValueError(f"{fastvideo_args.model_path} is a FastH3 PDD checkpoint; dmd_denoising_steps must be "
                         "unset.")
    self.pdd_step_indices = indices

    # Tunable settings: an unset run value takes the trained value; a different one wins, with a warning.
    trained_sparsity = inference_file["vsa_sparsity"]
    if fastvideo_args.VSA_sparsity is None:
        fastvideo_args.VSA_sparsity = trained_sparsity
    elif fastvideo_args.VSA_sparsity != trained_sparsity:
        _require_fraction("VSA_sparsity", fastvideo_args.VSA_sparsity, allow_zero=True)
        logger.warning("FastH3 PDD checkpoint was trained with VSA_sparsity=%s; this run uses %s.",
                       trained_sparsity, fastvideo_args.VSA_sparsity)
    trained_keep_rate = inference_file["vsa_ref_keep_rate"]
    if self.vsa_ref_keep_rate is None:
        self.vsa_ref_keep_rate = trained_keep_rate
    elif self.vsa_ref_keep_rate != trained_keep_rate:
        _require_fraction("vsa_ref_keep_rate", self.vsa_ref_keep_rate, allow_zero=False)
        logger.warning("FastH3 PDD checkpoint was trained with vsa_ref_keep_rate=%s; this run uses %s.",
                       trained_keep_rate, self.vsa_ref_keep_rate)

    logger.info(
        "FastH3 PDD checkpoint %s: %d fused blocks %s of a %d-interval grid; attention_backend=%s, "
        "VSA_tile_size=%s, VSA_sparsity=%s, vsa_ref_keep_rate=%s", fastvideo_args.model_path,
        len(indices) - 1, list(indices), indices[-1], fastvideo_args.attention_backend,
        fastvideo_args.VSA_tile_size, fastvideo_args.VSA_sparsity, self.vsa_ref_keep_rate)

Functions:

fastvideo.configs.pipelines.minimax_h3.parse_base_model_revision

parse_base_model_revision(value: Any) -> tuple[str, str]

Split a contract's base_model_revision, hf://<repo id>@<revision>, into repo id and revision.

Source code in fastvideo/configs/pipelines/minimax_h3.py
def parse_base_model_revision(value: Any) -> tuple[str, str]:
    """Split a contract's ``base_model_revision``, ``hf://<repo id>@<revision>``, into repo id and revision."""
    if isinstance(value, str) and value.startswith(_BASE_MODEL_REVISION_PREFIX):
        repo, separator, revision = value[len(_BASE_MODEL_REVISION_PREFIX):].partition("@")
        if separator and repo.strip() and revision.strip() and "@" not in revision:
            return repo, revision
    raise ValueError(f"FastH3 base_model_revision={value!r} must be hf://<repo id>@<revision>.")