MiniMaxH3PipelineConfig(model_path: str = '', pipeline_config_path: str | None = None, embedded_cfg_scale: float | None = None, flow_shift: float | None = None, flow_shift_sr: float | None = None, disable_autocast: bool = False, scheduler_step_in_fp32: bool = False, is_causal: bool = False, dit_config: MiniMaxH3Config = MiniMaxH3Config(), dit_precision: str = 'bf16', upsampler_config: UpsamplerConfig = UpsamplerConfig(), upsampler_precision: str = 'fp32', vae_config: VAEConfig = MiniMaxH3VideoVAEConfig(), vae_precision: str = 'fp32', vae_decode_precision: str | None = None, vae_tiling: bool = True, vae_sp: bool = False, image_encoder_config: EncoderConfig = EncoderConfig(), image_encoder_precision: str = 'fp32', image_encoder_configs: tuple[EncoderConfig, ...] | None = None, image_encoder_precisions: tuple[str, ...] | None = None, text_encoder_configs: tuple[EncoderConfig, ...] = (lambda: (MiniMaxH3Qwen3VLConfig(),))(), text_encoder_precisions: tuple[str, ...] = (lambda: ('bf16',))(), preprocess_text_funcs: tuple[Callable[[str], str], ...] = (lambda: (preprocess_text,))(), postprocess_text_funcs: tuple[Callable[[BaseEncoderOutput], tensor], ...] = (lambda: (postprocess_text,))(), dmd_denoising_steps: list[int] | None = None, ti2v_task: bool = False, lucy_edit_task: bool = False, boundary_ratio: float | None = None, audio_vae_config: VAEConfig = MiniMaxH3AudioVAEConfig(), pdd_step_indices: tuple[int, ...] | None = None, vsa_ref_keep_rate: float | None = None)
Bases: PipelineConfig
Component and precision policy shared by T2VA, FL2VA, and Ref2VA.
Methods:
fastvideo.configs.pipelines.minimax_h3.MiniMaxH3PipelineConfig.apply_request_constraints
Fix num_inference_steps to the block count of a PDD checkpoint.
A PDD checkpoint runs exactly one transformer forward per trained fused block: a request that leaves num_inference_steps unset gets the block count, and a request that sets another count raises. Checkpoints without a PDD partition return sampling_param unchanged.
Source code in fastvideo/configs/pipelines/minimax_h3.py
| def apply_request_constraints(self, request: GenerationRequest, sampling_param: SamplingParam) -> SamplingParam:
"""Fix ``num_inference_steps`` to the block count of a PDD checkpoint.
A PDD checkpoint runs exactly one transformer forward per trained fused
block: a request that leaves ``num_inference_steps`` unset gets the block
count, and a request that sets another count raises. Checkpoints without
a PDD partition return ``sampling_param`` unchanged.
"""
if self.pdd_step_indices is None:
return sampling_param
from fastvideo.api.compat import explicit_request_updates
num_blocks = len(self.pdd_step_indices) - 1
requested_steps = explicit_request_updates(request).get("num_inference_steps", num_blocks)
if requested_steps != num_blocks:
raise ValueError(f"This FastH3 PDD checkpoint runs exactly {num_blocks} transformer forwards; the request "
f"sets num_inference_steps={requested_steps}. Pass num_inference_steps={num_blocks}. Only "
"a request parsed from a mapping or a config file can leave it unset; a "
"GenerationRequest built in Python counts every field as set.")
sampling_param.num_inference_steps = num_blocks
return sampling_param
|
fastvideo.configs.pipelines.minimax_h3.MiniMaxH3PipelineConfig.resolve_checkpoint_settings
resolve_checkpoint_settings(fastvideo_args: FastVideoArgs) -> None
Apply a FastH3 Ref2VA PDD checkpoint's fastvideo_inference.json to this run.
The file is the one source of the PDD partition and the trained VSA settings, and a setting the run leaves unset takes the file's value. The partition, attention backend and tile size are fixed by training, so a different run value raises; VSA sparsity and the reference keep rate may differ, with a warning. For checkpoints without PDD fields (base MiniMax-H3, DMD exports) this method sets nothing.
Source code in fastvideo/configs/pipelines/minimax_h3.py
| def resolve_checkpoint_settings(self, fastvideo_args: FastVideoArgs) -> None:
"""Apply a FastH3 Ref2VA PDD checkpoint's fastvideo_inference.json to this run.
The file is the one source of the PDD partition and the trained VSA
settings, and a setting the run leaves unset takes the file's value.
The partition, attention backend and tile size are fixed by training,
so a different run value raises; VSA sparsity and the reference keep
rate may differ, with a warning. For checkpoints without PDD fields
(base MiniMax-H3, DMD exports) this method sets nothing.
"""
inference_file = _read_pdd_inference_file(fastvideo_args.model_path, fastvideo_args.revision)
if inference_file is None:
for name in ("pdd_step_indices", "vsa_ref_keep_rate"):
if getattr(self, name) is not None:
raise ValueError(f"{name} applies only to FastH3 PDD checkpoints, whose {FASTH3_INFERENCE_FILE} "
f"sets it; {fastvideo_args.model_path} has no PDD fields.")
return
# Training-fixed settings: an unset run value takes the file's value; a different one raises.
from fastvideo.attention.selector import coerce_attn_backend
trained_backend = inference_file["attention_backend"]
if fastvideo_args.attention_backend is None:
fastvideo_args.attention_backend = trained_backend
elif coerce_attn_backend(fastvideo_args.attention_backend) != coerce_attn_backend(trained_backend):
raise ValueError(f"This FastH3 PDD checkpoint was trained with attention_backend={trained_backend}; "
f"this run requests {fastvideo_args.attention_backend} (argument or "
"FASTVIDEO_ATTENTION_BACKEND). Leave it unset or pass the trained value.")
trained_tile = inference_file["vsa_tile_size"]
if fastvideo_args.VSA_tile_size is None:
fastvideo_args.VSA_tile_size = trained_tile
elif fastvideo_args.VSA_tile_size != trained_tile:
raise ValueError(f"This FastH3 PDD checkpoint was trained with VSA_tile_size={trained_tile}; this run "
f"requests {fastvideo_args.VSA_tile_size}. Leave it unset or pass the trained value.")
indices = tuple(inference_file["pdd_step_indices"])
if self.pdd_step_indices is not None and tuple(self.pdd_step_indices) != indices:
raise ValueError(f"pdd_step_indices comes from {FASTH3_INFERENCE_FILE} ({list(indices)}); this run sets "
f"{list(self.pdd_step_indices)}.")
if self.dmd_denoising_steps is not None:
raise ValueError(f"{fastvideo_args.model_path} is a FastH3 PDD checkpoint; dmd_denoising_steps must be "
"unset.")
self.pdd_step_indices = indices
# Tunable settings: an unset run value takes the trained value; a different one wins, with a warning.
trained_sparsity = inference_file["vsa_sparsity"]
if fastvideo_args.VSA_sparsity is None:
fastvideo_args.VSA_sparsity = trained_sparsity
elif fastvideo_args.VSA_sparsity != trained_sparsity:
_require_fraction("VSA_sparsity", fastvideo_args.VSA_sparsity, allow_zero=True)
logger.warning("FastH3 PDD checkpoint was trained with VSA_sparsity=%s; this run uses %s.",
trained_sparsity, fastvideo_args.VSA_sparsity)
trained_keep_rate = inference_file["vsa_ref_keep_rate"]
if self.vsa_ref_keep_rate is None:
self.vsa_ref_keep_rate = trained_keep_rate
elif self.vsa_ref_keep_rate != trained_keep_rate:
_require_fraction("vsa_ref_keep_rate", self.vsa_ref_keep_rate, allow_zero=False)
logger.warning("FastH3 PDD checkpoint was trained with vsa_ref_keep_rate=%s; this run uses %s.",
trained_keep_rate, self.vsa_ref_keep_rate)
logger.info(
"FastH3 PDD checkpoint %s: %d fused blocks %s of a %d-interval grid; attention_backend=%s, "
"VSA_tile_size=%s, VSA_sparsity=%s, vsa_ref_keep_rate=%s", fastvideo_args.model_path,
len(indices) - 1, list(indices), indices[-1], fastvideo_args.attention_backend,
fastvideo_args.VSA_tile_size, fastvideo_args.VSA_sparsity, self.vsa_ref_keep_rate)
|