Skip to content

ref2va_manifest

Raw dataset manifest support for MiniMax H3 Ref2VA training and validation.

Classes

fastvideo.pipelines.basic.minimax_h3.ref2va_manifest.MiniMaxH3RawReference dataclass

MiniMaxH3RawReference(media_type: RawReferenceType, image_path: Path | None = None, video_path: Path | None = None, audio_path: Path | None = None, fps: float | None = None, sample_rate: int | None = None)

One validated, ordered reference entry from a raw dataset manifest.

fastvideo.pipelines.basic.minimax_h3.ref2va_manifest.MiniMaxH3Ref2VARawSample dataclass

MiniMaxH3Ref2VARawSample(sample_id: str, target_file: str, target_video_path: Path, caption: str, references: tuple[MiniMaxH3RawReference, ...])

One validated raw Ref2VA sample with paths resolved against its manifest.

Functions:

fastvideo.pipelines.basic.minimax_h3.ref2va_manifest.build_minimax_h3_references

build_minimax_h3_references(references: tuple[MiniMaxH3RawReference, ...] | list[MiniMaxH3RawReference]) -> list[MiniMaxH3Reference]

Convert raw reference specs while preserving their declared order and audio semantics.

Source code in fastvideo/pipelines/basic/minimax_h3/ref2va_manifest.py
def build_minimax_h3_references(
    references: tuple[MiniMaxH3RawReference, ...] | list[MiniMaxH3RawReference], ) -> list[MiniMaxH3Reference]:
    """Convert raw reference specs while preserving their declared order and audio semantics."""
    converted: list[MiniMaxH3Reference] = []
    for index, reference in enumerate(references):
        if reference.media_type == "image":
            if reference.image_path is None:
                raise RuntimeError(f"Validated image reference {index} has no image path")
            converted.append(MiniMaxH3Reference(source=reference.image_path, media_type="image"))
            continue
        if reference.media_type == "audio":
            if reference.audio_path is None:
                raise RuntimeError(f"Validated audio reference {index} has no audio path")
            converted.append(
                MiniMaxH3Reference(
                    source=reference.audio_path,
                    media_type="audio",
                    sample_rate=reference.sample_rate,
                ))
            continue
        if reference.video_path is None:
            raise RuntimeError(f"Validated video reference {index} has no video path")
        if reference.media_type == "video":
            # A raw type=video is explicitly silent even if its container has
            # a soundtrack. Passing pixels avoids prepare_reference adopting it.
            frames, decoded_fps, decoded_soundtrack = decode_reference_video(reference.video_path)
            del decoded_soundtrack
            converted.append(
                MiniMaxH3Reference(
                    source=frames,
                    media_type="video",
                    fps=reference.fps if reference.fps is not None else decoded_fps,
                ))
            continue
        if reference.audio_path is None:
            raise RuntimeError(f"Validated video_audio reference {index} has no audio path")
        converted.append(
            MiniMaxH3Reference(
                source=reference.video_path,
                media_type="video",
                soundtrack=None if reference.video_path == reference.audio_path else reference.audio_path,
                fps=reference.fps,
                sample_rate=reference.sample_rate,
            ))

    return validate_references(converted)

fastvideo.pipelines.basic.minimax_h3.ref2va_manifest.load_minimax_h3_ref2va_raw_samples

load_minimax_h3_ref2va_raw_samples(manifest_path: Path) -> list[MiniMaxH3Ref2VARawSample]

Load pretty JSON, JSON arrays/wrappers, or standard one-object-per-line JSONL.

Source code in fastvideo/pipelines/basic/minimax_h3/ref2va_manifest.py
def load_minimax_h3_ref2va_raw_samples(manifest_path: Path) -> list[MiniMaxH3Ref2VARawSample]:
    """Load pretty JSON, JSON arrays/wrappers, or standard one-object-per-line JSONL."""
    manifest_path = manifest_path.expanduser().resolve()
    if not manifest_path.is_file():
        raise FileNotFoundError(f"MiniMax H3 raw manifest is missing at {manifest_path}")
    text = manifest_path.read_text(encoding="utf-8")
    if not text.strip():
        raise ValueError(f"MiniMax H3 raw manifest is empty at {manifest_path}")

    try:
        records = _records_from_json_document(json.loads(text), manifest_path)
    except json.JSONDecodeError:
        records = []
        for line_number, line in enumerate(text.splitlines(), start=1):
            if not line.strip():
                continue
            try:
                records.append(json.loads(line))
            except json.JSONDecodeError as error:
                raise ValueError(f"Manifest {manifest_path} is neither valid JSON nor one-object-per-line JSONL; "
                                 f"line {line_number} is invalid") from error

    if not records:
        raise ValueError(f"MiniMax H3 raw manifest contains no samples at {manifest_path}")
    samples = [parse_minimax_h3_ref2va_raw_sample(record, manifest_path=manifest_path) for record in records]
    sample_ids = [sample.sample_id for sample in samples]
    duplicate_ids = sorted(sample_id for sample_id, count in Counter(sample_ids).items() if count > 1)
    if duplicate_ids:
        raise ValueError(f"Manifest {manifest_path} contains duplicate sample ids: {duplicate_ids}")
    return samples

fastvideo.pipelines.basic.minimax_h3.ref2va_manifest.parse_minimax_h3_ref2va_raw_sample

parse_minimax_h3_ref2va_raw_sample(record: Any, *, manifest_path: Path, allow_extra_fields: bool = False) -> MiniMaxH3Ref2VARawSample

Validate one raw record and resolve every media path.

Source code in fastvideo/pipelines/basic/minimax_h3/ref2va_manifest.py
def parse_minimax_h3_ref2va_raw_sample(
    record: Any,
    *,
    manifest_path: Path,
    allow_extra_fields: bool = False,
) -> MiniMaxH3Ref2VARawSample:
    """Validate one raw record and resolve every media path."""
    manifest_path = manifest_path.expanduser().resolve()
    if not isinstance(record, dict):
        raise TypeError(f"A record in {manifest_path} must be a JSON object")

    required_fields = {"schema_version", "id", "target", "caption", "references"}
    record_keys = _non_null_keys(record)
    missing = required_fields - record_keys
    extra = record_keys - required_fields
    if missing or (extra and not allow_extra_fields):
        raise ValueError(f"A record in {manifest_path} requires exactly {sorted(required_fields)}; "
                         f"missing={sorted(missing)}, extra={sorted(extra)}")
    if record["schema_version"] != MINIMAX_H3_REF2VA_RAW_SCHEMA_VERSION:
        raise ValueError(f"Unsupported raw schema {record['schema_version']!r}; "
                         f"expected {MINIMAX_H3_REF2VA_RAW_SCHEMA_VERSION!r}")

    sample_id = record["id"]
    if not isinstance(sample_id, str) or not sample_id.strip():
        raise TypeError("Raw Ref2VA field 'id' must be a non-empty string")
    sample_id = sample_id.strip()

    target = record["target"]
    if not isinstance(target, dict) or _non_null_keys(target) != {"video"}:
        raise ValueError(f"Sample {sample_id!r} target must contain exactly one non-null 'video' field")
    target_file, target_video_path = _resolve_media_path(
        target["video"],
        manifest_path=manifest_path,
        field="video",
        context=f"Sample {sample_id!r} target",
    )

    caption = record["caption"]
    if not isinstance(caption, str) or not caption.strip():
        raise TypeError(f"Sample {sample_id!r} caption must be a non-empty string")
    caption = caption.strip()

    raw_references = record["references"]
    if not isinstance(raw_references, list):
        raise TypeError(f"Sample {sample_id!r} references must be an ordered list")
    references = tuple(
        _parse_reference(
            entry,
            manifest_path=manifest_path,
            sample_id=sample_id,
            index=index,
        ) for index, entry in enumerate(raw_references))

    return MiniMaxH3Ref2VARawSample(
        sample_id=sample_id,
        target_file=target_file,
        target_video_path=target_video_path,
        caption=caption,
        references=references,
    )