Skip to content

pinned_memory

Exact-size pinned host storage for inference offload.

Classes

fastvideo.hooks.pinned_memory.PinnedTensorArena

PinnedTensorArena(tensors: Iterable[tuple[str, Tensor]])

Pack tensors into one CUDA-registered CPU allocation, aligned to 256 bytes.

PyTorch's pinned allocator rounds large allocations to powers of two. Registering ordinary host storage avoids that overhead. Typed views retain this owner, so registration survives even if the module or offload state is dropped first. If registration is unavailable, allocate conventional pinned tensors instead. Call close only after all views and pending copies have been released.

Source code in fastvideo/hooks/pinned_memory.py
def __init__(self, tensors: Iterable[tuple[str, torch.Tensor]]) -> None:
    self.offsets: dict[str, tuple[int, int]] = {}
    size = 0
    for name, tensor in tensors:
        size = (size + _ALIGNMENT - 1) // _ALIGNMENT * _ALIGNMENT
        length = tensor.numel() * tensor.element_size()
        self.offsets[name] = (size, length)
        size += length
    self.nbytes = size
    self.buffer: torch.Tensor | None = None
    self._finalizer: weakref.finalize | None = None
    if not size:
        return
    # Register dedicated pages: small malloc allocations can otherwise share a
    # registered page with another arena. The extra space is bounded by 8 KiB.
    span = (size + _PAGE_ALIGNMENT - 1) // _PAGE_ALIGNMENT * _PAGE_ALIGNMENT
    allocation = torch.empty(span + _PAGE_ALIGNMENT - 1, dtype=torch.uint8, device="cpu")
    start = (-allocation.data_ptr()) % _PAGE_ALIGNMENT
    # Give the aligned region its own storage base. Tensor.is_pinned() queries
    # the storage pointer, which would precede the registered region for a
    # plain narrow() view. The memoryview retains the original allocation.
    buffer = torch.frombuffer(memoryview(allocation.numpy())[start:start + span], dtype=torch.uint8)
    device = torch.cuda.current_device()
    try:
        error = torch.cuda.cudart().cudaHostRegister(buffer.data_ptr(), span, 0)
        if error != 0:
            raise RuntimeError(f"cudaHostRegister returned {error}")
    except Exception as exc:
        logger.warning("Exact-size host registration failed; using the pinned allocator: %s", exc)
        return
    self.buffer = buffer
    self._finalizer = weakref.finalize(self, _unregister, buffer, device)

Methods:

fastvideo.hooks.pinned_memory.PinnedTensorArena.close
close() -> None

Unregister before releasing storage; safe to call more than once.

Source code in fastvideo/hooks/pinned_memory.py
def close(self) -> None:
    """Unregister before releasing storage; safe to call more than once."""
    if self._finalizer is not None:
        self._finalizer()
        self._finalizer = None
    self.buffer = None
fastvideo.hooks.pinned_memory.PinnedTensorArena.empty_like
empty_like(name: str, tensor: Tensor) -> Tensor

Return a contiguous host view with the source's dtype and shape.

Source code in fastvideo/hooks/pinned_memory.py
def empty_like(self, name: str, tensor: torch.Tensor) -> torch.Tensor:
    """Return a contiguous host view with the source's dtype and shape."""
    if self.buffer is None:
        return torch.empty(tensor.shape, dtype=tensor.dtype, device="cpu", pin_memory=True)
    offset, length = self.offsets[name]
    host = self.buffer.narrow(0, offset, length).view(tensor.dtype).reshape(tensor.shape)
    host._pinned_arena = self
    return host

Functions: