Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
6 changes: 3 additions & 3 deletions container/context.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -54,9 +54,9 @@ vllm:
base_image: nvcr.io/nvidia/cuda-dl-base
runtime_image: vllm/vllm-openai
base_image_tag: 25.11-cuda13.0-devel-ubuntu24.04
runtime_image_tag: v0.27.1-ubuntu2404
runtime_image_tag: v0.29.0-ubuntu2404
# Keep using the existing CUDA 13.0.2 compliance corpus for this Dingo CI
# image update. Refreshing the corpus for vLLM 0.27.1's CUDA 13.0.3 base is
# image update. Refreshing the corpus for the vLLM 0.29 runtime base is
# intentionally deferred because this workflow does not publish SBOM or
# provenance artifacts. Per-arch stem; the licenses stage appends
# -${TARGETARCH}.cdx.json.
Expand All @@ -74,7 +74,7 @@ vllm:
runtime_image_tag: v0.24.0
# baseline_sbom: not yet captured for cpu — runtime build runs without subtraction
flashinf_ref: v0.6.16.post3
vllm_omni_ref: "v0.27.0rc1"
vllm_omni_ref: "v0.29.0rc1"
nixl_ref: v1.3.1
max_jobs: "10"
enable_media_ffmpeg: "false"
Expand Down
7 changes: 6 additions & 1 deletion container/deps/vllm/install_vllm_omni.sh
Original file line number Diff line number Diff line change
Expand Up @@ -17,7 +17,7 @@ cleanup() {

trap cleanup EXIT

python3 - "${VLLM_OMNI_PROTECTED_PACKAGES_FILE}" <<'PY' > "${PROTECTED_CONSTRAINTS}"
python3 - "${VLLM_OMNI_PROTECTED_PACKAGES_FILE}" "${VLLM_OMNI_VERSION}" <<'PY' > "${PROTECTED_CONSTRAINTS}"
import importlib.metadata as md
from pathlib import Path
import sys
Expand All @@ -26,6 +26,11 @@ for raw_line in Path(sys.argv[1]).read_text().splitlines():
name = raw_line.strip()
if not name or name.startswith("#"):
continue
# Omni 0.29.0rc1 requires transformers>=5.13,<5.15, while the pinned
# upstream vLLM 0.29 image ships 5.16.1. Let Omni resolve this pure-Python
# API stack and its paired tokenizer, retaining the compiled core pins.
if sys.argv[2] == "0.29.0rc1" and name in {"transformers", "tokenizers"}:
continue
try:
dist = md.distribution(name)
except Exception:
Expand Down
118 changes: 118 additions & 0 deletions container/deps/vllm/validate_media_probe.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,118 @@
"""CPU-only smoke for the upstream FFmpeg path required by Ref2VA."""

import json
import shutil
import subprocess
import tempfile
from pathlib import Path


def main():
for executable in ("ffmpeg", "ffprobe"):
path = shutil.which(executable)
if not path:
raise RuntimeError(f"Missing upstream {executable}")
if Path(path).resolve().is_relative_to(Path("/usr/local")):
raise RuntimeError(
f"Expected upstream {executable}, found in-tree replacement {path!r}"
)
subprocess.run([path, "-version"], check=True, capture_output=True, timeout=15)

encoders = subprocess.run(
["ffmpeg", "-hide_banner", "-encoders"],
check=True,
capture_output=True,
text=True,
timeout=15,
).stdout
if "libx264rgb" not in encoders:
raise RuntimeError("Upstream FFmpeg is missing the libx264rgb encoder")

with tempfile.TemporaryDirectory(prefix="dingo-media-probe-") as directory:
directory_path = Path(directory)
source = directory_path / "source.rgb"
encoded = directory_path / "prepared.mp4"
decoded = directory_path / "decoded.rgb"
frame = bytes((index % 251 for index in range(32 * 32 * 3)))
source.write_bytes(frame * 4)
subprocess.run(
[
"ffmpeg",
"-y",
"-loglevel",
"error",
"-f",
"rawvideo",
"-pix_fmt",
"rgb24",
"-s",
"32x32",
"-r",
"8",
"-i",
str(source),
"-frames:v",
"4",
"-c:v",
"libx264rgb",
"-crf",
"0",
"-preset",
"veryfast",
"-pix_fmt",
"rgb24",
str(encoded),
],
check=True,
timeout=30,
)
subprocess.run(
[
"ffmpeg",
"-y",
"-loglevel",
"error",
"-i",
str(encoded),
"-f",
"rawvideo",
"-pix_fmt",
"rgb24",
str(decoded),
],
check=True,
timeout=30,
)
if decoded.read_bytes() != source.read_bytes():
raise RuntimeError("libx264rgb lossless round trip changed RGB pixels")
result = subprocess.run(
[
"ffprobe",
"-v",
"error",
"-count_frames",
"-show_streams",
"-show_format",
"-of",
"json",
str(encoded),
],
capture_output=True,
text=True,
check=True,
timeout=30,
)
document = json.loads(result.stdout)
streams = {stream["codec_type"]: stream for stream in document["streams"]}
assert streams["video"]["codec_name"] == "h264"
assert streams["video"]["pix_fmt"] == "gbrp"
assert (streams["video"]["width"], streams["video"]["height"]) == (32, 32)
assert int(streams["video"]["nb_read_frames"]) == 4
assert float(document["format"]["duration"]) > 0
print(
"DINGO_MEDIA_PROBE=PASS (upstream libx264rgb + rawvideo + ffprobe)"
)


if __name__ == "__main__":
main()
69 changes: 11 additions & 58 deletions container/templates/vllm_runtime.Dockerfile
Original file line number Diff line number Diff line change
Expand Up @@ -231,69 +231,22 @@ RUN --mount=type=cache,target=/root/.cache/uv,sharing=locked \
{% endif %}

{% if device == "cuda" %}
# The upstream vllm/vllm-openai base image ships a GPL/GPL-3.0 ffmpeg built
# against libx264/libx265/libmp3lame. Purge ONLY the explicitly-named ffmpeg +
# codec packages and replace them with the LGPL-only in-tree ffmpeg built in
# wheel_builder (--disable-gpl --disable-nonfree; H.264 via NVENC, VP9 via
# libvpx). PyAV, torchaudio, torchvision, soundfile and Pillow all bundle their
# own libraries and do not link the system ffmpeg/codecs, so removing them is
# safe. dpkg-query keeps the match robust across base-image/arch version
# suffixes (e.g. libavcodec58 vs 60).
#
# This grep is the COMPLETE, auditable set of what leaves the image: there is
# deliberately NO apt-get autoremove, so the removal can never cascade into
# unrelated auto-installed packages. That matters because the base image marks
# both the gcc/g++/make toolchain (torch.inductor/Triton JIT shell out to it at
# runtime) and the CUDA math libs (libcublas/libcusolver/libcusparse — the torch
# wheels here ship no bundled cublas and load the system copies) as
# auto-installed. A bare `autoremove --purge` sweeps all of those as "orphaned",
# which broke runtime JIT (missing C compiler) in the 1.3.0 rc image. Any
# LGPL/BSD media libs left orphaned (libva, libvdpau, ...) are license-clean
# dead weight, not a compliance issue.
RUN set -eux; \
purge=$(dpkg-query -W -f='${Package}\n' 2>/dev/null \
| grep -E '^(ffmpeg|libav[a-z]|libsw[a-z]|libpostproc|libx264|libx265|libmp3lame|libaom|libdav1d|libvpx|libtheora|libvorbis|libopus|libsoxr|libcaca|libcdio|libzvbi|libgme|libvidstab|libdc1394|libraw1394|libiec61883|libtwolame|libshine|libsrt[0-9]|libudfread|libsvtav1|libbs2b|librubberband|libchromaprint|libcodec2|libgsm|libass[0-9]|libbluray|libxvidcore|libflite)' \
|| true); \
if [ -n "$purge" ]; then \
DEBIAN_FRONTEND=noninteractive apt-get purge -y $purge; \
fi; \
rm -rf /var/lib/apt/lists/*

# Regression guard for the codec purge above: torch.inductor/Triton JIT shell
# out to a host C/C++ compiler at runtime, so a missing toolchain only surfaces
# on the first compile in production. Reproduce that compile path at build time
# (CPU-only) so a missing compiler aborts the build instead of shipping.
# Preserve the upstream vllm/vllm-openai FFmpeg stack. vLLM-Omni's Ref2VA
# preprocessing requires its software libx264rgb encoder and rawvideo support;
# replacing it with the reduced in-tree FFmpeg silently breaks video-reference
# requests. The runtime probe below exercises the same lossless RGB path.

# TorchInductor/Triton JIT shells out to a host C/C++ compiler at runtime.
# Reproduce that compile path at build time so a missing compiler aborts the
# build instead of surfacing on the first production request.
RUN --mount=type=bind,source=./container/deps/vllm/validate_torch_compile_smoke.py,target=/tmp/validate_torch_compile_smoke.py,readonly \
python3 /tmp/validate_torch_compile_smoke.py

# Copy the LGPL ffmpeg from wheel_builder: versioned shared libs (libav*.so*,
# libsw*.so*) + libvpx + the LGPL CLI binary that imageio/diffusers target via
# IMAGEIO_FFMPEG_EXE. Ungated by enable_media_ffmpeg because the base GPL ffmpeg
# was just purged, so the LGPL CLI must always be present for the omni
# video-export path to have something to encode with.
RUN --mount=type=bind,from=wheel_builder,source=/usr/local/,target=/tmp/usr/local/ \
mkdir -p /usr/local/lib/pkgconfig && \
cp -rnL /tmp/usr/local/include/libav* /tmp/usr/local/include/libsw* /usr/local/include/ && \
cp -nL /tmp/usr/local/lib/libav*.so* /tmp/usr/local/lib/libsw*.so* /usr/local/lib/ && \
cp -nL /tmp/usr/local/lib/lib*vpx*.so* /usr/local/lib/ 2>/dev/null || true && \
cp -nL /tmp/usr/local/lib/pkgconfig/libav*.pc /tmp/usr/local/lib/pkgconfig/libsw*.pc /usr/local/lib/pkgconfig/ && \
cp -nL /tmp/usr/local/bin/ffmpeg /usr/local/bin/ffmpeg && \
cp -r /tmp/usr/local/src/ffmpeg /usr/local/src/ && \
ldconfig
ENV IMAGEIO_FFMPEG_EXE=/usr/local/bin/ffmpeg
# Guard the upstream media tools and the exact Ref2VA codec path.
RUN --mount=type=bind,source=./container/deps/vllm/validate_media_probe.py,target=/tmp/validate_media_probe.py,readonly \
python3 /tmp/validate_media_probe.py
{% endif %}

# Replace the upstream vllm/vllm-openai image's imageio-ffmpeg (which ships a
# GPL-encumbered prebuilt ffmpeg binary in <site-packages>/imageio_ffmpeg/binaries/)
# with a source install that leaves no binary on disk. On cuda, IMAGEIO_FFMPEG_EXE
# (set above) points imageio at the LGPL CLI copied from wheel_builder. The
# --no-binary directive lives in the requirements file itself.
RUN --mount=type=bind,source=./container/deps/requirements.vllm.txt,target=/tmp/requirements.vllm.txt \
--mount=type=cache,target=/root/.cache/uv,sharing=locked \
export UV_CACHE_DIR=/root/.cache/uv && \
uv pip install {{ pip_target }} --reinstall-package imageio-ffmpeg --no-deps \
--requirement /tmp/requirements.vllm.txt

# Remove the vLLM source tree shipped in the base image to avoid pytest
# collection conflicts (duplicate conftest plugin registration) and stale
# tool scripts referencing files not present in Dynamo's build context.
Expand Down
8 changes: 8 additions & 0 deletions container/templates/wheel_builder.Dockerfile
Original file line number Diff line number Diff line change
Expand Up @@ -535,6 +535,11 @@ RUN --mount=type=secret,id=aws-web-identity-token,target=/run/secrets/aws-token

FROM wheel_builder_base AS runtime_wheel_builder

# Re-declare after FROM so --builder-image does not retain its baked-in
# compilation parallelism when the current build has a smaller resource budget.
ARG CARGO_BUILD_JOBS
ENV CARGO_BUILD_JOBS=${CARGO_BUILD_JOBS:-16}

{% if target not in ("dev", "local-dev") %}
# Copy source code (order matters for layer caching)
COPY .cargo/ /opt/dynamo/.cargo/
Expand Down Expand Up @@ -759,6 +764,9 @@ RUN --mount=type=secret,id=aws-web-identity-token,target=/run/secrets/aws-token
# so code-only commits never invalidate the published dependency image.
FROM reusable_builder_base AS wheel_builder

ARG CARGO_BUILD_JOBS
ENV CARGO_BUILD_JOBS=${CARGO_BUILD_JOBS:-16}

ARG TARGETARCH
ARG DEVICE
ARG USE_SCCACHE
Expand Down
65 changes: 65 additions & 0 deletions dingo/common/utils/runtime_termination.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,65 @@
# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
# SPDX-License-Identifier: Apache-2.0

"""Process-level guard for a permanently terminated Dynamo Runtime."""

from __future__ import annotations

import asyncio
from collections.abc import Awaitable
from typing import Any, TypeVar


_T = TypeVar("_T")


async def run_with_runtime_termination_guard(
operation: Awaitable[_T],
runtime: Any,
shutdown_event: asyncio.Event,
*,
component: str,
) -> _T:
"""Run ``operation`` until it completes or the Runtime terminates.

A Dynamo Runtime whose cancellation token has fired cannot rebuild its
discovery clients in place. An unexpected ``wait_shutdown`` completion
therefore cancels the component coroutine so its normal cleanup runs, then
raises and lets the process supervisor create a fresh Runtime.

``shutdown_event`` distinguishes this condition from an intentional signal
shutdown. Signal handling sets that event before calling
``runtime.shutdown()``, so the component retains its ordinary graceful
drain behavior in that path.
"""

operation_task = asyncio.ensure_future(operation)
runtime_task = asyncio.ensure_future(runtime.wait_shutdown())
try:
done, _pending = await asyncio.wait(
{operation_task, runtime_task}, return_when=asyncio.FIRST_COMPLETED
)
if runtime_task in done and not shutdown_event.is_set():
runtime_error: BaseException | None = None
if not runtime_task.cancelled():
runtime_error = runtime_task.exception()
if not operation_task.done():
operation_task.cancel()
await asyncio.gather(operation_task, return_exceptions=True)
error = RuntimeError(
f"{component} Dynamo Runtime terminated unexpectedly; "
"the process must restart"
)
if runtime_error is not None:
raise error from runtime_error
raise error
return await operation_task
except BaseException:
if not operation_task.done():
operation_task.cancel()
await asyncio.gather(operation_task, return_exceptions=True)
raise
finally:
if not runtime_task.done():
runtime_task.cancel()
await asyncio.gather(runtime_task, return_exceptions=True)
Loading
Loading