diff --git a/Dockerfile.sdk b/Dockerfile.sdk index 51e3980bc4..d7d01ba829 100644 --- a/Dockerfile.sdk +++ b/Dockerfile.sdk @@ -29,7 +29,7 @@ # # Base image on the minimum Triton container -ARG BASE_IMAGE=nvcr.io/nvidia/tritonserver:26.09-py3-min +ARG BASE_IMAGE=nvcr.io/nvidia/cuda-dl-base:26.09-cuda13.4-inference-runtime-ubuntu24.04 ARG TRITON_CLIENT_REPO_SUBDIR=clientrepo ARG TRITON_REPO_ORGANIZATION=http://github.com/triton-inference-server diff --git a/build.py b/build.py index 4410b0e818..14b3059838 100755 --- a/build.py +++ b/build.py @@ -59,6 +59,8 @@ # triton version -> # (triton container version, # upstream container version, +# cuda-dl-base version (container train and its CUDA version, which +# cuda-dl-base publishes as a matched pair and must be bumped together), # ORT version, # ORT OpenVINO version (use None to disable OpenVINO in ORT), # Standalone OpenVINO version, @@ -75,6 +77,7 @@ "release_version": "2.74.0dev", "triton_container_version": "26.10dev", "upstream_container_version": "26.09", + "cuda_dl_base_version": "26.09-cuda13.4", "ort_version": "1.30.0", "ort_openvino_version": "2026.3.1", "standalone_openvino_version": "2026.3.1", @@ -188,6 +191,18 @@ def target_machine(): return platform.machine().lower() +def default_build_container_image(): + """Image the compile toolchain runs in. + + Always CUDA-capable: the ONNX Runtime and OpenVINO backends build in this + container and need the CUDA toolchain even when Triton itself is being + built CPU-only. + """ + return "nvcr.io/nvidia/cuda-dl-base:{}-devel-ubuntu24.04".format( + FLAGS.cuda_dl_base_version + ) + + def container_versions(version, container_version, upstream_container_version): if container_version is None: container_version = FLAGS.triton_container_version @@ -661,21 +676,14 @@ def onnxruntime_cmake_args(images, library_paths): ) ) - if "base" in images: - cargs.append( - cmake_backend_arg( - "onnxruntime", "TRITON_BUILD_CONTAINER", None, images["base"] - ) - ) - else: - cargs.append( - cmake_backend_arg( - "onnxruntime", - "TRITON_BUILD_CONTAINER_VERSION", - None, - FLAGS.upstream_container_version, - ) + cargs.append( + cmake_backend_arg( + "onnxruntime", + "TRITON_BUILD_CONTAINER", + None, + images.get("base", default_build_container_image()), ) + ) # TODO: TPRD-333 OpenVino extension is not currently supported by our manylinux build if ( @@ -719,21 +727,14 @@ def openvino_cmake_args(): FLAGS.standalone_openvino_version, ) ] - if "base" in images: - cargs.append( - cmake_backend_arg( - "openvino", "TRITON_BUILD_CONTAINER", None, images["base"] - ) - ) - else: - cargs.append( - cmake_backend_arg( - "openvino", - "TRITON_BUILD_CONTAINER_VERSION", - None, - FLAGS.upstream_container_version, - ) + cargs.append( + cmake_backend_arg( + "openvino", + "TRITON_BUILD_CONTAINER", + None, + images.get("base", default_build_container_image()), ) + ) return cargs @@ -1693,14 +1694,24 @@ def create_build_dockerfiles( elif target_platform() == "rhel": raise KeyError("A base image must be specified when targeting RHEL") elif FLAGS.enable_gpu: - base_image = "nvcr.io/nvidia/tritonserver:{}-py3-min".format( - FLAGS.upstream_container_version - ) + base_image = default_build_container_image() else: base_image = "ubuntu:24.04" if "inference" in images: inference_image = images["inference"] + elif "base" in images: + # An explicit --image=base override has always supplied the runtime + # image as well. Leave it doing so, rather than silently swapping the + # final image for the default and discarding whatever runtime + # dependencies the override was chosen for. + inference_image = None + elif FLAGS.enable_gpu and target_platform() != "rhel" and "vllm" not in backends: + inference_image = ( + "nvcr.io/nvidia/cuda-dl-base:{}-inference-runtime-ubuntu24.04".format( + FLAGS.cuda_dl_base_version + ) + ) else: inference_image = None @@ -2541,7 +2552,7 @@ def enable_all(): "--image", action="append", required=False, - help='Use specified Docker image in build as ,. can be "base", "gpu-base", or "pytorch".', + help='Use specified Docker image in build as ,. can be "base", "gpu-base", "pytorch", or "inference".', ) parser.add_argument( @@ -2708,6 +2719,12 @@ def enable_all(): default=DEFAULT_TRITON_VERSION_MAP["upstream_container_version"], help="This flag sets the upstream container version for Triton Inference Server to be built. Default: the latest released version.", ) + parser.add_argument( + "--cuda-dl-base-version", + required=False, + default=DEFAULT_TRITON_VERSION_MAP["cuda_dl_base_version"], + help="This flag sets the cuda-dl-base container train and its CUDA version, as a matched pair (e.g. 26.09-cuda13.4), used for the default build and runtime base images. Default: the latest supported version.", + ) parser.add_argument( "--ort-version", required=False, diff --git a/compose.py b/compose.py index b5b5e691e8..9c10a51410 100755 --- a/compose.py +++ b/compose.py @@ -379,6 +379,16 @@ def create_argmap(images, skip_pull): "the container version will be chosen automatically based on the " "repository branch.", ) + parser.add_argument( + "--cuda-dl-base-version", + type=str, + required=False, + help="The cuda-dl-base container train and its CUDA version, as a matched " + "pair (e.g. 26.09-cuda13.4), used for the 'min' container. If not " + "specified the value from build.py's version map is used, which " + "corresponds to the branch compose.py is on rather than to " + "--container-version.", + ) parser.add_argument( "--image", action="append", @@ -465,14 +475,32 @@ def create_argmap(images, skip_pull): log('image "{}": "{}"'.format(parts[0], parts[1])) images[parts[0]] = parts[1] else: + container_version_specified = FLAGS.container_version is not None get_container_version_if_not_specified() if FLAGS.enable_gpu: + import build + + # cuda-dl-base couples its train to a CUDA version, so it cannot be + # derived from --container-version. Warn rather than silently pairing + # a requested `full` with a `min` from a different release. + if FLAGS.cuda_dl_base_version is None: + FLAGS.cuda_dl_base_version = build.DEFAULT_TRITON_VERSION_MAP[ + "cuda_dl_base_version" + ] + if container_version_specified: + log( + "warning: --container-version was specified but " + "--cuda-dl-base-version was not, so the 'min' container " + "will use {} and may not match the 'full' container".format( + FLAGS.cuda_dl_base_version + ) + ) images = { "full": "nvcr.io/nvidia/tritonserver:{}-py3".format( FLAGS.container_version ), - "min": "nvcr.io/nvidia/tritonserver:{}-py3-min".format( - FLAGS.container_version + "min": "nvcr.io/nvidia/cuda-dl-base:{}-inference-runtime-ubuntu24.04".format( + FLAGS.cuda_dl_base_version ), } else: diff --git a/deploy/gke-marketplace-app/README.md b/deploy/gke-marketplace-app/README.md index 595d4634ab..449fb49107 100644 --- a/deploy/gke-marketplace-app/README.md +++ b/deploy/gke-marketplace-app/README.md @@ -1,5 +1,5 @@