From 3437bb64e361f77bbcadde9573c6fffb2306e1fc Mon Sep 17 00:00:00 2001 From: "M. Chornyi" <99709299+mc-nv@users.noreply.github.com> Date: Fri, 2 Oct 2026 04:37:22 +0000 Subject: [PATCH 1/2] fix: Require TRITON_BUILD_CONTAINER instead of synthesizing a py3-min image When only TRITON_BUILD_CONTAINER_VERSION was given, the build container was invented as nvcr.io/nvidia/tritonserver:${VERSION}-py3-min. That image is being retired, so the fallback is a latent failure: configure succeeds and the build dies much later on a docker pull 404 for an image nobody asked for. Stop inventing one. Building ONNX Runtime from source now requires TRITON_BUILD_CONTAINER, which turns a missing image into a configure-time error naming exactly what to supply. TRITON_BUILD_CONTAINER_VERSION had no other use in this repo -- it only fed the synthesis -- so it goes with it and the two conditions collapse into one guard. Unlike the OpenVINO backend, this one needs the CUDA toolchain in its build container when GPU support is enabled, so both README examples name a CUDA-capable image and the header comment says so. Breaking for standalone cmake builds: configuring with only -DTRITON_BUILD_CONTAINER_VERSION now fails with "TRITON_BUILD_ONNXRUNTIME_VERSION requires TRITON_BUILD_CONTAINER". The README examples are updated in the same commit so the documented path does not point at a removed variable. Builds driven by server/build.py are unaffected -- it always passes a resolved image. --- CMakeLists.txt | 20 +++++++------------- README.md | 4 ++-- 2 files changed, 9 insertions(+), 15 deletions(-) diff --git a/CMakeLists.txt b/CMakeLists.txt index 43ef398..3dc5ecb 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -42,11 +42,10 @@ project(tritononnxruntimebackend LANGUAGES C CXX) # - Set TRITON_BUILD_ONNXRUNTIME_VERSION to the version of ONNX # Runtime that you want to be built for the backend. # -# - Set TRITON_BUILD_CONTAINER to the Triton container to use as a -# base for the build. On linux you can instead set -# TRITON_BUILD_CONTAINER_VERSION to the Triton version that you -# want to target with the build and the corresponding container -# from NGC will be used. +# - Set TRITON_BUILD_CONTAINER to the container to use as a base for +# the build. Required: the image is never inferred here, so that +# whoever drives the build decides it. It must provide the CUDA +# toolchain when GPU support is enabled. # # - Optionally set TRITON_BUILD_CUDA_VERSION and # TRITON_BUILD_CUDA_HOME. If not set these are automatically set @@ -91,8 +90,7 @@ option(TRITON_ENABLE_ONNXRUNTIME_TENSORRT "Enable TensorRT execution provider for ONNXRuntime backend in server" OFF) option(TRITON_ENABLE_ONNXRUNTIME_OPENVINO "Enable OpenVINO execution provider for ONNXRuntime backend in server" OFF) -set(TRITON_BUILD_CONTAINER "" CACHE STRING "Triton container to use a base for build") -set(TRITON_BUILD_CONTAINER_VERSION "" CACHE STRING "Triton container version to target") +set(TRITON_BUILD_CONTAINER "" CACHE STRING "Container to use as a base for build") set(TRITON_BUILD_ONNXRUNTIME_VERSION "" CACHE STRING "ONNXRuntime version to build") set(TRITON_BUILD_ONNXRUNTIME_OPENVINO_VERSION "" CACHE STRING "ONNXRuntime OpenVINO version to build") set(TRITON_BUILD_TARGET_PLATFORM "" CACHE STRING "Target platform for ONNXRuntime build") @@ -149,13 +147,9 @@ if(NOT TRITON_ONNXRUNTIME_DOCKER_BUILD) else() - if(NOT TRITON_BUILD_CONTAINER AND NOT TRITON_BUILD_CONTAINER_VERSION) - message(FATAL_ERROR - "TRITON_BUILD_ONNXRUNTIME_VERSION requires TRITON_BUILD_CONTAINER or TRITON_BUILD_CONTAINER_VERSION") - endif() - if(NOT TRITON_BUILD_CONTAINER) - set(TRITON_BUILD_CONTAINER "nvcr.io/nvidia/tritonserver:${TRITON_BUILD_CONTAINER_VERSION}-py3-min") + message(FATAL_ERROR + "TRITON_BUILD_ONNXRUNTIME_VERSION requires TRITON_BUILD_CONTAINER") endif() set(TRITON_ONNXRUNTIME_DOCKER_IMAGE "tritonserver_onnxruntime") diff --git a/README.md b/README.md index cc34621..ba4a74b 100644 --- a/README.md +++ b/README.md @@ -51,7 +51,7 @@ build.py](https://github.com/triton-inference-server/server/blob/r23.04/build.py ``` $ mkdir build $ cd build -$ cmake -DCMAKE_INSTALL_PREFIX:PATH=`pwd`/install -DTRITON_BUILD_ONNXRUNTIME_VERSION=1.14.1 -DTRITON_BUILD_CONTAINER_VERSION=23.04 .. +$ cmake -DCMAKE_INSTALL_PREFIX:PATH=`pwd`/install -DTRITON_BUILD_ONNXRUNTIME_VERSION=1.14.1 -DTRITON_BUILD_CONTAINER=nvcr.io/nvidia/cuda-dl-base:26.09-cuda13.4-devel-ubuntu24.04 .. $ make install ``` @@ -77,7 +77,7 @@ TensorRT and OpenVino support: ``` $ mkdir build $ cd build -$ cmake -DCMAKE_INSTALL_PREFIX:PATH=`pwd`/install -DTRITON_BUILD_ONNXRUNTIME_VERSION=1.14.1 -DTRITON_BUILD_CONTAINER_VERSION=23.04 -DTRITON_ENABLE_ONNXRUNTIME_TENSORRT=ON -DTRITON_ENABLE_ONNXRUNTIME_OPENVINO=ON -DTRITON_BUILD_ONNXRUNTIME_OPENVINO_VERSION=2021.2.200 .. +$ cmake -DCMAKE_INSTALL_PREFIX:PATH=`pwd`/install -DTRITON_BUILD_ONNXRUNTIME_VERSION=1.14.1 -DTRITON_BUILD_CONTAINER=nvcr.io/nvidia/cuda-dl-base:26.09-cuda13.4-devel-ubuntu24.04 -DTRITON_ENABLE_ONNXRUNTIME_TENSORRT=ON -DTRITON_ENABLE_ONNXRUNTIME_OPENVINO=ON -DTRITON_BUILD_ONNXRUNTIME_OPENVINO_VERSION=2021.2.200 .. $ make install ``` From f12b50f4e29ebe8a46f0c6ed344220405d7d4072 Mon Sep 17 00:00:00 2001 From: "M. Chornyi" <99709299+mc-nv@users.noreply.github.com> Date: Fri, 2 Oct 2026 18:13:41 +0000 Subject: [PATCH 2/2] docs: Align the build examples with the container image they now name The examples pinned ONNX Runtime 1.14.1 and OpenVINO 2021.2.200 -- versions the surrounding text ties to Triton 23.04 -- while naming a 26.09 CUDA 13.4 build container, leaving no example of a matched configuration. Move the versions and the worked reference to 26.09: ORT 1.30.0 and OpenVINO 2026.3.1, matching that release's TRITON_VERSION_MAP entry. Reported by Greptile on #366. --- README.md | 10 +++++----- 1 file changed, 5 insertions(+), 5 deletions(-) diff --git a/README.md b/README.md index ba4a74b..b935952 100644 --- a/README.md +++ b/README.md @@ -44,14 +44,14 @@ Runtime version and a Triton container version that you want to use with the backend. You can find the combination of versions used in a particular Triton release in the TRITON_VERSION_MAP at the top of build.py in the branch matching the Triton release you are interested -in. For example, to build the ONNX Runtime backend for Triton 23.04, -use the versions from TRITON_VERSION_MAP in the [r23.04 branch of -build.py](https://github.com/triton-inference-server/server/blob/r23.04/build.py#L73). +in. For example, to build the ONNX Runtime backend for Triton 26.09, +use the versions from TRITON_VERSION_MAP in the [r26.09 branch of +build.py](https://github.com/triton-inference-server/server/blob/r26.09/build.py#L73). ``` $ mkdir build $ cd build -$ cmake -DCMAKE_INSTALL_PREFIX:PATH=`pwd`/install -DTRITON_BUILD_ONNXRUNTIME_VERSION=1.14.1 -DTRITON_BUILD_CONTAINER=nvcr.io/nvidia/cuda-dl-base:26.09-cuda13.4-devel-ubuntu24.04 .. +$ cmake -DCMAKE_INSTALL_PREFIX:PATH=`pwd`/install -DTRITON_BUILD_ONNXRUNTIME_VERSION=1.30.0 -DTRITON_BUILD_CONTAINER=nvcr.io/nvidia/cuda-dl-base:26.09-cuda13.4-devel-ubuntu24.04 .. $ make install ``` @@ -77,7 +77,7 @@ TensorRT and OpenVino support: ``` $ mkdir build $ cd build -$ cmake -DCMAKE_INSTALL_PREFIX:PATH=`pwd`/install -DTRITON_BUILD_ONNXRUNTIME_VERSION=1.14.1 -DTRITON_BUILD_CONTAINER=nvcr.io/nvidia/cuda-dl-base:26.09-cuda13.4-devel-ubuntu24.04 -DTRITON_ENABLE_ONNXRUNTIME_TENSORRT=ON -DTRITON_ENABLE_ONNXRUNTIME_OPENVINO=ON -DTRITON_BUILD_ONNXRUNTIME_OPENVINO_VERSION=2021.2.200 .. +$ cmake -DCMAKE_INSTALL_PREFIX:PATH=`pwd`/install -DTRITON_BUILD_ONNXRUNTIME_VERSION=1.30.0 -DTRITON_BUILD_CONTAINER=nvcr.io/nvidia/cuda-dl-base:26.09-cuda13.4-devel-ubuntu24.04 -DTRITON_ENABLE_ONNXRUNTIME_TENSORRT=ON -DTRITON_ENABLE_ONNXRUNTIME_OPENVINO=ON -DTRITON_BUILD_ONNXRUNTIME_OPENVINO_VERSION=2026.3.1 .. $ make install ```