Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion Dockerfile.sdk
Original file line number Diff line number Diff line change
Expand Up @@ -29,7 +29,7 @@
#

# Base image on the minimum Triton container
ARG BASE_IMAGE=nvcr.io/nvidia/tritonserver:26.09-py3-min
ARG BASE_IMAGE=nvcr.io/nvidia/cuda-dl-base:26.09-cuda13.4-inference-runtime-ubuntu24.04
Comment thread
mc-nv marked this conversation as resolved.

ARG TRITON_CLIENT_REPO_SUBDIR=clientrepo
ARG TRITON_REPO_ORGANIZATION=http://github.com/triton-inference-server
Expand Down
81 changes: 49 additions & 32 deletions build.py
Original file line number Diff line number Diff line change
Expand Up @@ -59,6 +59,8 @@
# triton version ->
# (triton container version,
# upstream container version,
# cuda-dl-base version (container train and its CUDA version, which
# cuda-dl-base publishes as a matched pair and must be bumped together),
# ORT version,
# ORT OpenVINO version (use None to disable OpenVINO in ORT),
# Standalone OpenVINO version,
Expand All @@ -75,6 +77,7 @@
"release_version": "2.74.0dev",
"triton_container_version": "26.10dev",
"upstream_container_version": "26.09",
"cuda_dl_base_version": "26.09-cuda13.4",
"ort_version": "1.30.0",
"ort_openvino_version": "2026.3.1",
"standalone_openvino_version": "2026.3.1",
Expand Down Expand Up @@ -188,6 +191,18 @@ def target_machine():
return platform.machine().lower()


def default_build_container_image():
"""Image the compile toolchain runs in.

Always CUDA-capable: the ONNX Runtime and OpenVINO backends build in this
container and need the CUDA toolchain even when Triton itself is being
built CPU-only.
"""
return "nvcr.io/nvidia/cuda-dl-base:{}-devel-ubuntu24.04".format(
FLAGS.cuda_dl_base_version
)


def container_versions(version, container_version, upstream_container_version):
if container_version is None:
container_version = FLAGS.triton_container_version
Expand Down Expand Up @@ -661,21 +676,14 @@ def onnxruntime_cmake_args(images, library_paths):
)
)

if "base" in images:
cargs.append(
cmake_backend_arg(
"onnxruntime", "TRITON_BUILD_CONTAINER", None, images["base"]
)
)
else:
cargs.append(
cmake_backend_arg(
"onnxruntime",
"TRITON_BUILD_CONTAINER_VERSION",
None,
FLAGS.upstream_container_version,
)
cargs.append(
cmake_backend_arg(
"onnxruntime",
"TRITON_BUILD_CONTAINER",
None,
images.get("base", default_build_container_image()),
)
)

# TODO: TPRD-333 OpenVino extension is not currently supported by our manylinux build
if (
Expand Down Expand Up @@ -719,21 +727,14 @@ def openvino_cmake_args():
FLAGS.standalone_openvino_version,
)
]
if "base" in images:
cargs.append(
cmake_backend_arg(
"openvino", "TRITON_BUILD_CONTAINER", None, images["base"]
)
)
else:
cargs.append(
cmake_backend_arg(
"openvino",
"TRITON_BUILD_CONTAINER_VERSION",
None,
FLAGS.upstream_container_version,
)
cargs.append(
cmake_backend_arg(
"openvino",
"TRITON_BUILD_CONTAINER",
None,
images.get("base", default_build_container_image()),
)
)
return cargs


Expand Down Expand Up @@ -1693,14 +1694,24 @@ def create_build_dockerfiles(
elif target_platform() == "rhel":
raise KeyError("A base image must be specified when targeting RHEL")
elif FLAGS.enable_gpu:
base_image = "nvcr.io/nvidia/tritonserver:{}-py3-min".format(
FLAGS.upstream_container_version
)
base_image = default_build_container_image()
else:
base_image = "ubuntu:24.04"

if "inference" in images:
inference_image = images["inference"]
elif "base" in images:
# An explicit --image=base override has always supplied the runtime
# image as well. Leave it doing so, rather than silently swapping the
# final image for the default and discarding whatever runtime
# dependencies the override was chosen for.
inference_image = None
elif FLAGS.enable_gpu and target_platform() != "rhel" and "vllm" not in backends:
inference_image = (
"nvcr.io/nvidia/cuda-dl-base:{}-inference-runtime-ubuntu24.04".format(
FLAGS.cuda_dl_base_version
)
)
Comment thread
greptile-apps[bot] marked this conversation as resolved.
else:
inference_image = None

Expand Down Expand Up @@ -2541,7 +2552,7 @@ def enable_all():
"--image",
action="append",
required=False,
help='Use specified Docker image in build as <image-name>,<full-image-name>. <image-name> can be "base", "gpu-base", or "pytorch".',
help='Use specified Docker image in build as <image-name>,<full-image-name>. <image-name> can be "base", "gpu-base", "pytorch", or "inference".',
)

parser.add_argument(
Expand Down Expand Up @@ -2708,6 +2719,12 @@ def enable_all():
default=DEFAULT_TRITON_VERSION_MAP["upstream_container_version"],
help="This flag sets the upstream container version for Triton Inference Server to be built. Default: the latest released version.",
)
parser.add_argument(
"--cuda-dl-base-version",
required=False,
default=DEFAULT_TRITON_VERSION_MAP["cuda_dl_base_version"],
help="This flag sets the cuda-dl-base container train and its CUDA version, as a matched pair (e.g. 26.09-cuda13.4), used for the default build and runtime base images. Default: the latest supported version.",
)
parser.add_argument(
"--ort-version",
required=False,
Expand Down
32 changes: 30 additions & 2 deletions compose.py
Original file line number Diff line number Diff line change
Expand Up @@ -90,7 +90,7 @@
images["min"]
)

import build

Check notice

Code scanning / CodeQL

Module is imported more than once Note

This import of module build is redundant, as it was previously imported
on line 481
.

df += build.dockerfile_prepare_container_linux(
argmap, backends, FLAGS.enable_gpu, platform.machine().lower()
Expand Down Expand Up @@ -176,7 +176,7 @@
# Read from TRITON_VERSION file in server repo to determine version
with open("TRITON_VERSION", "r") as vfile:
version = vfile.readline().strip()
import build

Check notice

Code scanning / CodeQL

Module is imported more than once Note

This import of module build is redundant, as it was previously imported
on line 481
.

_, FLAGS.container_version = build.container_versions(
version, None, FLAGS.container_version
Expand Down Expand Up @@ -379,6 +379,16 @@
"the container version will be chosen automatically based on the "
"repository branch.",
)
parser.add_argument(
"--cuda-dl-base-version",
type=str,
required=False,
help="The cuda-dl-base container train and its CUDA version, as a matched "
"pair (e.g. 26.09-cuda13.4), used for the 'min' container. If not "
"specified the value from build.py's version map is used, which "
"corresponds to the branch compose.py is on rather than to "
"--container-version.",
)
parser.add_argument(
"--image",
action="append",
Expand Down Expand Up @@ -465,14 +475,32 @@
log('image "{}": "{}"'.format(parts[0], parts[1]))
images[parts[0]] = parts[1]
else:
container_version_specified = FLAGS.container_version is not None
get_container_version_if_not_specified()
if FLAGS.enable_gpu:
import build

# cuda-dl-base couples its train to a CUDA version, so it cannot be
# derived from --container-version. Warn rather than silently pairing
# a requested `full` with a `min` from a different release.
if FLAGS.cuda_dl_base_version is None:
FLAGS.cuda_dl_base_version = build.DEFAULT_TRITON_VERSION_MAP[
"cuda_dl_base_version"
]
if container_version_specified:
log(
"warning: --container-version was specified but "
"--cuda-dl-base-version was not, so the 'min' container "
"will use {} and may not match the 'full' container".format(
FLAGS.cuda_dl_base_version
)
)
images = {
"full": "nvcr.io/nvidia/tritonserver:{}-py3".format(
FLAGS.container_version
),
"min": "nvcr.io/nvidia/tritonserver:{}-py3-min".format(
FLAGS.container_version
"min": "nvcr.io/nvidia/cuda-dl-base:{}-inference-runtime-ubuntu24.04".format(
FLAGS.cuda_dl_base_version
),
}
else:
Expand Down
6 changes: 3 additions & 3 deletions deploy/gke-marketplace-app/README.md
Original file line number Diff line number Diff line change
@@ -1,5 +1,5 @@
<!--
# Copyright (c) 2021-2024, NVIDIA CORPORATION & AFFILIATES. All rights reserved.
# Copyright (c) 2021-2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved.
#
# Redistribution and use in source and binary forms, with or without
# modification, are permitted provided that the following conditions
Expand Down Expand Up @@ -141,13 +141,13 @@ Please note that A100 MIG in GKE does not support GPU metrics yet, also Triton G

Second, go to this [GKE Marketplace link](https://console.cloud.google.com/marketplace/details/nvidia-ngc-public/triton-inference-server) to deploy Triton application.

Users can leave everything as default if their models have already been tested/validated with Triton. They can provide a GCS path pointing to the model repository containing their models. By default, we provide a BERT large model optimized by TensorRT in a public demo GCS bucket that is compatible with the `xx.yy` release of Triton Server in `gs://triton_sample_models/xx_yy`. However, please take note of the following about this demo bucket:
Users can leave everything as default if their models have already been tested/validated with Triton. They can provide a GCS path pointing to the model repository containing their models. By default, we provide a BERT large model optimized by TensorRT in a public demo GCS bucket that is compatible with the `YY.MM` release of Triton Server in `gs://triton_sample_models/YY.MM`. However, please take note of the following about this demo bucket:
- The TensorRT engine provided in the demo bucket is only compatible with Tesla T4 GPUs.
- This bucket is located in `us-central1`, so loading from this bucket into Triton in other regions may be affected.
- The first deployment of this Triton GKE application will be slower than consecutive runs because the image needs to be pulled into the GKE cluster.
- You can find an example of how this model is generated and uploaded [here](trt-engine/README.md).

Where <xx.yy> is the version of NGC Triton container needed.
Where <YY.MM> is the version of NGC Triton container needed.

![GKE Marketplace Application UI](ui.png)

Expand Down
4 changes: 2 additions & 2 deletions docs/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -46,10 +46,10 @@ Containers](http://docs.nvidia.com/deeplearning/dgx/preparing-containers/index.h
Pull the image using the following command.

```
$ docker pull nvcr.io/nvidia/tritonserver:<yy.mm>-py3
$ docker pull nvcr.io/nvidia/tritonserver:<YY.MM>-py3
```

Where \<yy.mm\> is the version of Triton that you want to pull. For a complete list of all the variants and versions of the Triton Inference Server Container, visit the [NGC Page](https://catalog.ngc.nvidia.com/orgs/nvidia/containers/tritonserver). More information about customizing the Triton Container can be found in [this section](customization_guide/compose.md) of the User Guide.
Where \<YY.MM\> is the version of Triton that you want to pull. For a complete list of all the variants and versions of the Triton Inference Server Container, visit the [NGC Page](https://catalog.ngc.nvidia.com/orgs/nvidia/containers/tritonserver). More information about customizing the Triton Container can be found in [this section](customization_guide/compose.md) of the User Guide.

## **Getting Started**

Expand Down
14 changes: 6 additions & 8 deletions docs/customization_guide/build.md
Original file line number Diff line number Diff line change
Expand Up @@ -103,11 +103,11 @@ building with Docker.
*tritonserver_buildbase* image is based on a minimal/base
image. When building with GPU support (--enable-gpu), the *min*
image is the
[\<xx.yy\>-py3-min](https://catalog.ngc.nvidia.com/orgs/nvidia/containers/tritonserver)
[cuda-dl-base \<YY.MM\>-cuda\<X.Y\>-devel](https://catalog.ngc.nvidia.com/orgs/nvidia/containers/cuda-dl-base)
image pulled from [NGC](https://ngc.nvidia.com) that contains the
CUDA, cuDNN, TensorRT and other dependencies that are required to
build Triton. When building without GPU support, the *min* image
is the standard ubuntu:22.04 image.
is the standard ubuntu:24.04 image.

* Run the cmake_build script within the *tritonserver_buildbase*
image to actually build Triton. The cmake_build script performs
Expand Down Expand Up @@ -339,7 +339,7 @@ available for a non-GPU / CPU-only build: `identity`, `repeat`, `ensemble`,
CPU-only builds of the PyTorch backends require some CUDA stubs
and runtime dependencies that are not present in the CPU-only base container.
These are retrieved from a GPU base container, which can be changed with the
`--image=gpu-base,nvcr.io/nvidia/tritonserver:<xx.yy>-py3-min` flag.
`--image=gpu-base,nvcr.io/nvidia/tritonserver:<YY.MM>-py3-min` flag.

### Building Without Docker

Expand All @@ -362,11 +362,9 @@ $ ./build.py -v --enable-all
From Dockerfile.buildbase you can see what dependencies you need to
install on your host system. Note that when building with --enable-gpu
(or --enable-all), Dockerfile.buildbase depends on the
[\<xx.yy\>-py3-min](https://catalog.ngc.nvidia.com/orgs/nvidia/containers/tritonserver)
image pulled from [NGC](https://ngc.nvidia.com). Unfortunately, a
Dockerfile is not currently available for the
[\<xx.yy\>-py3-min](https://catalog.ngc.nvidia.com/orgs/nvidia/containers/tritonserver)
image. Instead, you must manually install [CUDA and
[cuda-dl-base \<YY.MM\>-cuda\<X.Y\>-devel](https://catalog.ngc.nvidia.com/orgs/nvidia/containers/cuda-dl-base)
image pulled from [NGC](https://ngc.nvidia.com). Rather than reproducing
that image, you must manually install [CUDA and
cuDNN](#cuda-cublas-cudnn) and [TensorRT](#tensorrt) dependencies as
described below.

Expand Down
9 changes: 5 additions & 4 deletions docs/customization_guide/compose.md
Original file line number Diff line number Diff line change
Expand Up @@ -72,14 +72,15 @@ This may result in different GPU statistic reporting behavior.

`compose.py` requires two containers: a `min` container which is the
base the compose container is built from and a `full` container from which the
script will extract components. The version of the `min` and `full` container
is determined by the branch of Triton `compose.py` is on.
script will extract components. The version of the `full` container is
determined by the branch of Triton `compose.py` is on, while the `min`
container is the CUDA DL base image that branch builds against.
For example, running
```
python3 compose.py --backend pytorch --repoagent checksum
```
on branch [r26.09](https://github.com/triton-inference-server/server/tree/r26.09) pulls:
- `min` container `nvcr.io/nvidia/tritonserver:26.09-py3-min`
- `min` container `nvcr.io/nvidia/cuda-dl-base:26.09-cuda13.4-inference-runtime-ubuntu24.04`
- `full` container `nvcr.io/nvidia/tritonserver:26.09-py3`

Alternatively, users can specify the version of Triton container to pull from
Expand All @@ -91,7 +92,7 @@ python3 compose.py --backend pytorch --repoagent checksum --container-version 26
2. Specifying `--image min,<min container image name> --image full,<full container image name>`.
The user is responsible for specifying compatible `min` and `full` containers.
```
python3 compose.py --backend pytorch --repoagent checksum --image min,nvcr.io/nvidia/tritonserver:26.09-py3-min --image full,nvcr.io/nvidia/tritonserver:26.09-py3
python3 compose.py --backend pytorch --repoagent checksum --image min,nvcr.io/nvidia/cuda-dl-base:26.09-cuda13.4-inference-runtime-ubuntu24.04 --image full,nvcr.io/nvidia/tritonserver:26.09-py3
```
Method 1 and 2 will result in the same composed container. Furthermore,
`--image` flag overrides the `--container-version` flag when both are specified.
Expand Down
16 changes: 8 additions & 8 deletions docs/getting_started/quickstart.md
Original file line number Diff line number Diff line change
@@ -1,5 +1,5 @@
<!--
# Copyright (c) 2018-2024, NVIDIA CORPORATION. All rights reserved.
# Copyright (c) 2018-2026, NVIDIA CORPORATION. All rights reserved.
#
# Redistribution and use in source and binary forms, with or without
# modification, are permitted provided that the following conditions
Expand Down Expand Up @@ -73,10 +73,10 @@ for Docker to recognize the GPU(s). The --gpus=1 flag indicates that 1
system GPU should be made available to Triton for inferencing.

```
$ docker run --gpus=1 --rm -p8000:8000 -p8001:8001 -p8002:8002 -v/full/path/to/docs/examples/model_repository:/models nvcr.io/nvidia/tritonserver:<xx.yy>-py3 tritonserver --model-repository=/models
$ docker run --gpus=1 --rm -p8000:8000 -p8001:8001 -p8002:8002 -v/full/path/to/docs/examples/model_repository:/models nvcr.io/nvidia/tritonserver:<YY.MM>-py3 tritonserver --model-repository=/models
```

Where \<xx.yy\> is the version of Triton that you want to use (and
Where \<YY.MM\> is the version of Triton that you want to use (and
pulled above). After you start Triton you will see output on the
console showing the server starting up and loading the model. When you
see output like the following, Triton is ready to accept inference
Expand Down Expand Up @@ -106,7 +106,7 @@ On a system without GPUs, Triton should be run without using the
above.

```
$ docker run --rm -p8000:8000 -p8001:8001 -p8002:8002 -v/full/path/to/docs/examples/model_repository:/models nvcr.io/nvidia/tritonserver:<xx.yy>-py3 tritonserver --model-repository=/models
$ docker run --rm -p8000:8000 -p8001:8001 -p8002:8002 -v/full/path/to/docs/examples/model_repository:/models nvcr.io/nvidia/tritonserver:<YY.MM>-py3 tritonserver --model-repository=/models
```

Because the --gpus flag is not used, a GPU is not available and Triton
Expand Down Expand Up @@ -136,17 +136,17 @@ Use docker pull to get the client libraries and examples image
from NGC.

```
$ docker pull nvcr.io/nvidia/tritonserver:<xx.yy>-py3-sdk
$ docker pull nvcr.io/nvidia/tritonserver:<YY.MM>-py3-sdk
```

Where \<xx.yy\> is the version that you want to pull. Run the client
Where \<YY.MM\> is the version that you want to pull. Run the client
image.

```
$ docker run -it --rm --net=host nvcr.io/nvidia/tritonserver:<xx.yy>-py3-sdk
$ docker run -it --rm --net=host nvcr.io/nvidia/tritonserver:<YY.MM>-py3-sdk
```

From within the nvcr.io/nvidia/tritonserver:<xx.yy>-py3-sdk
From within the nvcr.io/nvidia/tritonserver:<YY.MM>-py3-sdk
image, run the example image-client application to perform image
classification using the example densenet_onnx model.

Expand Down
Loading