diff --git a/.github/workflows/example_tests.yml b/.github/workflows/example_tests.yml index 297366149..d0d23df68 100644 --- a/.github/workflows/example_tests.yml +++ b/.github/workflows/example_tests.yml @@ -93,11 +93,11 @@ jobs: strategy: fail-fast: false matrix: - example: [llm_ptq] + example: [llm_ptq, vlm_ptq] uses: ./.github/workflows/_example_tests_runner.yml secrets: inherit with: - docker_image: "nvcr.io/nvidia/tensorrt-llm/release:1.1.0rc2.post2" + docker_image: "nvcr.io/nvidia/tensorrt-llm/release:1.2.0rc4" example: ${{ matrix.example }} pip_install_extras: "[hf,dev-test]" runner: linux-amd64-gpu-h100-latest-1 @@ -111,7 +111,7 @@ jobs: uses: ./.github/workflows/_example_tests_runner.yml secrets: inherit with: - docker_image: "nvcr.io/nvidia/tensorrt-llm/release:1.1.0rc2.post2" + docker_image: "nvcr.io/nvidia/tensorrt-llm/release:1.2.0rc4" example: ${{ matrix.example }} pip_install_extras: "[hf,dev-test]" runner: linux-amd64-gpu-h100-latest-2 diff --git a/CHANGELOG.rst b/CHANGELOG.rst index f6a323441..643f446ac 100755 --- a/CHANGELOG.rst +++ b/CHANGELOG.rst @@ -27,6 +27,7 @@ Model Optimizer Changelog (Linux) **Misc** +- Bump TensorRT-LLM docker to 1.2.0rc4. - Bump minimum recommended transformers version to 4.53. - Replace ONNX simplification package from ``onnxsim`` to ``onnxslim``. diff --git a/docs/source/getting_started/_installation_for_Linux.rst b/docs/source/getting_started/_installation_for_Linux.rst index f6e60ade7..0a82ecd1e 100644 --- a/docs/source/getting_started/_installation_for_Linux.rst +++ b/docs/source/getting_started/_installation_for_Linux.rst @@ -18,7 +18,7 @@ Latest Model Optimizer (``nvidia-modelopt``) currently has the following system +-------------------------+-----------------------------+ | PyTorch | >=2.6 | +-------------------------+-----------------------------+ -| TensorRT-LLM (Optional) | 1.1.0rc2.post2 | +| TensorRT-LLM (Optional) | 1.2.0rc4 | +-------------------------+-----------------------------+ | ONNX Runtime (Optional) | 1.22 | +-------------------------+-----------------------------+ diff --git a/examples/llm_ptq/README.md b/examples/llm_ptq/README.md index 4009fbd6d..46260db8f 100755 --- a/examples/llm_ptq/README.md +++ b/examples/llm_ptq/README.md @@ -27,7 +27,7 @@ This section focuses on Post-training quantization, a technique that reduces mod ### Docker -For Hugging Face models, please use the TensorRT-LLM docker image (e.g., `nvcr.io/nvidia/tensorrt-llm/release:1.1.0rc2.post2`). +For Hugging Face models, please use the TensorRT-LLM docker image (e.g., `nvcr.io/nvidia/tensorrt-llm/release:1.2.0rc4`). For NeMo models, use the NeMo container (e.g., `nvcr.io/nvidia/nemo:25.09`). Visit our [installation docs](https://nvidia.github.io/TensorRT-Model-Optimizer/getting_started/2_installation.html) for more information. diff --git a/examples/specdec_bench/README.md b/examples/specdec_bench/README.md index b0e955b4a..50a7bc817 100644 --- a/examples/specdec_bench/README.md +++ b/examples/specdec_bench/README.md @@ -4,7 +4,7 @@ This benchmark is meant to be a lightweight layer ontop of an existing vLLM/SGLang/TRTLLM installation. For example, no install is required if one is running in the following dockers: `vllm/vllm-openai:v0.11.0` (vLLM), `lmsysorg/sglang:v0.5.4.post2` (SGLang), or -`nvcr.io/nvidia/tensorrt-llm/release:1.2.0rc1` (TRT-LLM). +`nvcr.io/nvidia/tensorrt-llm/release:1.2.0rc4` (TRT-LLM). Next @@ -16,7 +16,7 @@ cd examples/specdec_bench Collect relevant metrics on acceptance rate, timing, and outputs for Speculative Decoding methods. Acceptance rate refers to the number of tokens generated on every iteration. For a standard Autoregressive LLM, this number -is just 1. +is just 1. ## Getting Started diff --git a/examples/speculative_decoding/collect_hidden_states/slurm_dump.sh b/examples/speculative_decoding/collect_hidden_states/slurm_dump.sh index b95557b4c..9c9e7a71f 100644 --- a/examples/speculative_decoding/collect_hidden_states/slurm_dump.sh +++ b/examples/speculative_decoding/collect_hidden_states/slurm_dump.sh @@ -20,8 +20,8 @@ # THE BIWEEKLY CAPACITY MEETING. IF YOU DON'T KNOW WHO IS THE PIC OF YOUR CSRG PPP # MANAGEMET, GO WITH `-p backfill -t 00:25:00`. -#SBATCH -A coreai_dlalgo_modelopt -#SBATCH --job-name=coreai_dlalgo_modelopt-generate_eagle_hidden_states +#SBATCH -A +#SBATCH --job-name= #SBATCH --nodes=1 --ntasks-per-node=4 --gpus-per-node=4 #SBATCH -p batch #SBATCH -t 04:00:00 @@ -29,7 +29,7 @@ echo "SLURM_ARRAY_TASK_ID: $SLURM_ARRAY_TASK_ID" echo "SLURM_ARRAY_TASK_COUNT: $SLURM_ARRAY_TASK_COUNT" -CONTAINER="nvcr.io#nvidia/tensorrt-llm/release:1.2.0rc0" +CONTAINER="nvcr.io/nvidia/tensorrt-llm/release:1.2.0rc4" INPUT_DIR="" DUMP_DIR="