From ec595a8e297b0bb80e1b41daf2f4ea28d0dbcd93 Mon Sep 17 00:00:00 2001 From: Yiqing Yan Date: Sun, 31 Aug 2025 10:20:38 +0800 Subject: [PATCH 1/5] [None][chore] Bump version to 1.1.0rc2 (#7394) Signed-off-by: Yiqing Yan --- README.md | 2 +- examples/constraints.txt | 2 +- tensorrt_llm/version.py | 2 +- 3 files changed, 3 insertions(+), 3 deletions(-) diff --git a/README.md b/README.md index e4f4bd230cec..2b9c95586d3d 100644 --- a/README.md +++ b/README.md @@ -9,7 +9,7 @@ TensorRT-LLM [![python](https://img.shields.io/badge/python-3.10-green)](https://www.python.org/downloads/release/python-31012/) [![cuda](https://img.shields.io/badge/cuda-12.9.1-green)](https://developer.nvidia.com/cuda-downloads) [![trt](https://img.shields.io/badge/TRT-10.11.0-green)](https://developer.nvidia.com/tensorrt) -[![version](https://img.shields.io/badge/release-1.1.0rc2-green)](./tensorrt_llm/version.py) +[![version](https://img.shields.io/badge/release-1.1.0rc3-green)](./tensorrt_llm/version.py) [![license](https://img.shields.io/badge/license-Apache%202-blue)](./LICENSE) [Architecture](./docs/source/torch/arch_overview.md)   |   [Performance](./docs/source/performance/perf-overview.md)   |   [Examples](https://nvidia.github.io/TensorRT-LLM/quick-start-guide.html)   |   [Documentation](./docs/source/)   |   [Roadmap](https://github.com/NVIDIA/TensorRT-LLM/issues?q=is%3Aissue%20state%3Aopen%20label%3Aroadmap) diff --git a/examples/constraints.txt b/examples/constraints.txt index 8b0d1a009303..527f9c8201b8 100644 --- a/examples/constraints.txt +++ b/examples/constraints.txt @@ -1,3 +1,3 @@ -tensorrt_llm==1.1.0rc2 +tensorrt_llm==1.1.0rc3 evaluate~=0.4.1 rouge_score~=0.1.2 diff --git a/tensorrt_llm/version.py b/tensorrt_llm/version.py index 93b6027df5cc..2059111b7cb7 100644 --- a/tensorrt_llm/version.py +++ b/tensorrt_llm/version.py @@ -12,4 +12,4 @@ # WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. # See the License for the specific language governing permissions and # limitations under the License. -__version__ = "1.1.0rc2" +__version__ = "1.1.0rc3" From 12a5d78daf18c028c3a8de1f89228212acbc0cf2 Mon Sep 17 00:00:00 2001 From: Kaiyu Xie <26294424+kaiyux@users.noreply.github.com> Date: Sun, 31 Aug 2025 15:04:42 -0700 Subject: [PATCH 2/5] Fix nsys in slurm scripts Signed-off-by: Kaiyu Xie <26294424+kaiyux@users.noreply.github.com> --- examples/disaggregated/slurm/benchmark/disaggr_torch.slurm | 5 +++-- examples/disaggregated/slurm/benchmark/start_worker.sh | 4 ++-- tensorrt_llm/_torch/pyexecutor/py_executor.py | 4 ++-- 3 files changed, 7 insertions(+), 6 deletions(-) diff --git a/examples/disaggregated/slurm/benchmark/disaggr_torch.slurm b/examples/disaggregated/slurm/benchmark/disaggr_torch.slurm index 58613e081455..303c2689c4cc 100644 --- a/examples/disaggregated/slurm/benchmark/disaggr_torch.slurm +++ b/examples/disaggregated/slurm/benchmark/disaggr_torch.slurm @@ -83,8 +83,6 @@ echo "ntasks_per_node: ${ntasks_per_node}" echo "===========================================" -nsys_on="" -# nsys_on=${full_logdir} # Uncomment this line to enable Nsys profiling numa_bind=true # Only allocate memory from nodes, this only works on GB200 ctx_max_seq_len=$((isl + 10)) gen_max_seq_len=$((isl + osl + 10)) @@ -96,6 +94,9 @@ logdir=${workdir}/slurm-${SLURM_JOB_ID}/benchmark-${isl}-${osl} mkdir -p ${logdir} full_logdir=${logdir}/ctx${num_ctx_servers}_gen${num_gen_servers}_dep${gen_tp_size}_batch${gen_batch_size}_eplb${eplb_num_slots}_mtp${mtp_size} +# nsys_on="" +nsys_on=${full_logdir} # Uncomment this line to enable Nsys profiling + echo "concurrency: ${concurrency}" ctx_gpus=$((num_ctx_servers * ctx_tp_size * ctx_pp_size)) diff --git a/examples/disaggregated/slurm/benchmark/start_worker.sh b/examples/disaggregated/slurm/benchmark/start_worker.sh index 1a5041b1e33c..9c2f48860e36 100644 --- a/examples/disaggregated/slurm/benchmark/start_worker.sh +++ b/examples/disaggregated/slurm/benchmark/start_worker.sh @@ -65,14 +65,14 @@ else nsys_file=${nsys_folder}/nsys_worker_proc_${instance_id}_${SLURM_PROCID} export TLLM_PROFILE_RECORD_GC=1 export TLLM_NVTX_DEBUG=1 - if [ "${role}" = "GEN" ]; then + if [ "${role}" = "GEN" ] && [ "$SLURM_PROCID" = "0" ]; then export TLLM_PROFILE_START_STOP=200-250 nsys_prefix="nsys profile -e \"NSYS_MPI_STORE_TEAMS_PER_RANK=1\" -o ${nsys_file} -f true -t cuda,nvtx,python-gil -c cudaProfilerApi --cuda-graph-trace node --capture-range-end=stop --gpu-metrics-devices=none" echo "nsys_prefix: ${nsys_prefix}" elif [ "${role}" = "CTX" ]; then echo "nsys is not enabled on ctx_gpus" fi - trtllm-llmapi-launch ${numa_bind_cmd} ${nsys_prefix} \ + ${nsys_prefix} trtllm-llmapi-launch ${numa_bind_cmd} \ trtllm-serve ${model_path} \ --host $(hostname) --port ${port} \ --extra_llm_api_options ${config_file} diff --git a/tensorrt_llm/_torch/pyexecutor/py_executor.py b/tensorrt_llm/_torch/pyexecutor/py_executor.py index 4b3315560f86..e27bc9b4b625 100644 --- a/tensorrt_llm/_torch/pyexecutor/py_executor.py +++ b/tensorrt_llm/_torch/pyexecutor/py_executor.py @@ -561,9 +561,9 @@ def profile_step(): if enable_torch_trace: torch_profiler.stop() torch_profiler.export_chrome_trace(torch_trace_path) - logger.info(f"Profiling stopped at iteration {it}, " - f"trace saved to {torch_trace_path}") + logger.info(f"Torch profiler trace saved to {torch_trace_path}.") torch.cuda.cudart().cudaProfilerStop() + logger.info(f"Profiling stopped at iteration {it}.") def _get_init_iter_stats(self, num_new_active_requests, new_active_requests_queue_latency_ms): From f8cb1378accfcd3b7c4c035fca487eeff87cf7d7 Mon Sep 17 00:00:00 2001 From: Kaiyu Xie <26294424+kaiyux@users.noreply.github.com> Date: Sun, 31 Aug 2025 22:19:58 -0700 Subject: [PATCH 3/5] Minor update Signed-off-by: Kaiyu Xie <26294424+kaiyux@users.noreply.github.com> --- examples/disaggregated/slurm/benchmark/disaggr_torch.slurm | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/examples/disaggregated/slurm/benchmark/disaggr_torch.slurm b/examples/disaggregated/slurm/benchmark/disaggr_torch.slurm index 303c2689c4cc..c29f995f3c68 100644 --- a/examples/disaggregated/slurm/benchmark/disaggr_torch.slurm +++ b/examples/disaggregated/slurm/benchmark/disaggr_torch.slurm @@ -94,8 +94,8 @@ logdir=${workdir}/slurm-${SLURM_JOB_ID}/benchmark-${isl}-${osl} mkdir -p ${logdir} full_logdir=${logdir}/ctx${num_ctx_servers}_gen${num_gen_servers}_dep${gen_tp_size}_batch${gen_batch_size}_eplb${eplb_num_slots}_mtp${mtp_size} -# nsys_on="" -nsys_on=${full_logdir} # Uncomment this line to enable Nsys profiling +nsys_on="" +# nsys_on=${full_logdir} # Uncomment this line to enable Nsys profiling echo "concurrency: ${concurrency}" From 48c3760e63d147a65be1769a5733536b65675203 Mon Sep 17 00:00:00 2001 From: Kaiyu Xie <26294424+kaiyux@users.noreply.github.com> Date: Sun, 31 Aug 2025 23:07:40 -0700 Subject: [PATCH 4/5] Revert "[None][chore] Bump version to 1.1.0rc2 (#7394)" This reverts commit ec595a8e297b0bb80e1b41daf2f4ea28d0dbcd93. --- README.md | 2 +- examples/constraints.txt | 2 +- tensorrt_llm/version.py | 2 +- 3 files changed, 3 insertions(+), 3 deletions(-) diff --git a/README.md b/README.md index 2b9c95586d3d..e4f4bd230cec 100644 --- a/README.md +++ b/README.md @@ -9,7 +9,7 @@ TensorRT-LLM [![python](https://img.shields.io/badge/python-3.10-green)](https://www.python.org/downloads/release/python-31012/) [![cuda](https://img.shields.io/badge/cuda-12.9.1-green)](https://developer.nvidia.com/cuda-downloads) [![trt](https://img.shields.io/badge/TRT-10.11.0-green)](https://developer.nvidia.com/tensorrt) -[![version](https://img.shields.io/badge/release-1.1.0rc3-green)](./tensorrt_llm/version.py) +[![version](https://img.shields.io/badge/release-1.1.0rc2-green)](./tensorrt_llm/version.py) [![license](https://img.shields.io/badge/license-Apache%202-blue)](./LICENSE) [Architecture](./docs/source/torch/arch_overview.md)   |   [Performance](./docs/source/performance/perf-overview.md)   |   [Examples](https://nvidia.github.io/TensorRT-LLM/quick-start-guide.html)   |   [Documentation](./docs/source/)   |   [Roadmap](https://github.com/NVIDIA/TensorRT-LLM/issues?q=is%3Aissue%20state%3Aopen%20label%3Aroadmap) diff --git a/examples/constraints.txt b/examples/constraints.txt index 527f9c8201b8..8b0d1a009303 100644 --- a/examples/constraints.txt +++ b/examples/constraints.txt @@ -1,3 +1,3 @@ -tensorrt_llm==1.1.0rc3 +tensorrt_llm==1.1.0rc2 evaluate~=0.4.1 rouge_score~=0.1.2 diff --git a/tensorrt_llm/version.py b/tensorrt_llm/version.py index 2059111b7cb7..93b6027df5cc 100644 --- a/tensorrt_llm/version.py +++ b/tensorrt_llm/version.py @@ -12,4 +12,4 @@ # WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. # See the License for the specific language governing permissions and # limitations under the License. -__version__ = "1.1.0rc3" +__version__ = "1.1.0rc2" From 2fe7013de79357940e5649074d900af19af493c5 Mon Sep 17 00:00:00 2001 From: Kaiyu Xie <26294424+kaiyux@users.noreply.github.com> Date: Sun, 31 Aug 2025 23:31:00 -0700 Subject: [PATCH 5/5] Update Signed-off-by: Kaiyu Xie <26294424+kaiyux@users.noreply.github.com> --- tensorrt_llm/_torch/pyexecutor/py_executor.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/tensorrt_llm/_torch/pyexecutor/py_executor.py b/tensorrt_llm/_torch/pyexecutor/py_executor.py index e27bc9b4b625..4b3315560f86 100644 --- a/tensorrt_llm/_torch/pyexecutor/py_executor.py +++ b/tensorrt_llm/_torch/pyexecutor/py_executor.py @@ -561,9 +561,9 @@ def profile_step(): if enable_torch_trace: torch_profiler.stop() torch_profiler.export_chrome_trace(torch_trace_path) - logger.info(f"Torch profiler trace saved to {torch_trace_path}.") + logger.info(f"Profiling stopped at iteration {it}, " + f"trace saved to {torch_trace_path}") torch.cuda.cudart().cudaProfilerStop() - logger.info(f"Profiling stopped at iteration {it}.") def _get_init_iter_stats(self, num_new_active_requests, new_active_requests_queue_latency_ms):