Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Original file line number Diff line number Diff line change
Expand Up @@ -30,6 +30,7 @@ resources:
backend:
type: trtllm
prefill_environment:
OMP_NUM_THREADS: '1'
TLLM_LOG_LEVEL: INFO
TRTLLM_SERVER_DISABLE_GC: '1'
TRTLLM_WORKER_DISABLE_GC: '1'
Expand All @@ -43,6 +44,7 @@ backend:
PYTORCH_CUDA_ALLOC_CONF: expandable_segments:True
CUDA_SCALE_LAUNCH_QUEUES: 4x
decode_environment:
OMP_NUM_THREADS: '1'
TLLM_LOG_LEVEL: INFO
TRTLLM_SERVER_DISABLE_GC: '1'
TRTLLM_WORKER_DISABLE_GC: '1'
Expand Down Expand Up @@ -147,7 +149,6 @@ frontend:
PRECISION: fp4
CONC: '388'
DURATION: '3600'
KV_OFFLOADING: none
ETCD_LEASE_TTL: '120'
DYN_ROUTER_QUEUE_THRESHOLD: None
DYN_TOKENIZER_CACHE: '1'
Expand All @@ -173,7 +174,6 @@ benchmark:
PRECISION: fp4
CONC: '388'
DURATION: '3600'
KV_OFFLOADING: none
INFMAX_CONTAINER_WORKSPACE: /infmax-workspace
RESULT_DIR: /logs/agentic
PORT: '8000'
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -30,6 +30,7 @@ resources:
backend:
type: trtllm
prefill_environment:
OMP_NUM_THREADS: '1'
TLLM_LOG_LEVEL: INFO
TRTLLM_SERVER_DISABLE_GC: '1'
TRTLLM_WORKER_DISABLE_GC: '1'
Expand All @@ -43,6 +44,7 @@ backend:
PYTORCH_CUDA_ALLOC_CONF: expandable_segments:True
CUDA_SCALE_LAUNCH_QUEUES: 4x
decode_environment:
OMP_NUM_THREADS: '1'
TLLM_LOG_LEVEL: INFO
TRTLLM_SERVER_DISABLE_GC: '1'
TRTLLM_WORKER_DISABLE_GC: '1'
Expand Down Expand Up @@ -147,7 +149,6 @@ frontend:
PRECISION: fp4
CONC: '4'
DURATION: '3600'
KV_OFFLOADING: none
ETCD_LEASE_TTL: '120'
DYN_ROUTER_QUEUE_THRESHOLD: None
DYN_TOKENIZER_CACHE: '1'
Expand All @@ -173,7 +174,6 @@ benchmark:
PRECISION: fp4
CONC: '4'
DURATION: '3600'
KV_OFFLOADING: none
INFMAX_CONTAINER_WORKSPACE: /infmax-workspace
RESULT_DIR: /logs/agentic
PORT: '8000'
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -30,6 +30,7 @@ resources:
backend:
type: trtllm
prefill_environment:
OMP_NUM_THREADS: '1'
TLLM_LOG_LEVEL: INFO
TRTLLM_SERVER_DISABLE_GC: '1'
TRTLLM_WORKER_DISABLE_GC: '1'
Expand All @@ -43,6 +44,7 @@ backend:
PYTORCH_CUDA_ALLOC_CONF: expandable_segments:True
CUDA_SCALE_LAUNCH_QUEUES: 4x
decode_environment:
OMP_NUM_THREADS: '1'
TLLM_LOG_LEVEL: INFO
TRTLLM_SERVER_DISABLE_GC: '1'
TRTLLM_WORKER_DISABLE_GC: '1'
Expand Down Expand Up @@ -147,7 +149,6 @@ frontend:
PRECISION: fp4
CONC: '24'
DURATION: '3600'
KV_OFFLOADING: none
ETCD_LEASE_TTL: '120'
DYN_ROUTER_QUEUE_THRESHOLD: None
DYN_TOKENIZER_CACHE: '1'
Expand All @@ -173,7 +174,6 @@ benchmark:
PRECISION: fp4
CONC: '24'
DURATION: '3600'
KV_OFFLOADING: none
INFMAX_CONTAINER_WORKSPACE: /infmax-workspace
RESULT_DIR: /logs/agentic
PORT: '8000'
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -30,6 +30,7 @@ resources:
backend:
type: trtllm
prefill_environment:
OMP_NUM_THREADS: '1'
TLLM_LOG_LEVEL: INFO
TRTLLM_SERVER_DISABLE_GC: '1'
TRTLLM_WORKER_DISABLE_GC: '1'
Expand All @@ -43,6 +44,7 @@ backend:
PYTORCH_CUDA_ALLOC_CONF: expandable_segments:True
CUDA_SCALE_LAUNCH_QUEUES: 4x
decode_environment:
OMP_NUM_THREADS: '1'
TLLM_LOG_LEVEL: INFO
TRTLLM_SERVER_DISABLE_GC: '1'
TRTLLM_WORKER_DISABLE_GC: '1'
Expand Down Expand Up @@ -148,7 +150,6 @@ frontend:
PRECISION: fp4
CONC: '736'
DURATION: '3600'
KV_OFFLOADING: none
ETCD_LEASE_TTL: '120'
DYN_ROUTER_QUEUE_THRESHOLD: None
DYN_TOKENIZER_CACHE: '1'
Expand All @@ -174,7 +175,6 @@ benchmark:
PRECISION: fp4
CONC: '736'
DURATION: '3600'
KV_OFFLOADING: none
INFMAX_CONTAINER_WORKSPACE: /infmax-workspace
RESULT_DIR: /logs/agentic
PORT: '8000'
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -30,6 +30,7 @@ resources:
backend:
type: trtllm
prefill_environment:
OMP_NUM_THREADS: '1'
TLLM_LOG_LEVEL: INFO
TRTLLM_SERVER_DISABLE_GC: '1'
TRTLLM_WORKER_DISABLE_GC: '1'
Expand All @@ -43,6 +44,7 @@ backend:
PYTORCH_CUDA_ALLOC_CONF: expandable_segments:True
CUDA_SCALE_LAUNCH_QUEUES: 4x
decode_environment:
OMP_NUM_THREADS: '1'
TLLM_LOG_LEVEL: INFO
TRTLLM_SERVER_DISABLE_GC: '1'
TRTLLM_WORKER_DISABLE_GC: '1'
Expand Down Expand Up @@ -154,7 +156,6 @@ frontend:
PRECISION: fp4
CONC: '1152'
DURATION: '3600'
KV_OFFLOADING: none
ETCD_LEASE_TTL: '120'
DYN_ROUTER_QUEUE_THRESHOLD: None
DYN_TOKENIZER_CACHE: '1'
Expand All @@ -180,7 +181,6 @@ benchmark:
PRECISION: fp4
CONC: '1152'
DURATION: '3600'
KV_OFFLOADING: none
INFMAX_CONTAINER_WORKSPACE: /infmax-workspace
RESULT_DIR: /logs/agentic
PORT: '8000'
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -30,6 +30,7 @@ resources:
backend:
type: trtllm
prefill_environment:
OMP_NUM_THREADS: '1'
TLLM_LOG_LEVEL: INFO
TRTLLM_SERVER_DISABLE_GC: '1'
TRTLLM_WORKER_DISABLE_GC: '1'
Expand All @@ -43,6 +44,7 @@ backend:
PYTORCH_CUDA_ALLOC_CONF: expandable_segments:True
CUDA_SCALE_LAUNCH_QUEUES: 4x
decode_environment:
OMP_NUM_THREADS: '1'
TLLM_LOG_LEVEL: INFO
TRTLLM_SERVER_DISABLE_GC: '1'
TRTLLM_WORKER_DISABLE_GC: '1'
Expand Down Expand Up @@ -170,7 +172,6 @@ frontend:
PRECISION: fp4
CONC: '2626'
DURATION: '3600'
KV_OFFLOADING: none
ETCD_LEASE_TTL: '120'
DYN_ROUTER_QUEUE_THRESHOLD: None
DYN_TOKENIZER_CACHE: '1'
Expand All @@ -196,7 +197,6 @@ benchmark:
PRECISION: fp4
CONC: '2626'
DURATION: '3600'
KV_OFFLOADING: none
INFMAX_CONTAINER_WORKSPACE: /infmax-workspace
RESULT_DIR: /logs/agentic
PORT: '8000'
Expand Down
9 changes: 9 additions & 0 deletions perf-changelog.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -6547,3 +6547,12 @@
- "Filter AgentX traces at the same 202,752-token context limit used by both TileRT roles so oversized Weka trajectories are excluded before replay."
- "Pin SemiAnalysisAI/srt-slurm PR #10 commit d1e6c97b3baf3e87103b6d83189544c3c7d61c38, stacked on the AMD/native-router PR #7 and base runtime PR #1, including explicit native HTTP dependencies, GLM-5.1-compatible Transformers v5 router tokenization, incomplete-snapshot recovery, backend-declared conversion GPU resources, pre-container NVIDIA driver-hook activation, and lossless Slurm container-environment exports."
pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/2750

- config-keys:
- dsv4-fp4-gb300-dynamo-trt-agentx
scenario-type:
- agentic-coding
description:
- "Set OMP_NUM_THREADS=1 for the prefill and decode workers across the GB300 Dynamo-TensorRT-LLM AgentX recipe matrix."
- "Allow the frontend and benchmark environments to inherit the runtime KV offloading setting."
pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/2711
Loading