From 8cc40578336d26b2ed4805d69480022ee17c17af Mon Sep 17 00:00:00 2001 From: Victor Oliveira Date: Mon, 14 Sep 2026 18:59:04 +0000 Subject: [PATCH] Fix Qwen3 ONNX export bug related to 0/1 specialization --- tensorrt_edgellm/models/default/modeling_default.py | 8 +++++--- 1 file changed, 5 insertions(+), 3 deletions(-) diff --git a/tensorrt_edgellm/models/default/modeling_default.py b/tensorrt_edgellm/models/default/modeling_default.py index 44fb3c0d7..f12b9db0d 100644 --- a/tensorrt_edgellm/models/default/modeling_default.py +++ b/tensorrt_edgellm/models/default/modeling_default.py @@ -68,9 +68,11 @@ # ONNX export spec # --------------------------------------------------------------------------- -_BATCH_SIZE = 1 -_SEQ_LEN = 1 -_PAST_LEN = 1 +# PyTorch applies 0/1 specialization which breaks ONNX dynamic batching +# Using size = 2 avoids this issue +_BATCH_SIZE = 2 +_SEQ_LEN = 2 +_PAST_LEN = 2 _MAX_POS = 4096