From 958759b83ba13e46df3ee03abc5741cb00f819b1 Mon Sep 17 00:00:00 2001 From: David Fan Date: Fri, 25 Sep 2026 21:06:49 +0000 Subject: [PATCH 1/5] Add GPT-OSS mixed-width QMoE recipe --- .../int4_cuda_int2_int4_qmoe/README.md | 40 +++++++++++++++++++ .../int4_cuda_int2_int4_qmoe/gpt-oss-20b.sh | 15 +++++++ 2 files changed, 55 insertions(+) create mode 100644 gpt-oss-20b/int4_cuda_int2_int4_qmoe/README.md create mode 100644 gpt-oss-20b/int4_cuda_int2_int4_qmoe/gpt-oss-20b.sh diff --git a/gpt-oss-20b/int4_cuda_int2_int4_qmoe/README.md b/gpt-oss-20b/int4_cuda_int2_int4_qmoe/README.md new file mode 100644 index 000000000..08879bd00 --- /dev/null +++ b/gpt-oss-20b/int4_cuda_int2_int4_qmoe/README.md @@ -0,0 +1,40 @@ +# Export GPT-OSS-20B with Mixed-Width CUDA QMoE + +This recipe exports the dense weights as INT4 and requantizes the GPT-OSS +MXFP4 experts to symmetric mixed-width QMoE weights: + +- gate/up projections (FC1/FC3): INT2 +- down projection (FC2): INT4 +- QMoE block size: 64 + +The generated model targets the CUDA execution provider. ONNX Runtime performs +the final expert-weight prepacking when the model is loaded. + +## Prerequisites + +Install the latest Olive and ONNX Runtime GenAI CUDA packages: + +```bash +python -m pip install -r ../requirements.txt +``` + +Until the changes are available in nightly packages, build from the branches in: + +- [ONNX Runtime GenAI #2624](https://github.com/microsoft/onnxruntime-genai/pull/2624) +- [ONNX Runtime #32761](https://github.com/microsoft/onnxruntime/pull/32761) + +## Export + +```bash +./gpt-oss-20b.sh +``` + +The exported model is saved in `int4_cuda_int2_int4_qmoe`. + +## Run + +Use the ONNX Runtime GenAI sample chat application: + +```bash +python model-chat.py -m int4_cuda_int2_int4_qmoe/model -e cuda +``` \ No newline at end of file diff --git a/gpt-oss-20b/int4_cuda_int2_int4_qmoe/gpt-oss-20b.sh b/gpt-oss-20b/int4_cuda_int2_int4_qmoe/gpt-oss-20b.sh new file mode 100644 index 000000000..6db91ad0f --- /dev/null +++ b/gpt-oss-20b/int4_cuda_int2_int4_qmoe/gpt-oss-20b.sh @@ -0,0 +1,15 @@ +#!/bin/bash + +olive capture-onnx-graph \ + --model_name_or_path openai/gpt-oss-20b \ + --trust_remote_code \ + --execution_provider CUDAExecutionProvider \ + --precision int4 \ + --use_model_builder \ + --use_ort_genai \ + --extra_mb_options op_types_to_quantize=MatMul/Gather \ + moe_quant_type=int4 \ + qmoe_fc1_type=int2 \ + qmoe_fc2_type=int4 \ + qmoe_block_size=64 \ + -o int4_cuda_int2_int4_qmoe \ No newline at end of file From 3145bbb83867c422be6a029ac0636cead68c2b9e Mon Sep 17 00:00:00 2001 From: David Fan Date: Sun, 27 Sep 2026 14:09:55 +0000 Subject: [PATCH 2/5] Fix GPT-OSS mixed-width QMoE recipe --- gpt-oss-20b/int4_cuda_int2_int4_qmoe/README.md | 7 +++++-- .../int4_cuda_int2_int4_qmoe/gpt-oss-20b.sh | 10 ++++------ gpt-oss-20b/int4_cuda_int2_int4_qmoe/info.yml | 9 +++++++++ .../target-options.json | 17 +++++++++++++++++ 4 files changed, 35 insertions(+), 8 deletions(-) create mode 100644 gpt-oss-20b/int4_cuda_int2_int4_qmoe/info.yml create mode 100644 gpt-oss-20b/int4_cuda_int2_int4_qmoe/target-options.json diff --git a/gpt-oss-20b/int4_cuda_int2_int4_qmoe/README.md b/gpt-oss-20b/int4_cuda_int2_int4_qmoe/README.md index 08879bd00..7d566eb23 100644 --- a/gpt-oss-20b/int4_cuda_int2_int4_qmoe/README.md +++ b/gpt-oss-20b/int4_cuda_int2_int4_qmoe/README.md @@ -10,6 +10,9 @@ MXFP4 experts to symmetric mixed-width QMoE weights: The generated model targets the CUDA execution provider. ONNX Runtime performs the final expert-weight prepacking when the model is loaded. +The recipe uses the structured mixed-width QMoE configuration introduced by +[ONNX Runtime GenAI #2624](https://github.com/microsoft/onnxruntime-genai/pull/2624). + ## Prerequisites Install the latest Olive and ONNX Runtime GenAI CUDA packages: @@ -26,7 +29,7 @@ Until the changes are available in nightly packages, build from the branches in: ## Export ```bash -./gpt-oss-20b.sh +bash gpt-oss-20b.sh ``` The exported model is saved in `int4_cuda_int2_int4_qmoe`. @@ -37,4 +40,4 @@ Use the ONNX Runtime GenAI sample chat application: ```bash python model-chat.py -m int4_cuda_int2_int4_qmoe/model -e cuda -``` \ No newline at end of file +``` diff --git a/gpt-oss-20b/int4_cuda_int2_int4_qmoe/gpt-oss-20b.sh b/gpt-oss-20b/int4_cuda_int2_int4_qmoe/gpt-oss-20b.sh index 6db91ad0f..f2cd014f8 100644 --- a/gpt-oss-20b/int4_cuda_int2_int4_qmoe/gpt-oss-20b.sh +++ b/gpt-oss-20b/int4_cuda_int2_int4_qmoe/gpt-oss-20b.sh @@ -1,5 +1,7 @@ #!/bin/bash +SCRIPT_DIR=$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" && pwd) + olive capture-onnx-graph \ --model_name_or_path openai/gpt-oss-20b \ --trust_remote_code \ @@ -7,9 +9,5 @@ olive capture-onnx-graph \ --precision int4 \ --use_model_builder \ --use_ort_genai \ - --extra_mb_options op_types_to_quantize=MatMul/Gather \ - moe_quant_type=int4 \ - qmoe_fc1_type=int2 \ - qmoe_fc2_type=int4 \ - qmoe_block_size=64 \ - -o int4_cuda_int2_int4_qmoe \ No newline at end of file + --extra_mb_options "builder_config_version=2,target_options=${SCRIPT_DIR}/target-options.json" \ + -o int4_cuda_int2_int4_qmoe diff --git a/gpt-oss-20b/int4_cuda_int2_int4_qmoe/info.yml b/gpt-oss-20b/int4_cuda_int2_int4_qmoe/info.yml new file mode 100644 index 000000000..f7217b3b1 --- /dev/null +++ b/gpt-oss-20b/int4_cuda_int2_int4_qmoe/info.yml @@ -0,0 +1,9 @@ +keywords: + - olive-ai +recipes: + - name: gpt-oss-20b + file: gpt-oss-20b.sh + eps: + - CUDAExecutionProvider + device: + - gpu diff --git a/gpt-oss-20b/int4_cuda_int2_int4_qmoe/target-options.json b/gpt-oss-20b/int4_cuda_int2_int4_qmoe/target-options.json new file mode 100644 index 000000000..6131452b4 --- /dev/null +++ b/gpt-oss-20b/int4_cuda_int2_int4_qmoe/target-options.json @@ -0,0 +1,17 @@ +{ + "quant_config": { + "weights": { + "type": "int4", + "op_types": [ + "MatMul", + "Gather" + ] + }, + "moe": { + "type": "int4", + "fc1_type": "int2", + "fc2_type": "int4", + "block_size": 64 + } + } +} From 97be9adb2472546152147fb00241382dcc27e038 Mon Sep 17 00:00:00 2001 From: David Fan Date: Mon, 28 Sep 2026 03:54:55 +0000 Subject: [PATCH 3/5] Document GPT-OSS mixed QMoE runtime limits --- .../int4_cuda_int2_int4_qmoe/README.md | 34 +++++++++++++------ gpt-oss-20b/requirements.txt | 2 ++ 2 files changed, 26 insertions(+), 10 deletions(-) diff --git a/gpt-oss-20b/int4_cuda_int2_int4_qmoe/README.md b/gpt-oss-20b/int4_cuda_int2_int4_qmoe/README.md index 7d566eb23..7d606b834 100644 --- a/gpt-oss-20b/int4_cuda_int2_int4_qmoe/README.md +++ b/gpt-oss-20b/int4_cuda_int2_int4_qmoe/README.md @@ -11,7 +11,9 @@ The generated model targets the CUDA execution provider. ONNX Runtime performs the final expert-weight prepacking when the model is loaded. The recipe uses the structured mixed-width QMoE configuration introduced by -[ONNX Runtime GenAI #2624](https://github.com/microsoft/onnxruntime-genai/pull/2624). +[ONNX Runtime GenAI #2624](https://github.com/microsoft/onnxruntime-genai/pull/2624) +and the packed CUDA decode support introduced by +[ONNX Runtime #32761](https://github.com/microsoft/onnxruntime/pull/32761). ## Prerequisites @@ -21,10 +23,9 @@ Install the latest Olive and ONNX Runtime GenAI CUDA packages: python -m pip install -r ../requirements.txt ``` -Until the changes are available in nightly packages, build from the branches in: - -- [ONNX Runtime GenAI #2624](https://github.com/microsoft/onnxruntime-genai/pull/2624) -- [ONNX Runtime #32761](https://github.com/microsoft/onnxruntime/pull/32761) +Both changes have been merged. Use package versions that include them. Exporting +the official GPT-OSS checkpoint also requires an ONNX Runtime GenAI build that +loads expert tensors from the checkpoint's `model.layers.*.mlp.experts` keys. ## Export @@ -34,10 +35,23 @@ bash gpt-oss-20b.sh The exported model is saved in `int4_cuda_int2_int4_qmoe`. -## Run +The resulting GPT-OSS-20B model contains 24 mixed-width QMoE nodes with INT2 +gate/up projections, INT4 down projections, and block size 64. -Use the ONNX Runtime GenAI sample chat application: +## Runtime Status -```bash -python model-chat.py -m int4_cuda_int2_int4_qmoe/model -e cuda -``` +The packed INT2/INT4 CUDA kernel currently targets decode workloads with at most +eight expanded rows. GPT-OSS routes each token to four experts, so this covers up +to two input tokens per QMoE invocation. Longer prefill inputs currently select +the dense dequantization fallback and can exceed its default scratch-memory +limit. + +Model loading and token-by-token decode are supported, but the standard +`model-chat.py` application performs multi-token prefill and is not supported by +this recipe until ONNX Runtime provides a bounded mixed-width prefill path. Do +not raise `ep.cuda.qmoe_int_dequant_max_scratch_bytes` as a production +workaround because that fallback materializes the expert weights. + +The model directory passed to ONNX Runtime GenAI is +`int4_cuda_int2_int4_qmoe` (the directory directly containing `model.onnx` and +`genai_config.json`). diff --git a/gpt-oss-20b/requirements.txt b/gpt-oss-20b/requirements.txt index c9c5334a0..327ea97e7 100644 --- a/gpt-oss-20b/requirements.txt +++ b/gpt-oss-20b/requirements.txt @@ -1,3 +1,5 @@ --extra-index-url https://aiinfra.pkgs.visualstudio.com/PublicPackages/_packaging/ORT-Nightly/pypi/simple olive-ai onnxruntime-genai-cuda +accelerate +kernels>=0.16,<0.17 From 37f91b22bd74c7bb557cf351c7f791fde01901da Mon Sep 17 00:00:00 2001 From: David Fan Date: Mon, 28 Sep 2026 04:16:27 +0000 Subject: [PATCH 4/5] Fix GPT-OSS requirements ordering --- gpt-oss-20b/requirements.txt | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/gpt-oss-20b/requirements.txt b/gpt-oss-20b/requirements.txt index 327ea97e7..e74fafb51 100644 --- a/gpt-oss-20b/requirements.txt +++ b/gpt-oss-20b/requirements.txt @@ -1,5 +1,5 @@ --extra-index-url https://aiinfra.pkgs.visualstudio.com/PublicPackages/_packaging/ORT-Nightly/pypi/simple -olive-ai -onnxruntime-genai-cuda accelerate kernels>=0.16,<0.17 +olive-ai +onnxruntime-genai-cuda From 0accea38caccc9266a3bbcdb17be389998804ec5 Mon Sep 17 00:00:00 2001 From: David Fan Date: Mon, 28 Sep 2026 18:28:49 +0000 Subject: [PATCH 5/5] Address mixed QMoE recipe review --- gpt-oss-20b/int4_cuda_int2_int4_qmoe/README.md | 16 +++++++++++++++- .../int4_cuda_int2_int4_qmoe/gpt-oss-20b.sh | 13 ------------- gpt-oss-20b/int4_cuda_int2_int4_qmoe/info.yml | 2 +- 3 files changed, 16 insertions(+), 15 deletions(-) delete mode 100644 gpt-oss-20b/int4_cuda_int2_int4_qmoe/gpt-oss-20b.sh diff --git a/gpt-oss-20b/int4_cuda_int2_int4_qmoe/README.md b/gpt-oss-20b/int4_cuda_int2_int4_qmoe/README.md index 7d606b834..3063a1017 100644 --- a/gpt-oss-20b/int4_cuda_int2_int4_qmoe/README.md +++ b/gpt-oss-20b/int4_cuda_int2_int4_qmoe/README.md @@ -23,14 +23,28 @@ Install the latest Olive and ONNX Runtime GenAI CUDA packages: python -m pip install -r ../requirements.txt ``` +The `accelerate` and `kernels` packages in the shared requirements are needed +to load the official GPT-OSS MXFP4 checkpoint through Transformers during +export. They are not dependencies of the exported ONNX model. + Both changes have been merged. Use package versions that include them. Exporting the official GPT-OSS checkpoint also requires an ONNX Runtime GenAI build that loads expert tensors from the checkpoint's `model.layers.*.mlp.experts` keys. ## Export +Run the command from this recipe directory: + ```bash -bash gpt-oss-20b.sh +olive capture-onnx-graph \ + --model_name_or_path openai/gpt-oss-20b \ + --trust_remote_code \ + --execution_provider CUDAExecutionProvider \ + --precision int4 \ + --use_model_builder \ + --use_ort_genai \ + --extra_mb_options "builder_config_version=2,target_options=target-options.json" \ + -o int4_cuda_int2_int4_qmoe ``` The exported model is saved in `int4_cuda_int2_int4_qmoe`. diff --git a/gpt-oss-20b/int4_cuda_int2_int4_qmoe/gpt-oss-20b.sh b/gpt-oss-20b/int4_cuda_int2_int4_qmoe/gpt-oss-20b.sh deleted file mode 100644 index f2cd014f8..000000000 --- a/gpt-oss-20b/int4_cuda_int2_int4_qmoe/gpt-oss-20b.sh +++ /dev/null @@ -1,13 +0,0 @@ -#!/bin/bash - -SCRIPT_DIR=$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" && pwd) - -olive capture-onnx-graph \ - --model_name_or_path openai/gpt-oss-20b \ - --trust_remote_code \ - --execution_provider CUDAExecutionProvider \ - --precision int4 \ - --use_model_builder \ - --use_ort_genai \ - --extra_mb_options "builder_config_version=2,target_options=${SCRIPT_DIR}/target-options.json" \ - -o int4_cuda_int2_int4_qmoe diff --git a/gpt-oss-20b/int4_cuda_int2_int4_qmoe/info.yml b/gpt-oss-20b/int4_cuda_int2_int4_qmoe/info.yml index f7217b3b1..8744f80cc 100644 --- a/gpt-oss-20b/int4_cuda_int2_int4_qmoe/info.yml +++ b/gpt-oss-20b/int4_cuda_int2_int4_qmoe/info.yml @@ -2,7 +2,7 @@ keywords: - olive-ai recipes: - name: gpt-oss-20b - file: gpt-oss-20b.sh + file: README.md eps: - CUDAExecutionProvider device: