From 9367c767dbe6afac4e0f438eccd9017809c187c1 Mon Sep 17 00:00:00 2001 From: shaahji <96227573+shaahji@users.noreply.github.com> Date: Thu, 16 Jul 2026 17:28:23 -0700 Subject: [PATCH 01/10] [Scanner]: Ran scanner and update README.md (#555) Co-authored-by: jambayk --- README.md | 385 +++++++++++++++++++++++++++--------------------------- 1 file changed, 196 insertions(+), 189 deletions(-) diff --git a/README.md b/README.md index cca067ca6..3f9ddb3bf 100644 --- a/README.md +++ b/README.md @@ -20,8 +20,8 @@ Below are list of available recipes grouped by different criteria. Click the lin | bert | clip | deepseek | gemma | hiera | llama | llama3 | mistral | mobilenet | phi3 | phi4 | qwen2 | resnet | sam | sd | vit | whisper | | :---: | :---: | :---: | :---: | :---: | :---: | :---: | :---: | :---: | :---: | :---: | :---: | :---: | :---: | :---: | :---: | :---: | | [google-bert-bert-base-multilingual-cased](google-bert-bert-base-multilingual-cased/QNN) | [OFA-Sys-chinese-clip-vit-base-patch16](OFA-Sys-chinese-clip-vit-base-patch16/aitk) | [deepseek-ai-DeepSeek-R1-Distill-Llama-8B](deepseek-ai-DeepSeek-R1-Distill-Llama-8B/aitk) | [google-gemma-3-1b-it](google-gemma-3-1b-it/OpenVINO) | [sam2.1-hiera-small](sam2.1-hiera-small/QNN) | [deepseek-ai-DeepSeek-R1-Distill-Llama-8B](deepseek-ai-DeepSeek-R1-Distill-Llama-8B/NvTensorRtRtx) | [meta-llama-Llama-3.1-8B-Instruct](meta-llama-Llama-3.1-8B-Instruct/aitk) | [mistralai-Mistral-7B-Instruct-v0.2](mistralai-Mistral-7B-Instruct-v0.2/NvTensorRtRtx) | [timm-mobilenetv3_small_100.lamb_in1k](timm-mobilenetv3_small_100.lamb_in1k/VitisAI) | [microsoft-Phi-3-mini-128k-instruct](microsoft-Phi-3-mini-128k-instruct/NvTensorRtRtx) | [microsoft-Phi-4-mini-instruct](microsoft-Phi-4-mini-instruct/NvTensorRtRtx) | [Qwen-Qwen2.5-0.5B-Instruct](Qwen-Qwen2.5-0.5B-Instruct/NvTensorRtRtx) | [microsoft-resnet-50](microsoft-resnet-50/aitk) | [sam-vit-base](sam-vit-base/aitk) | [sd-legacy-stable-diffusion-v1-5](sd-legacy-stable-diffusion-v1-5/aitk) | [google-vit-base-patch16-224](google-vit-base-patch16-224/OpenVINO) | [openai-whisper-large-v3-turbo](openai-whisper-large-v3-turbo/OpenVINO) | -| [google-bert-bert-base-multilingual-cased](google-bert-bert-base-multilingual-cased/aitk) | [laion-CLIP-ViT-B-32-laion2B-s34B-b79K](laion-CLIP-ViT-B-32-laion2B-s34B-b79K/QNN) | [deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B](deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B/QNN) | | | [meta-llama-Llama-3.1-8B-Instruct](meta-llama-Llama-3.1-8B-Instruct/NvTensorRtRtx) | [meta-llama-Llama-3.2-1B-Instruct](meta-llama-Llama-3.2-1B-Instruct/QNN) | [mistralai-Mistral-7B-Instruct-v0.2](mistralai-Mistral-7B-Instruct-v0.2/aitk) | | [microsoft-Phi-3-mini-128k-instruct](microsoft-Phi-3-mini-128k-instruct/QNN) | [microsoft-Phi-4-mini-instruct](microsoft-Phi-4-mini-instruct/aitk) | [Qwen-Qwen2.5-0.5B-Instruct](Qwen-Qwen2.5-0.5B-Instruct/aitk) | | [sam2.1-hiera-small](sam2.1-hiera-small/aitk) | [sd2-community-stable-diffusion-2-1](sd2-community-stable-diffusion-2-1/aitk) | [google-vit-base-patch16-224](google-vit-base-patch16-224/QNN) | [openai-whisper-large-v3-turbo](openai-whisper-large-v3-turbo/VitisAI) | -| [intel-bert-base-uncased-mrpc](intel-bert-base-uncased-mrpc/QNN) | [laion-CLIP-ViT-B-32-laion2B-s34B-b79K](laion-CLIP-ViT-B-32-laion2B-s34B-b79K/aitk) | [deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B](deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B/aitk) | | | [meta-llama-Llama-3.2-1B-Instruct](meta-llama-Llama-3.2-1B-Instruct/NvTensorRtRtx) | [meta-llama-Llama-3.2-1B-Instruct](meta-llama-Llama-3.2-1B-Instruct/aitk) | [mistralai-Mistral-7B-Instruct-v0.3](mistralai-Mistral-7B-Instruct-v0.3/aitk) | | [microsoft-Phi-3-mini-128k-instruct](microsoft-Phi-3-mini-128k-instruct/aitk) | [microsoft-Phi-4-mini-instruct](microsoft-Phi-4-mini-instruct/olive) | [Qwen-Qwen2.5-0.5B](Qwen-Qwen2.5-0.5B/aitk) | | | | [google-vit-base-patch16-224](google-vit-base-patch16-224/aitk) | [openai-whisper-large-v3-turbo](openai-whisper-large-v3-turbo/aitk) | +| [google-bert-bert-base-multilingual-cased](google-bert-bert-base-multilingual-cased/aitk) | [laion-CLIP-ViT-B-32-laion2B-s34B-b79K](laion-CLIP-ViT-B-32-laion2B-s34B-b79K/QNN) | [deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B](deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B/QNN) | [google-gemma-4-E2B-it](google-gemma-4-E2B-it) | | [meta-llama-Llama-3.1-8B-Instruct](meta-llama-Llama-3.1-8B-Instruct/NvTensorRtRtx) | [meta-llama-Llama-3.2-1B-Instruct](meta-llama-Llama-3.2-1B-Instruct/QNN) | [mistralai-Mistral-7B-Instruct-v0.2](mistralai-Mistral-7B-Instruct-v0.2/aitk) | | [microsoft-Phi-3-mini-128k-instruct](microsoft-Phi-3-mini-128k-instruct/QNN) | [microsoft-Phi-4-mini-instruct](microsoft-Phi-4-mini-instruct/aitk) | [Qwen-Qwen2.5-0.5B-Instruct](Qwen-Qwen2.5-0.5B-Instruct/aitk) | | [sam2.1-hiera-small](sam2.1-hiera-small/OpenVINO) | [sd2-community-stable-diffusion-2-1](sd2-community-stable-diffusion-2-1/aitk) | [google-vit-base-patch16-224](google-vit-base-patch16-224/QNN) | [openai-whisper-large-v3-turbo](openai-whisper-large-v3-turbo/VitisAI) | +| [intel-bert-base-uncased-mrpc](intel-bert-base-uncased-mrpc/QNN) | [laion-CLIP-ViT-B-32-laion2B-s34B-b79K](laion-CLIP-ViT-B-32-laion2B-s34B-b79K/aitk) | [deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B](deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B/aitk) | | | [meta-llama-Llama-3.2-1B-Instruct](meta-llama-Llama-3.2-1B-Instruct/NvTensorRtRtx) | [meta-llama-Llama-3.2-1B-Instruct](meta-llama-Llama-3.2-1B-Instruct/aitk) | [mistralai-Mistral-7B-Instruct-v0.3](mistralai-Mistral-7B-Instruct-v0.3/aitk) | | [microsoft-Phi-3-mini-128k-instruct](microsoft-Phi-3-mini-128k-instruct/aitk) | [microsoft-Phi-4-mini-instruct](microsoft-Phi-4-mini-instruct/olive) | [Qwen-Qwen2.5-0.5B](Qwen-Qwen2.5-0.5B/aitk) | | [sam2.1-hiera-small](sam2.1-hiera-small/aitk) | | [google-vit-base-patch16-224](google-vit-base-patch16-224/aitk) | [openai-whisper-large-v3-turbo](openai-whisper-large-v3-turbo/aitk) | | [intel-bert-base-uncased-mrpc](intel-bert-base-uncased-mrpc/aitk) | [openai-clip-vit-base-patch16](openai-clip-vit-base-patch16/QNN) | [deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B](deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B/olive) | | | | [meta-llama-Llama-3.2-1B-Instruct](meta-llama-Llama-3.2-1B-Instruct/olive) | | | [microsoft-Phi-3-mini-4k-instruct](microsoft-Phi-3-mini-4k-instruct/NvTensorRtRtx) | [microsoft-Phi-4-mini-reasoning](microsoft-Phi-4-mini-reasoning/aitk) | [Qwen-Qwen2.5-1.5B-Instruct](Qwen-Qwen2.5-1.5B-Instruct/NvTensorRtRtx) | | | | [sam-vit-base](sam-vit-base/QNN) | [openai-whisper-large-v3-turbo](openai-whisper-large-v3-turbo/olive) | | | [openai-clip-vit-base-patch16](openai-clip-vit-base-patch16/aitk) | [deepseek-ai-DeepSeek-R1-Distill-Qwen-14B](deepseek-ai-DeepSeek-R1-Distill-Qwen-14B/aitk) | | | | [meta-llama-Meta-Llama-3-8B](meta-llama-Meta-Llama-3-8B/olive) | | | [microsoft-Phi-3-mini-4k-instruct](microsoft-Phi-3-mini-4k-instruct/QNN) | [microsoft-Phi-4-reasoning-plus](microsoft-Phi-4-reasoning-plus/aitk) | [Qwen-Qwen2.5-1.5B-Instruct](Qwen-Qwen2.5-1.5B-Instruct/QNN) | | | | | [openai-whisper-medium](openai-whisper-medium/VitisAI) | | | [openai-clip-vit-base-patch32](openai-clip-vit-base-patch32/QNN) | [deepseek-ai-DeepSeek-R1-Distill-Qwen-7B](deepseek-ai-DeepSeek-R1-Distill-Qwen-7B/aitk) | | | | | | | [microsoft-Phi-3-mini-4k-instruct](microsoft-Phi-3-mini-4k-instruct/aitk) | [microsoft-Phi-4-reasoning](microsoft-Phi-4-reasoning/aitk) | [Qwen-Qwen2.5-1.5B-Instruct](Qwen-Qwen2.5-1.5B-Instruct/aitk) | | | | | [openai-whisper-small](openai-whisper-small/VitisAI) | @@ -71,114 +71,118 @@ Below are list of available recipes grouped by different criteria. Click the lin | [deepseek-ai-DeepSeek-R1-Distill-Qwen-7B](deepseek-ai-DeepSeek-R1-Distill-Qwen-7B/aitk/deepseek_ov_config.json) | [Qwen-Qwen2.5-1.5B-Instruct](Qwen-Qwen2.5-1.5B-Instruct/QNN/config_gpu_ctxbin.json) | [Qwen-Qwen2.5-Coder-14B-Instruct](Qwen-Qwen2.5-Coder-14B-Instruct/aitk/qwen2_5_ov_npu_config.json) | | [facebook-opt-125m-splicegpt](facebook-opt-125m/olive/slicegpt.json) | [Qwen-Qwen2.5-1.5B-Instruct](Qwen-Qwen2.5-1.5B-Instruct/aitk/qwen2_5_dml_config.json) | [Qwen-Qwen2.5-Coder-3B-Instruct](Qwen-Qwen2.5-Coder-3B-Instruct/aitk/qwen2_5_ov_npu_config.json) | | [gemma-3-1b-it_model_builder_cpu_FP32](google-gemma/olive/gemma-3-1b-it_model_builder_cpu_fp32.json) | [Qwen-Qwen2.5-1.5B-Instruct](Qwen-Qwen2.5-1.5B-Instruct/aitk/qwen2_5_ov_gpu_config.json) | [Qwen-Qwen2.5-Coder-7B-Instruct](Qwen-Qwen2.5-Coder-7B-Instruct/aitk/qwen2_5_ov_npu_config.json) | -| [google-bert-bert-base-multilingual-cased](google-bert-bert-base-multilingual-cased/aitk/bert-base-multilingual-cased_context_ov_static.json) | [Qwen-Qwen2.5-1.5B-Instruct](Qwen-Qwen2.5-1.5B-Instruct/aitk/qwen2_5_qnn_gpu_config.json) | [Qwen-Qwen2.5-Coder-7B-Instruct](Qwen-Qwen2.5-Coder-7B-Instruct/aitk/qwen2_5_vitis_ai_config.json) | -| [google-gemma](google-gemma/olive/README.md) | [Qwen-Qwen2.5-1.5B-Instruct](Qwen-Qwen2.5-1.5B-Instruct/aitk/qwen2_5_trtrtx_config.json) | [deepseek-ai-DeepSeek-R1-Distill-Llama-8B](deepseek-ai-DeepSeek-R1-Distill-Llama-8B/aitk/deepseek_ov_npu_config.json) | -| [google-vit-base-patch16-224](google-vit-base-patch16-224/aitk/vit_base_patch16_224_context_ov_static.json) | [Qwen-Qwen2.5-14B-Instruct](Qwen-Qwen2.5-14B-Instruct/aitk/qwen2_5_ov_config.json) | [deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B](deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B/aitk/deepseek_ov_config.json) | -| [gpt-oss-20b](gpt-oss-20b/int4_cuda_int4_qmoe/gpt-oss-20b.sh) | [Qwen-Qwen2.5-14B-Instruct](Qwen-Qwen2.5-14B-Instruct/aitk/qwen2_5_trtrtx.json) | [deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B](deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B/aitk/deepseek_qnn_config.json) | -| [intel-bert-base-uncased-mrpc (ov)](intel-bert-base-uncased-mrpc/aitk/bert_ov.json) | [Qwen-Qwen2.5-3B-Instruct](Qwen-Qwen2.5-3B-Instruct/aitk/qwen2_5_ov_config.json) | [deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B](deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B/aitk/deepseek_vitis_ai_config.json) | -| [intel-bert-base-uncased-mrpc-inc-smooth-quant](intel-bert-base-uncased-mrpc/oci/cpu/inc_smooth_quant.json) | [Qwen-Qwen2.5-7B-Instruct](Qwen-Qwen2.5-7B-Instruct/aitk/qwen2_5_ov_config.json) | [deepseek-ai-DeepSeek-R1-Distill-Qwen-14B](deepseek-ai-DeepSeek-R1-Distill-Qwen-14B/aitk/deepseek_ov_npu_config.json) | -| [intel-bert-base-uncased-mrpc-ptq](intel-bert-base-uncased-mrpc/oci/cpu/ptq.json) | [Qwen-Qwen2.5-7B-Instruct](Qwen-Qwen2.5-7B-Instruct/aitk/qwen2_5_trtrtx.json) | [deepseek-ai-DeepSeek-R1-Distill-Qwen-7B](deepseek-ai-DeepSeek-R1-Distill-Qwen-7B/aitk/deepseek_ov_npu_config.json) | -| [laion-CLIP-ViT-B-32-laion2B-s34B-b79K](laion-CLIP-ViT-B-32-laion2B-s34B-b79K/aitk/laion_clip_ov.json) | [Qwen-Qwen2.5-Coder-0.5B-Instruct](Qwen-Qwen2.5-Coder-0.5B-Instruct/aitk/qwen2_5_ov_config.json) | [deepseek-ai-DeepSeek-R1-Distill-Qwen-7B](deepseek-ai-DeepSeek-R1-Distill-Qwen-7B/aitk/deepseek_vitis_ai_config.json) | -| [meta-llama-Llama-3.1-8B-Instruct](meta-llama-Llama-3.1-8B-Instruct/aitk/llama3_1_ov_gpu_config.json) | [Qwen-Qwen2.5-Coder-0.5B-Instruct](Qwen-Qwen2.5-Coder-0.5B-Instruct/aitk/qwen2_5_trtrtx.json) | [google-bert-bert-base-multilingual-cased](google-bert-bert-base-multilingual-cased/aitk/bert-base-multilingual-cased_context_ov_static.json) | -| [meta-llama-Llama-3.2-1B-Instruct-dora](meta-llama-Llama-3.2-1B-Instruct/olive/dora.json) | [Qwen-Qwen2.5-Coder-1.5B-Instruct](Qwen-Qwen2.5-Coder-1.5B-Instruct/aitk/qwen2_5_ov_config.json) | [google-bert-bert-base-multilingual-cased](google-bert-bert-base-multilingual-cased/aitk/bert-base-multilingual-cased_qdq_amd.json) | -| [meta-llama-Llama-3.2-1B-Instruct-hqq](meta-llama-Llama-3.2-1B-Instruct/olive/hqq.json) | [Qwen-Qwen2.5-Coder-1.5B-Instruct](Qwen-Qwen2.5-Coder-1.5B-Instruct/aitk/qwen2_5_trtrtx.json) | [google-bert-bert-base-multilingual-cased](google-bert-bert-base-multilingual-cased/aitk/bert-base-multilingual-cased_qdq_qnn.json) | -| [meta-llama-Llama-3.2-1B-Instruct-lmeval-onnx](meta-llama-Llama-3.2-1B-Instruct/olive/lmeval_onnx.json) | [Qwen-Qwen2.5-Coder-14B-Instruct](Qwen-Qwen2.5-Coder-14B-Instruct/aitk/qwen2_5_ov_config.json) | [google-gemma-3-1b-it](google-gemma-3-1b-it/OpenVINO/gemma_3_1b_it_context_ov_npu_config.json) | -| [meta-llama-Llama-3.2-1B-Instruct-lmeval](meta-llama-Llama-3.2-1B-Instruct/olive/lmeval.json) | [Qwen-Qwen2.5-Coder-14B-Instruct](Qwen-Qwen2.5-Coder-14B-Instruct/aitk/qwen2_5_trtrtx.json) | [google-vit-base-patch16-224](google-vit-base-patch16-224/OpenVINO/vit_base_patch16_224_context_ov_static.json) | -| [meta-llama-Llama-3.2-1B-Instruct-loha](meta-llama-Llama-3.2-1B-Instruct/olive/loha.json) | [Qwen-Qwen2.5-Coder-3B-Instruct](Qwen-Qwen2.5-Coder-3B-Instruct/aitk/qwen2_5_ov_config.json) | [google-vit-base-patch16-224](google-vit-base-patch16-224/QNN/vit_qnn_fp32_ctx.json) | -| [meta-llama-Llama-3.2-1B-Instruct-lokr](meta-llama-Llama-3.2-1B-Instruct/olive/lokr.json) | [Qwen-Qwen2.5-Coder-7B-Instruct](Qwen-Qwen2.5-Coder-7B-Instruct/aitk/qwen2_5_ov_config.json) | [google-vit-base-patch16-224](google-vit-base-patch16-224/aitk/vit-base-patch16-224_qdq_amd.json) | -| [meta-llama-Llama-3.2-1B-Instruct-mixed](meta-llama-Llama-3.2-1B-Instruct/olive/mixed.json) | [Qwen-Qwen2.5-Coder-7B-Instruct](Qwen-Qwen2.5-Coder-7B-Instruct/aitk/qwen2_5_trtrtx.json) | [google-vit-base-patch16-224](google-vit-base-patch16-224/aitk/vit-base-patch16-224_qdq_qnn.json) | -| [meta-llama-Llama-3.2-1B-Instruct-qlora](meta-llama-Llama-3.2-1B-Instruct/olive/qlora.json) | [Qwen2.5-0.5B-Instruct_Model_Builder_FP16](Qwen-Qwen2.5-0.5B-Instruct/NvTensorRtRtx/Qwen2.5-0.5B-Instruct_model_builder_fp16.json) | [google-vit-base-patch16-224](google-vit-base-patch16-224/aitk/vit_base_patch16_224_context_ov_static.json) | -| [meta-llama-Llama-3.2-1B-Instruct](meta-llama-Llama-3.2-1B-Instruct/aitk/llama3_2_ov_gpu_config.json) | [Qwen2.5-14B-Instruct_Model_Builder_INT4](Qwen-Qwen2.5-14B-Instruct/NvTensorRtRtx/Qwen2.5-14B-Instruct_model_builder_int4.json) | [intel-bert-base-uncased-mrpc (AMD)](intel-bert-base-uncased-mrpc/aitk/bert_qdq_amd.json) | -| [meta-llama-Meta-Llama-3-8B](meta-llama-Meta-Llama-3-8B/olive/config.json) | [Qwen2.5-7B-Instruct_Model_Builder_INT4](Qwen-Qwen2.5-7B-Instruct/NvTensorRtRtx/Qwen2.5-7B-Instruct_model_builder_int4.json) | [intel-bert-base-uncased-mrpc (ov)](intel-bert-base-uncased-mrpc/aitk/bert_ov.json) | -| [microsoft-Phi-3-mini-128k-instruct](microsoft-Phi-3-mini-128k-instruct/aitk/phi3_ov_config.json) | [Qwen2.5-Coder-0.5B-Instruct_Model_Builder_FP16](Qwen-Qwen2.5-Coder-0.5B-Instruct/NvTensorRtRtx/Qwen2.5-Coder-0.5B-Instruct_model_builder_fp16.json) | [intel-bert-base-uncased-mrpc](intel-bert-base-uncased-mrpc/aitk/bert_qdq_qnn.json) | -| [microsoft-Phi-3-mini-4k-instruct](microsoft-Phi-3-mini-4k-instruct/aitk/phi3_ov_config.json) | [Qwen2.5-Coder-1.5B-Instruct_Model_Builder_FP16](Qwen-Qwen2.5-Coder-1.5B-Instruct/NvTensorRtRtx/Qwen2.5-Coder-1.5B-Instruct_model_builder_fp16.json) | [laion-CLIP-ViT-B-32-laion2B-s34B-b79K](laion-CLIP-ViT-B-32-laion2B-s34B-b79K/aitk/laion_clip_ov.json) | -| [microsoft-Phi-3.5-mini-instruct](microsoft-Phi-3.5-mini-instruct/aitk/phi3_5_ov_gpu_config.json) | [Qwen2.5-Coder-14B-Instruct_Model_Builder_INT4](Qwen-Qwen2.5-Coder-14B-Instruct/NvTensorRtRtx/Qwen2.5-Coder-14B-Instruct_model_builder_int4.json) | [laion-CLIP-ViT-B-32-laion2B-s34B-b79K](laion-CLIP-ViT-B-32-laion2B-s34B-b79K/aitk/laion_clip_qdq_amd.json) | -| [microsoft-Phi-4-mini-instruct](microsoft-Phi-4-mini-instruct/aitk/phi4_ov_config.json) | [Qwen2.5-Coder-7B-Instruct_Model_Builder_INT4](Qwen-Qwen2.5-Coder-7B-Instruct/NvTensorRtRtx/Qwen2.5-Coder-7B-Instruct_model_builder_int4.json) | [laion-CLIP-ViT-B-32-laion2B-s34B-b79K](laion-CLIP-ViT-B-32-laion2B-s34B-b79K/aitk/laion_clip_qnn.json) | -| [microsoft-Phi-4-mini-reasoning](microsoft-Phi-4-mini-reasoning/aitk/phi4_ov_gpu_config.json) | [Qwen2.5_1.5B_Instruct_Model_Builder_FP16](Qwen-Qwen2.5-1.5B-Instruct/NvTensorRtRtx/Qwen2.5-1.5B-Instruct_model_builder_fp16.json) | [llama3.1-8b-instruct-x-elite](meta-llama-Llama-3.1-8B-Instruct/QNN/x_elite_config.json) | -| [microsoft-Phi-4](microsoft-Phi-4/aitk/phi4_ov_config.json) | [deepseek-ai-DeepSeek-R1-Distill-Llama-8B](deepseek-ai-DeepSeek-R1-Distill-Llama-8B/aitk/deepseek_ov_config.json) | [llama3.1-8b-instruct-x2-elite](meta-llama-Llama-3.1-8B-Instruct/QNN/x2_elite_config.json) | -| [microsoft-deberta-base-mnli](microsoft-deberta-base-mnli/aml/deberta.json) | [deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B](deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B/QNN/config_gpu.json) | [meta-llama-Llama-3.1-8B-Instruct](meta-llama-Llama-3.1-8B-Instruct/aitk/llama3_1_ov_config.json) | -| [microsoft-resnet-50](microsoft-resnet-50/aitk/resnet_context_ov_static.json) | [deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B](deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B/QNN/config_gpu_ctxbin.json) | [meta-llama-Llama-3.1-8B-Instruct](meta-llama-Llama-3.1-8B-Instruct/aitk/llama3_1_qnn_config.json) | -| [mistralai-Mistral-7B-Instruct-v0.2](mistralai-Mistral-7B-Instruct-v0.2/aitk/Mistral_7B_Instruct_v0.2_gpu_context_ov_dy.json) | [deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B](deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B/aitk/deepseek_dml_config.json) | [meta-llama-Llama-3.1-8B-Instruct](meta-llama-Llama-3.1-8B-Instruct/aitk/llama3_1_vitis_ai_config.json) | -| [mistralai-Mistral-7B-Instruct-v0.3](mistralai-Mistral-7B-Instruct-v0.3/aitk/mistral-7b-instruct-v0.3-ov.json) | [deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B](deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B/aitk/deepseek_ov_gpu_config.json) | [meta-llama-Llama-3.2-1B-Instruct](meta-llama-Llama-3.2-1B-Instruct/aitk/llama3_2_ov_config.json) | -| [moonshine-tiny](usefulSensors-moonshine-tiny/cli/optimize.sh) | [deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B](deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B/aitk/deepseek_qnn_gpu_config.json) | [meta-llama-Llama-3.2-1B-Instruct](meta-llama-Llama-3.2-1B-Instruct/aitk/llama3_2_qnn_config.json) | -| [openai-clip-vit-base-patch16](openai-clip-vit-base-patch16/aitk/openai_clip_ov.json) | [deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B](deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B/aitk/deepseek_trtrtx_config.json) | [meta-llama-Llama-3.2-1B-Instruct](meta-llama-Llama-3.2-1B-Instruct/aitk/llama3_2_vitis_ai_config.json) | -| [openai-clip-vit-base-patch32](openai-clip-vit-base-patch32/aitk/openai_clip_ov.json) | [deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B](deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B/olive/mixed.json) | [microsoft-Phi-3-mini-128k-instruct](microsoft-Phi-3-mini-128k-instruct/QNN/config.json) | -| [openai-clip-vit-large-patch14](openai-clip-vit-large-patch14/aitk/openai_clip_ov.json) | [deepseek-ai-DeepSeek-R1-Distill-Qwen-14B](deepseek-ai-DeepSeek-R1-Distill-Qwen-14B/aitk/deepseek_ov_config.json) | [microsoft-Phi-3-mini-128k-instruct](microsoft-Phi-3-mini-128k-instruct/aitk/phi3_ov_npu_config.json) | -| [openai-whisper-base-cpu-int8](openai-whisper-base/cpu/whisper-base_cpu_int8.json) | [deepseek-ai-DeepSeek-R1-Distill-Qwen-14B](deepseek-ai-DeepSeek-R1-Distill-Qwen-14B/aitk/deepseek_trtrtx.json) | [microsoft-Phi-3-mini-128k-instruct](microsoft-Phi-3-mini-128k-instruct/aitk/phi3_qnn.json) | -| [openai-whisper-base.en-cpu-int8](openai-whisper-base.en/cpu/whisper-base.en_cpu_int8.json) | [deepseek-ai-DeepSeek-R1-Distill-Qwen-7B](deepseek-ai-DeepSeek-R1-Distill-Qwen-7B/aitk/deepseek_ov_config.json) | [microsoft-Phi-3-mini-128k-instruct](microsoft-Phi-3-mini-128k-instruct/aitk/phi3_vitis_ai_config.json) | -| [openai-whisper-large-cpu-int8](openai-whisper-large/cpu/whisper-large_cpu_int8.json) | [deepseek-ai-DeepSeek-R1-Distill-Qwen-7B](deepseek-ai-DeepSeek-R1-Distill-Qwen-7B/aitk/deepseek_trtrtx.json) | [microsoft-Phi-3-mini-4k-instruct](microsoft-Phi-3-mini-4k-instruct/QNN/config.json) | -| [openai-whisper-large-v2-cpu-int8](openai-whisper-large-v2/cpu/whisper-large-v2_cpu_int8.json) | [facebook-opt-125m-splicegpt](facebook-opt-125m/olive/slicegpt.json) | [microsoft-Phi-3-mini-4k-instruct](microsoft-Phi-3-mini-4k-instruct/aitk/phi3_ov_npu_config.json) | -| [openai-whisper-large-v3-cpu-int8](openai-whisper-large-v3/cpu/whisper-large-v3_cpu_int8.json) | [google-bert-bert-base-multilingual-cased](google-bert-bert-base-multilingual-cased/QNN/config_gpu_fp32.json) | [microsoft-Phi-3-mini-4k-instruct](microsoft-Phi-3-mini-4k-instruct/aitk/phi3_qnn.json) | -| [openai-whisper-large-v3-turbo-cpu-int8](openai-whisper-large-v3-turbo/cpu/whisper-large-v3-turbo_cpu_int8.json) | [google-bert-bert-base-multilingual-cased](google-bert-bert-base-multilingual-cased/aitk/bert-base-multilingual-cased_context_ov_static.json) | [microsoft-Phi-3-mini-4k-instruct](microsoft-Phi-3-mini-4k-instruct/aitk/phi3_vitis_ai_config.json) | -| [openai-whisper-large-v3-turbo](openai-whisper-large-v3-turbo/olive/whisper_large_v3_turbo_decoder_fp32.json) | [google-bert-bert-base-multilingual-cased](google-bert-bert-base-multilingual-cased/aitk/bert-base-multilingual-cased_dml.json) | [microsoft-Phi-3.5-mini-instruct](microsoft-Phi-3.5-mini-instruct/QNN/config.json) | -| [openai-whisper-large-v3-turbo](openai-whisper-large-v3-turbo/olive/whisper_large_v3_turbo_decoder_qdq.json) | [google-bert-bert-base-multilingual-cased](google-bert-bert-base-multilingual-cased/aitk/bert-base-multilingual-cased_migraphx.json) | [microsoft-Phi-3.5-mini-instruct](microsoft-Phi-3.5-mini-instruct/QNN/config_fp16.json) | -| [openai-whisper-large-v3-turbo](openai-whisper-large-v3-turbo/olive/whisper_large_v3_turbo_encoder_fp32.json) | [google-bert-bert-base-multilingual-cased](google-bert-bert-base-multilingual-cased/aitk/bert-base-multilingual-cased_qnn_gpu.json) | [microsoft-Phi-3.5-mini-instruct](microsoft-Phi-3.5-mini-instruct/aitk/phi3_5_ov_config.json) | -| [openai-whisper-large-v3-turbo](openai-whisper-large-v3-turbo/olive/whisper_large_v3_turbo_encoder_qdq.json) | [google-bert-bert-base-multilingual-cased](google-bert-bert-base-multilingual-cased/aitk/bert-base-multilingual-cased_trtrtx.json) | [microsoft-Phi-3.5-mini-instruct](microsoft-Phi-3.5-mini-instruct/aitk/phi3_5_qnn_config.json) | -| [openai-whisper-medium-cpu-int8](openai-whisper-medium/cpu/whisper-medium_cpu_int8.json) | [google-gemma-3-1b-it](google-gemma-3-1b-it/OpenVINO/gemma_3_1b_it_context_ov_gpu_config.json) | [microsoft-Phi-3.5-mini-instruct](microsoft-Phi-3.5-mini-instruct/aitk/phi3_5_vitis_ai_config.json) | -| [openai-whisper-medium.en-cpu-int8](openai-whisper-medium.en/cpu/whisper-medium.en_cpu_int8.json) | [google-gemma](google-gemma/olive/README.md) | [microsoft-Phi-4-mini-instruct](microsoft-Phi-4-mini-instruct/QNN/phi4_mini_qnn_docker.json) | -| [openai-whisper-small-cpu-int8](openai-whisper-small/cpu/whisper-small_cpu_int8.json) | [google-vit-base-patch16-224](google-vit-base-patch16-224/QNN/config_gpu_fp32.json) | [microsoft-Phi-4-mini-instruct](microsoft-Phi-4-mini-instruct/aitk/phi4_ov_npu_config.json) | -| [openai-whisper-small.en-cpu-int8](openai-whisper-small.en/cpu/whisper-small.en_cpu_int8.json) | [google-vit-base-patch16-224](google-vit-base-patch16-224/aitk/vit-base-patch16-224_dml.json) | [microsoft-Phi-4-mini-instruct](microsoft-Phi-4-mini-instruct/aitk/phi4_qnn.json) | -| [openai-whisper-tiny-cpu-int8](openai-whisper-tiny/cpu/whisper-tiny_cpu_int8.json) | [google-vit-base-patch16-224](google-vit-base-patch16-224/aitk/vit-base-patch16-224_migraphx.json) | [microsoft-Phi-4-mini-instruct](microsoft-Phi-4-mini-instruct/aitk/phi4_vitis_ai_config.json) | -| [openai-whisper-tiny.en-cpu-int8](openai-whisper-tiny.en/cpu/whisper-tiny.en_cpu_int8.json) | [google-vit-base-patch16-224](google-vit-base-patch16-224/aitk/vit-base-patch16-224_qnn_gpu.json) | [microsoft-Phi-4-mini-reasoning](microsoft-Phi-4-mini-reasoning/aitk/phi4_ov_config.json) | -| [qwen2.5-vl-3B-Instruct](Qwen-Qwen2.5-VL-3B-Instruct/builtin/optimize.py) | [google-vit-base-patch16-224](google-vit-base-patch16-224/aitk/vit-base-patch16-224_trtrtx.json) | [microsoft-Phi-4-mini-reasoning](microsoft-Phi-4-mini-reasoning/aitk/phi4_vitis_ai_config.json) | -| [qwen3.5-0.8B-Instruct](Qwen-Qwen3.5-0.8B/builtin/optimize.py) | [google-vit-base-patch16-224](google-vit-base-patch16-224/aitk/vit_base_patch16_224_context_ov_static.json) | [microsoft-Phi-4-reasoning-plus](microsoft-Phi-4-reasoning-plus/aitk/phi4_ov_config.json) | -| [qwen3.5-2B](Qwen-Qwen3.5-2B/builtin/optimize.py) | [gpt-oss-20b](gpt-oss-20b/int4_cuda_int4_qmoe/gpt-oss-20b.sh) | [microsoft-Phi-4-reasoning](microsoft-Phi-4-reasoning/QAIRT/htp_sc8380xp.json) | -| [qwen3.5-4B](Qwen-Qwen3.5-4B/builtin/optimize.py) | [intel-bert-base-uncased-mrpc (ov)](intel-bert-base-uncased-mrpc/aitk/bert_ov.json) | [microsoft-Phi-4-reasoning](microsoft-Phi-4-reasoning/QAIRT/htp_sc8480xp.json) | -| [qwen3.5-9B](Qwen-Qwen3.5-9B/builtin/optimize.py) | [intel-bert-base-uncased-mrpc-inc-quant](intel-bert-base-uncased-mrpc/oci/gpu/cuda/inc_quant.json) | [microsoft-Phi-4-reasoning](microsoft-Phi-4-reasoning/QNN/config.json) | -| [qwen3vl-2B-Instruct](Qwen-Qwen3-VL-2B-Instruct/builtin/optimize.py) | [intel-bert-base-uncased-mrpc-ptq](intel-bert-base-uncased-mrpc/oci/gpu/cuda/ptq.json) | [microsoft-Phi-4-reasoning](microsoft-Phi-4-reasoning/aitk/phi4_ov_config.json) | -| [qwen3vl-4B-Instruct](Qwen-Qwen3-VL-4B-Instruct/builtin/optimize.py) | [intel-bert-base-uncased-mrpc](intel-bert-base-uncased-mrpc/QNN/config_gpu_fp32.json) | [microsoft-resnet-50](microsoft-resnet-50/aitk/resnet_context_ov_static.json) | -| [qwen3vl-8B-Instruct](Qwen-Qwen3-VL-8B-Instruct/builtin/optimize.py) | [intel-bert-base-uncased-mrpc](intel-bert-base-uncased-mrpc/aitk/bert_dml.json) | [microsoft-resnet-50](microsoft-resnet-50/aitk/resnet_qdq_amd.json) | -| [sd-legacy-stable-diffusion-v1-5](sd-legacy-stable-diffusion-v1-5/aitk/sd_ov_workflow.json) | [intel-bert-base-uncased-mrpc](intel-bert-base-uncased-mrpc/aitk/bert_migraphx.json) | [microsoft-resnet-50](microsoft-resnet-50/aitk/resnet_qdq_qnn.json) | -| [sd2-community-stable-diffusion-2-1](sd2-community-stable-diffusion-2-1/aitk/sd_ov_workflow.json) | [intel-bert-base-uncased-mrpc](intel-bert-base-uncased-mrpc/aitk/bert_qnn_gpu.json) | [microsoft-table-transformer-detection](microsoft-table-transformer-detection/QNN/ttd_config.json) | -| [sshleifer-tiny-gpt2-sparsegpt](sshleifer-tiny-gpt2/olive/sparsegpt.json) | [intel-bert-base-uncased-mrpc](intel-bert-base-uncased-mrpc/aitk/bert_trtrtx.json) | [mistralai-Mistral-7B-Instruct-v0.2](mistralai-Mistral-7B-Instruct-v0.2/aitk/Mistral_7B_Instruct_v0.2_npu_context_ov_dy.json) | -| [stable-diffusion-v1-4-safety-checker](compvis-stable-diffusion-v1-4/olive/config_safety_checker.json) | [laion-CLIP-ViT-B-32-laion2B-s34B-b79K](laion-CLIP-ViT-B-32-laion2B-s34B-b79K/QNN/config_gpu_fp32.json) | [mistralai-Mistral-7B-Instruct-v0.2](mistralai-Mistral-7B-Instruct-v0.2/aitk/Mistral_7B_Instruct_v0.2_vitis_ai_config.json) | -| [stable-diffusion-v1-4-text-encoder](compvis-stable-diffusion-v1-4/olive/config_text_encoder.json) | [laion-CLIP-ViT-B-32-laion2B-s34B-b79K](laion-CLIP-ViT-B-32-laion2B-s34B-b79K/aitk/laion_clip_dml.json) | [openai-clip-vit-base-patch16](openai-clip-vit-base-patch16/aitk/openai_clip_ov.json) | -| [stable-diffusion-v1-4-unet](compvis-stable-diffusion-v1-4/olive/config_unet.json) | [laion-CLIP-ViT-B-32-laion2B-s34B-b79K](laion-CLIP-ViT-B-32-laion2B-s34B-b79K/aitk/laion_clip_migraphx.json) | [openai-clip-vit-base-patch16](openai-clip-vit-base-patch16/aitk/openai_clip_qdq_amd.json) | -| [stable-diffusion-v1-4-vae-decoder](compvis-stable-diffusion-v1-4/olive/config_vae_decoder.json) | [laion-CLIP-ViT-B-32-laion2B-s34B-b79K](laion-CLIP-ViT-B-32-laion2B-s34B-b79K/aitk/laion_clip_ov.json) | [openai-clip-vit-base-patch16](openai-clip-vit-base-patch16/aitk/openai_clip_qnn.json) | -| [stable-diffusion-v1-4-vae-encoder](compvis-stable-diffusion-v1-4/olive/config_vae_encoder.json) | [laion-CLIP-ViT-B-32-laion2B-s34B-b79K](laion-CLIP-ViT-B-32-laion2B-s34B-b79K/aitk/laion_clip_qnn_gpu.json) | [openai-clip-vit-base-patch32](openai-clip-vit-base-patch32/aitk/openai_clip_ov.json) | -| [stable-diffusion-v1-5](sd-legacy-stable-diffusion-v1-5/olive/optimize.sh) | [laion-CLIP-ViT-B-32-laion2B-s34B-b79K](laion-CLIP-ViT-B-32-laion2B-s34B-b79K/aitk/laion_clip_trtrtx.json) | [openai-clip-vit-base-patch32](openai-clip-vit-base-patch32/aitk/openai_clip_qdq_amd.json) | -| [stable-diffusion-xl-base-1.0](stabilityai-stable-diffusion-xl-base-1.0/olive/stable_diffusion_xl.py) | [meta-llama-Llama-3.1-8B-Instruct](meta-llama-Llama-3.1-8B-Instruct/aitk/llama3_1_dml_config.json) | [openai-clip-vit-base-patch32](openai-clip-vit-base-patch32/aitk/openai_clip_qnn.json) | -| [timm-mobilenetv3_small_100.lamb_in1k](timm-mobilenetv3_small_100.lamb_in1k/olive/config.json) | [meta-llama-Llama-3.1-8B-Instruct](meta-llama-Llama-3.1-8B-Instruct/aitk/llama3_1_ov_gpu_config.json) | [openai-clip-vit-large-patch14](openai-clip-vit-large-patch14/aitk/openai_clip_ov.json) | -| | [meta-llama-Llama-3.1-8B-Instruct](meta-llama-Llama-3.1-8B-Instruct/aitk/llama3_1_trtrtx_config.json) | [openai-clip-vit-large-patch14](openai-clip-vit-large-patch14/aitk/openai_clip_qdq_amd.json) | -| | [meta-llama-Llama-3.2-1B-Instruct-dora](meta-llama-Llama-3.2-1B-Instruct/olive/dora.json) | [openai-clip-vit-large-patch14](openai-clip-vit-large-patch14/aitk/openai_clip_qnn.json) | -| | [meta-llama-Llama-3.2-1B-Instruct-hqq](meta-llama-Llama-3.2-1B-Instruct/olive/hqq.json) | [openai-whisper-large-v3-turbo-vitisai](openai-whisper-large-v3-turbo/VitisAI/run_whisper.py) | -| | [meta-llama-Llama-3.2-1B-Instruct-lmeval-onnx](meta-llama-Llama-3.2-1B-Instruct/olive/lmeval_onnx.json) | [openai-whisper-large-v3-turbo](openai-whisper-large-v3-turbo/OpenVINO/whisper_large_v3_turbo_default_ov_npu.json) | -| | [meta-llama-Llama-3.2-1B-Instruct-lmeval](meta-llama-Llama-3.2-1B-Instruct/olive/lmeval.json) | [openai-whisper-large-v3-turbo](openai-whisper-large-v3-turbo/aitk/qnn_workflow.json) | -| | [meta-llama-Llama-3.2-1B-Instruct-loha](meta-llama-Llama-3.2-1B-Instruct/olive/loha.json) | [openai-whisper-large-v3-turbo](openai-whisper-large-v3-turbo/olive/whisper_large_v3_turbo_decoder_fp32.json) | -| | [meta-llama-Llama-3.2-1B-Instruct-lokr](meta-llama-Llama-3.2-1B-Instruct/olive/lokr.json) | [openai-whisper-large-v3-turbo](openai-whisper-large-v3-turbo/olive/whisper_large_v3_turbo_decoder_qdq.json) | -| | [meta-llama-Llama-3.2-1B-Instruct-mixed](meta-llama-Llama-3.2-1B-Instruct/olive/mixed.json) | [openai-whisper-large-v3-turbo](openai-whisper-large-v3-turbo/olive/whisper_large_v3_turbo_decoder_qdq_ctx.json) | -| | [meta-llama-Llama-3.2-1B-Instruct-qlora](meta-llama-Llama-3.2-1B-Instruct/olive/qlora.json) | [openai-whisper-large-v3-turbo](openai-whisper-large-v3-turbo/olive/whisper_large_v3_turbo_encoder_fp32.json) | -| | [meta-llama-Llama-3.2-1B-Instruct](meta-llama-Llama-3.2-1B-Instruct/QNN/config_gpu.json) | [openai-whisper-large-v3-turbo](openai-whisper-large-v3-turbo/olive/whisper_large_v3_turbo_encoder_qdq.json) | -| | [meta-llama-Llama-3.2-1B-Instruct](meta-llama-Llama-3.2-1B-Instruct/QNN/config_gpu_ctxbin.json) | [openai-whisper-large-v3-turbo](openai-whisper-large-v3-turbo/olive/whisper_large_v3_turbo_encoder_qdq_ctx.json) | -| | [meta-llama-Llama-3.2-1B-Instruct](meta-llama-Llama-3.2-1B-Instruct/aitk/llama3_2_dml_config.json) | [openai-whisper-medium-vitisai](openai-whisper-medium/VitisAI/run_whisper.py) | -| | [meta-llama-Llama-3.2-1B-Instruct](meta-llama-Llama-3.2-1B-Instruct/aitk/llama3_2_ov_gpu_config.json) | [openai-whisper-small-vitisai](openai-whisper-small/VitisAI/run_whisper.py) | -| | [meta-llama-Llama-3.2-1B-Instruct](meta-llama-Llama-3.2-1B-Instruct/aitk/llama3_2_qnn_gpu_config.json) | [qwen2.5-7b-instruct](Qwen-Qwen2.5-7B-Instruct/QNN/config.json) | -| | [meta-llama-Llama-3.2-1B-Instruct](meta-llama-Llama-3.2-1B-Instruct/aitk/llama3_2_trtrtx_config.json) | [sam-vit-base](sam-vit-base/QNN/sam_mask_box_decoder_qnn_fp16.json) | -| | [microsoft-Phi-3-mini-128k-instruct](microsoft-Phi-3-mini-128k-instruct/aitk/phi3_ov_config.json) | [sam-vit-base](sam-vit-base/QNN/sam_mask_decoder_qnn_fp16_ctx.json) | -| | [microsoft-Phi-3-mini-128k-instruct](microsoft-Phi-3-mini-128k-instruct/aitk/phi3_trtrtx.json) | [sam-vit-base](sam-vit-base/QNN/sam_mask_point_decoder_qnn_fp16.json) | -| | [microsoft-Phi-3-mini-4k-instruct](microsoft-Phi-3-mini-4k-instruct/aitk/phi3_ov_config.json) | [sam-vit-base](sam-vit-base/QNN/sam_vision_encoder_qnn_w8a8_ctx.json) | -| | [microsoft-Phi-3-mini-4k-instruct](microsoft-Phi-3-mini-4k-instruct/aitk/phi3_trtrtx.json) | [sam-vit-base](sam-vit-base/aitk/sam_qnn_workflow.json) | +| [gemma4-e2b-fp32-cpu](google-gemma-4-E2B-it/cpu/fp32/config.json) | [Qwen-Qwen2.5-1.5B-Instruct](Qwen-Qwen2.5-1.5B-Instruct/aitk/qwen2_5_qnn_gpu_config.json) | [Qwen-Qwen2.5-Coder-7B-Instruct](Qwen-Qwen2.5-Coder-7B-Instruct/aitk/qwen2_5_vitis_ai_config.json) | +| [gemma4-e2b-int4-kquant-cpu](google-gemma-4-E2B-it/cpu/int4/config.json) | [Qwen-Qwen2.5-1.5B-Instruct](Qwen-Qwen2.5-1.5B-Instruct/aitk/qwen2_5_trtrtx_config.json) | [deepseek-ai-DeepSeek-R1-Distill-Llama-8B](deepseek-ai-DeepSeek-R1-Distill-Llama-8B/aitk/deepseek_ov_npu_config.json) | +| [google-bert-bert-base-multilingual-cased](google-bert-bert-base-multilingual-cased/aitk/bert-base-multilingual-cased_context_ov_static.json) | [Qwen-Qwen2.5-14B-Instruct](Qwen-Qwen2.5-14B-Instruct/aitk/qwen2_5_ov_config.json) | [deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B](deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B/aitk/deepseek_ov_config.json) | +| [google-gemma](google-gemma/olive/README.md) | [Qwen-Qwen2.5-14B-Instruct](Qwen-Qwen2.5-14B-Instruct/aitk/qwen2_5_trtrtx.json) | [deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B](deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B/aitk/deepseek_qnn_config.json) | +| [google-vit-base-patch16-224](google-vit-base-patch16-224/aitk/vit_base_patch16_224_context_ov_static.json) | [Qwen-Qwen2.5-3B-Instruct](Qwen-Qwen2.5-3B-Instruct/aitk/qwen2_5_ov_config.json) | [deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B](deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B/aitk/deepseek_vitis_ai_config.json) | +| [gpt-oss-20b](gpt-oss-20b/int4_cuda_int4_qmoe/gpt-oss-20b.sh) | [Qwen-Qwen2.5-7B-Instruct](Qwen-Qwen2.5-7B-Instruct/aitk/qwen2_5_ov_config.json) | [deepseek-ai-DeepSeek-R1-Distill-Qwen-14B](deepseek-ai-DeepSeek-R1-Distill-Qwen-14B/aitk/deepseek_ov_npu_config.json) | +| [intel-bert-base-uncased-mrpc (ov)](intel-bert-base-uncased-mrpc/aitk/bert_ov.json) | [Qwen-Qwen2.5-7B-Instruct](Qwen-Qwen2.5-7B-Instruct/aitk/qwen2_5_trtrtx.json) | [deepseek-ai-DeepSeek-R1-Distill-Qwen-7B](deepseek-ai-DeepSeek-R1-Distill-Qwen-7B/aitk/deepseek_ov_npu_config.json) | +| [intel-bert-base-uncased-mrpc-inc-smooth-quant](intel-bert-base-uncased-mrpc/oci/cpu/inc_smooth_quant.json) | [Qwen-Qwen2.5-Coder-0.5B-Instruct](Qwen-Qwen2.5-Coder-0.5B-Instruct/aitk/qwen2_5_ov_config.json) | [deepseek-ai-DeepSeek-R1-Distill-Qwen-7B](deepseek-ai-DeepSeek-R1-Distill-Qwen-7B/aitk/deepseek_vitis_ai_config.json) | +| [intel-bert-base-uncased-mrpc-ptq](intel-bert-base-uncased-mrpc/oci/cpu/ptq.json) | [Qwen-Qwen2.5-Coder-0.5B-Instruct](Qwen-Qwen2.5-Coder-0.5B-Instruct/aitk/qwen2_5_trtrtx.json) | [google-bert-bert-base-multilingual-cased](google-bert-bert-base-multilingual-cased/aitk/bert-base-multilingual-cased_context_ov_static.json) | +| [laion-CLIP-ViT-B-32-laion2B-s34B-b79K](laion-CLIP-ViT-B-32-laion2B-s34B-b79K/aitk/laion_clip_ov.json) | [Qwen-Qwen2.5-Coder-1.5B-Instruct](Qwen-Qwen2.5-Coder-1.5B-Instruct/aitk/qwen2_5_ov_config.json) | [google-bert-bert-base-multilingual-cased](google-bert-bert-base-multilingual-cased/aitk/bert-base-multilingual-cased_qdq_amd.json) | +| [meta-llama-Llama-3.1-8B-Instruct](meta-llama-Llama-3.1-8B-Instruct/aitk/llama3_1_ov_gpu_config.json) | [Qwen-Qwen2.5-Coder-1.5B-Instruct](Qwen-Qwen2.5-Coder-1.5B-Instruct/aitk/qwen2_5_trtrtx.json) | [google-bert-bert-base-multilingual-cased](google-bert-bert-base-multilingual-cased/aitk/bert-base-multilingual-cased_qdq_qnn.json) | +| [meta-llama-Llama-3.2-1B-Instruct-dora](meta-llama-Llama-3.2-1B-Instruct/olive/dora.json) | [Qwen-Qwen2.5-Coder-14B-Instruct](Qwen-Qwen2.5-Coder-14B-Instruct/aitk/qwen2_5_ov_config.json) | [google-gemma-3-1b-it](google-gemma-3-1b-it/OpenVINO/gemma_3_1b_it_context_ov_npu_config.json) | +| [meta-llama-Llama-3.2-1B-Instruct-hqq](meta-llama-Llama-3.2-1B-Instruct/olive/hqq.json) | [Qwen-Qwen2.5-Coder-14B-Instruct](Qwen-Qwen2.5-Coder-14B-Instruct/aitk/qwen2_5_trtrtx.json) | [google-vit-base-patch16-224](google-vit-base-patch16-224/OpenVINO/vit_base_patch16_224_context_ov_static.json) | +| [meta-llama-Llama-3.2-1B-Instruct-lmeval-onnx](meta-llama-Llama-3.2-1B-Instruct/olive/lmeval_onnx.json) | [Qwen-Qwen2.5-Coder-3B-Instruct](Qwen-Qwen2.5-Coder-3B-Instruct/aitk/qwen2_5_ov_config.json) | [google-vit-base-patch16-224](google-vit-base-patch16-224/QNN/vit_qnn_fp32_ctx.json) | +| [meta-llama-Llama-3.2-1B-Instruct-lmeval](meta-llama-Llama-3.2-1B-Instruct/olive/lmeval.json) | [Qwen-Qwen2.5-Coder-7B-Instruct](Qwen-Qwen2.5-Coder-7B-Instruct/aitk/qwen2_5_ov_config.json) | [google-vit-base-patch16-224](google-vit-base-patch16-224/aitk/vit-base-patch16-224_qdq_amd.json) | +| [meta-llama-Llama-3.2-1B-Instruct-loha](meta-llama-Llama-3.2-1B-Instruct/olive/loha.json) | [Qwen-Qwen2.5-Coder-7B-Instruct](Qwen-Qwen2.5-Coder-7B-Instruct/aitk/qwen2_5_trtrtx.json) | [google-vit-base-patch16-224](google-vit-base-patch16-224/aitk/vit-base-patch16-224_qdq_qnn.json) | +| [meta-llama-Llama-3.2-1B-Instruct-lokr](meta-llama-Llama-3.2-1B-Instruct/olive/lokr.json) | [Qwen2.5-0.5B-Instruct_Model_Builder_FP16](Qwen-Qwen2.5-0.5B-Instruct/NvTensorRtRtx/Qwen2.5-0.5B-Instruct_model_builder_fp16.json) | [google-vit-base-patch16-224](google-vit-base-patch16-224/aitk/vit_base_patch16_224_context_ov_static.json) | +| [meta-llama-Llama-3.2-1B-Instruct-mixed](meta-llama-Llama-3.2-1B-Instruct/olive/mixed.json) | [Qwen2.5-14B-Instruct_Model_Builder_INT4](Qwen-Qwen2.5-14B-Instruct/NvTensorRtRtx/Qwen2.5-14B-Instruct_model_builder_int4.json) | [intel-bert-base-uncased-mrpc (AMD)](intel-bert-base-uncased-mrpc/aitk/bert_qdq_amd.json) | +| [meta-llama-Llama-3.2-1B-Instruct-qlora](meta-llama-Llama-3.2-1B-Instruct/olive/qlora.json) | [Qwen2.5-7B-Instruct_Model_Builder_INT4](Qwen-Qwen2.5-7B-Instruct/NvTensorRtRtx/Qwen2.5-7B-Instruct_model_builder_int4.json) | [intel-bert-base-uncased-mrpc (ov)](intel-bert-base-uncased-mrpc/aitk/bert_ov.json) | +| [meta-llama-Llama-3.2-1B-Instruct](meta-llama-Llama-3.2-1B-Instruct/aitk/llama3_2_ov_gpu_config.json) | [Qwen2.5-Coder-0.5B-Instruct_Model_Builder_FP16](Qwen-Qwen2.5-Coder-0.5B-Instruct/NvTensorRtRtx/Qwen2.5-Coder-0.5B-Instruct_model_builder_fp16.json) | [intel-bert-base-uncased-mrpc](intel-bert-base-uncased-mrpc/aitk/bert_qdq_qnn.json) | +| [meta-llama-Meta-Llama-3-8B](meta-llama-Meta-Llama-3-8B/olive/config.json) | [Qwen2.5-Coder-1.5B-Instruct_Model_Builder_FP16](Qwen-Qwen2.5-Coder-1.5B-Instruct/NvTensorRtRtx/Qwen2.5-Coder-1.5B-Instruct_model_builder_fp16.json) | [laion-CLIP-ViT-B-32-laion2B-s34B-b79K](laion-CLIP-ViT-B-32-laion2B-s34B-b79K/aitk/laion_clip_ov.json) | +| [microsoft-Phi-3-mini-128k-instruct](microsoft-Phi-3-mini-128k-instruct/aitk/phi3_ov_config.json) | [Qwen2.5-Coder-14B-Instruct_Model_Builder_INT4](Qwen-Qwen2.5-Coder-14B-Instruct/NvTensorRtRtx/Qwen2.5-Coder-14B-Instruct_model_builder_int4.json) | [laion-CLIP-ViT-B-32-laion2B-s34B-b79K](laion-CLIP-ViT-B-32-laion2B-s34B-b79K/aitk/laion_clip_qdq_amd.json) | +| [microsoft-Phi-3-mini-4k-instruct](microsoft-Phi-3-mini-4k-instruct/aitk/phi3_ov_config.json) | [Qwen2.5-Coder-7B-Instruct_Model_Builder_INT4](Qwen-Qwen2.5-Coder-7B-Instruct/NvTensorRtRtx/Qwen2.5-Coder-7B-Instruct_model_builder_int4.json) | [laion-CLIP-ViT-B-32-laion2B-s34B-b79K](laion-CLIP-ViT-B-32-laion2B-s34B-b79K/aitk/laion_clip_qnn.json) | +| [microsoft-Phi-3.5-mini-instruct](microsoft-Phi-3.5-mini-instruct/aitk/phi3_5_ov_gpu_config.json) | [Qwen2.5_1.5B_Instruct_Model_Builder_FP16](Qwen-Qwen2.5-1.5B-Instruct/NvTensorRtRtx/Qwen2.5-1.5B-Instruct_model_builder_fp16.json) | [llama3.1-8b-instruct-x-elite](meta-llama-Llama-3.1-8B-Instruct/QNN/x_elite_config.json) | +| [microsoft-Phi-4-mini-instruct](microsoft-Phi-4-mini-instruct/aitk/phi4_ov_config.json) | [deepseek-ai-DeepSeek-R1-Distill-Llama-8B](deepseek-ai-DeepSeek-R1-Distill-Llama-8B/aitk/deepseek_ov_config.json) | [llama3.1-8b-instruct-x2-elite](meta-llama-Llama-3.1-8B-Instruct/QNN/x2_elite_config.json) | +| [microsoft-Phi-4-mini-reasoning](microsoft-Phi-4-mini-reasoning/aitk/phi4_ov_gpu_config.json) | [deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B](deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B/QNN/config_gpu.json) | [meta-llama-Llama-3.1-8B-Instruct](meta-llama-Llama-3.1-8B-Instruct/aitk/llama3_1_ov_config.json) | +| [microsoft-Phi-4](microsoft-Phi-4/aitk/phi4_ov_config.json) | [deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B](deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B/QNN/config_gpu_ctxbin.json) | [meta-llama-Llama-3.1-8B-Instruct](meta-llama-Llama-3.1-8B-Instruct/aitk/llama3_1_qnn_config.json) | +| [microsoft-deberta-base-mnli](microsoft-deberta-base-mnli/aml/deberta.json) | [deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B](deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B/aitk/deepseek_dml_config.json) | [meta-llama-Llama-3.1-8B-Instruct](meta-llama-Llama-3.1-8B-Instruct/aitk/llama3_1_vitis_ai_config.json) | +| [microsoft-resnet-50](microsoft-resnet-50/aitk/resnet_context_ov_static.json) | [deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B](deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B/aitk/deepseek_ov_gpu_config.json) | [meta-llama-Llama-3.2-1B-Instruct](meta-llama-Llama-3.2-1B-Instruct/aitk/llama3_2_ov_config.json) | +| [ministral_3_3b](mistralai-Ministral-3-3B-Instruct-2512/builtin/optimize.py) | [deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B](deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B/aitk/deepseek_qnn_gpu_config.json) | [meta-llama-Llama-3.2-1B-Instruct](meta-llama-Llama-3.2-1B-Instruct/aitk/llama3_2_qnn_config.json) | +| [mistralai-Mistral-7B-Instruct-v0.2](mistralai-Mistral-7B-Instruct-v0.2/aitk/Mistral_7B_Instruct_v0.2_gpu_context_ov_dy.json) | [deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B](deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B/aitk/deepseek_trtrtx_config.json) | [meta-llama-Llama-3.2-1B-Instruct](meta-llama-Llama-3.2-1B-Instruct/aitk/llama3_2_vitis_ai_config.json) | +| [mistralai-Mistral-7B-Instruct-v0.3](mistralai-Mistral-7B-Instruct-v0.3/aitk/mistral-7b-instruct-v0.3-ov.json) | [deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B](deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B/olive/mixed.json) | [microsoft-Phi-3-mini-128k-instruct](microsoft-Phi-3-mini-128k-instruct/QNN/config.json) | +| [moonshine-tiny](usefulSensors-moonshine-tiny/cli/optimize.sh) | [deepseek-ai-DeepSeek-R1-Distill-Qwen-14B](deepseek-ai-DeepSeek-R1-Distill-Qwen-14B/aitk/deepseek_ov_config.json) | [microsoft-Phi-3-mini-128k-instruct](microsoft-Phi-3-mini-128k-instruct/aitk/phi3_ov_npu_config.json) | +| [openai-clip-vit-base-patch16](openai-clip-vit-base-patch16/aitk/openai_clip_ov.json) | [deepseek-ai-DeepSeek-R1-Distill-Qwen-14B](deepseek-ai-DeepSeek-R1-Distill-Qwen-14B/aitk/deepseek_trtrtx.json) | [microsoft-Phi-3-mini-128k-instruct](microsoft-Phi-3-mini-128k-instruct/aitk/phi3_qnn.json) | +| [openai-clip-vit-base-patch32](openai-clip-vit-base-patch32/aitk/openai_clip_ov.json) | [deepseek-ai-DeepSeek-R1-Distill-Qwen-7B](deepseek-ai-DeepSeek-R1-Distill-Qwen-7B/aitk/deepseek_ov_config.json) | [microsoft-Phi-3-mini-128k-instruct](microsoft-Phi-3-mini-128k-instruct/aitk/phi3_vitis_ai_config.json) | +| [openai-clip-vit-large-patch14](openai-clip-vit-large-patch14/aitk/openai_clip_ov.json) | [deepseek-ai-DeepSeek-R1-Distill-Qwen-7B](deepseek-ai-DeepSeek-R1-Distill-Qwen-7B/aitk/deepseek_trtrtx.json) | [microsoft-Phi-3-mini-4k-instruct](microsoft-Phi-3-mini-4k-instruct/QNN/config.json) | +| [openai-whisper-base-cpu-int8](openai-whisper-base/cpu/whisper-base_cpu_int8.json) | [facebook-opt-125m-splicegpt](facebook-opt-125m/olive/slicegpt.json) | [microsoft-Phi-3-mini-4k-instruct](microsoft-Phi-3-mini-4k-instruct/aitk/phi3_ov_npu_config.json) | +| [openai-whisper-base.en-cpu-int8](openai-whisper-base.en/cpu/whisper-base.en_cpu_int8.json) | [gemma4-e2b-fp16-cuda](google-gemma-4-E2B-it/cuda/fp16/config.json) | [microsoft-Phi-3-mini-4k-instruct](microsoft-Phi-3-mini-4k-instruct/aitk/phi3_qnn.json) | +| [openai-whisper-large-cpu-int8](openai-whisper-large/cpu/whisper-large_cpu_int8.json) | [gemma4-e2b-int4-kquant-cuda](google-gemma-4-E2B-it/cuda/int4/config.json) | [microsoft-Phi-3-mini-4k-instruct](microsoft-Phi-3-mini-4k-instruct/aitk/phi3_vitis_ai_config.json) | +| [openai-whisper-large-v2-cpu-int8](openai-whisper-large-v2/cpu/whisper-large-v2_cpu_int8.json) | [google-bert-bert-base-multilingual-cased](google-bert-bert-base-multilingual-cased/QNN/config_gpu_fp32.json) | [microsoft-Phi-3.5-mini-instruct](microsoft-Phi-3.5-mini-instruct/QNN/config.json) | +| [openai-whisper-large-v3-cpu-int8](openai-whisper-large-v3/cpu/whisper-large-v3_cpu_int8.json) | [google-bert-bert-base-multilingual-cased](google-bert-bert-base-multilingual-cased/aitk/bert-base-multilingual-cased_context_ov_static.json) | [microsoft-Phi-3.5-mini-instruct](microsoft-Phi-3.5-mini-instruct/QNN/config_fp16.json) | +| [openai-whisper-large-v3-turbo-cpu-int8](openai-whisper-large-v3-turbo/cpu/whisper-large-v3-turbo_cpu_int8.json) | [google-bert-bert-base-multilingual-cased](google-bert-bert-base-multilingual-cased/aitk/bert-base-multilingual-cased_dml.json) | [microsoft-Phi-3.5-mini-instruct](microsoft-Phi-3.5-mini-instruct/aitk/phi3_5_ov_config.json) | +| [openai-whisper-large-v3-turbo](openai-whisper-large-v3-turbo/aitk/ov_workflow.json) | [google-bert-bert-base-multilingual-cased](google-bert-bert-base-multilingual-cased/aitk/bert-base-multilingual-cased_migraphx.json) | [microsoft-Phi-3.5-mini-instruct](microsoft-Phi-3.5-mini-instruct/aitk/phi3_5_qnn_config.json) | +| [openai-whisper-large-v3-turbo](openai-whisper-large-v3-turbo/olive/whisper_large_v3_turbo_decoder_fp32.json) | [google-bert-bert-base-multilingual-cased](google-bert-bert-base-multilingual-cased/aitk/bert-base-multilingual-cased_qnn_gpu.json) | [microsoft-Phi-3.5-mini-instruct](microsoft-Phi-3.5-mini-instruct/aitk/phi3_5_vitis_ai_config.json) | +| [openai-whisper-large-v3-turbo](openai-whisper-large-v3-turbo/olive/whisper_large_v3_turbo_decoder_qdq.json) | [google-bert-bert-base-multilingual-cased](google-bert-bert-base-multilingual-cased/aitk/bert-base-multilingual-cased_trtrtx.json) | [microsoft-Phi-4-mini-instruct](microsoft-Phi-4-mini-instruct/QNN/phi4_mini_qnn_docker.json) | +| [openai-whisper-large-v3-turbo](openai-whisper-large-v3-turbo/olive/whisper_large_v3_turbo_encoder_fp32.json) | [google-gemma-3-1b-it](google-gemma-3-1b-it/OpenVINO/gemma_3_1b_it_context_ov_gpu_config.json) | [microsoft-Phi-4-mini-instruct](microsoft-Phi-4-mini-instruct/aitk/phi4_ov_npu_config.json) | +| [openai-whisper-large-v3-turbo](openai-whisper-large-v3-turbo/olive/whisper_large_v3_turbo_encoder_qdq.json) | [google-gemma](google-gemma/olive/README.md) | [microsoft-Phi-4-mini-instruct](microsoft-Phi-4-mini-instruct/aitk/phi4_qnn.json) | +| [openai-whisper-medium-cpu-int8](openai-whisper-medium/cpu/whisper-medium_cpu_int8.json) | [google-vit-base-patch16-224](google-vit-base-patch16-224/QNN/config_gpu_fp32.json) | [microsoft-Phi-4-mini-instruct](microsoft-Phi-4-mini-instruct/aitk/phi4_vitis_ai_config.json) | +| [openai-whisper-medium.en-cpu-int8](openai-whisper-medium.en/cpu/whisper-medium.en_cpu_int8.json) | [google-vit-base-patch16-224](google-vit-base-patch16-224/aitk/vit-base-patch16-224_dml.json) | [microsoft-Phi-4-mini-reasoning](microsoft-Phi-4-mini-reasoning/aitk/phi4_ov_config.json) | +| [openai-whisper-small-cpu-int8](openai-whisper-small/cpu/whisper-small_cpu_int8.json) | [google-vit-base-patch16-224](google-vit-base-patch16-224/aitk/vit-base-patch16-224_migraphx.json) | [microsoft-Phi-4-mini-reasoning](microsoft-Phi-4-mini-reasoning/aitk/phi4_vitis_ai_config.json) | +| [openai-whisper-small.en-cpu-int8](openai-whisper-small.en/cpu/whisper-small.en_cpu_int8.json) | [google-vit-base-patch16-224](google-vit-base-patch16-224/aitk/vit-base-patch16-224_qnn_gpu.json) | [microsoft-Phi-4-reasoning-plus](microsoft-Phi-4-reasoning-plus/aitk/phi4_ov_config.json) | +| [openai-whisper-tiny-cpu-int8](openai-whisper-tiny/cpu/whisper-tiny_cpu_int8.json) | [google-vit-base-patch16-224](google-vit-base-patch16-224/aitk/vit-base-patch16-224_trtrtx.json) | [microsoft-Phi-4-reasoning](microsoft-Phi-4-reasoning/QAIRT/htp_sc8380xp.json) | +| [openai-whisper-tiny.en-cpu-int8](openai-whisper-tiny.en/cpu/whisper-tiny.en_cpu_int8.json) | [google-vit-base-patch16-224](google-vit-base-patch16-224/aitk/vit_base_patch16_224_context_ov_static.json) | [microsoft-Phi-4-reasoning](microsoft-Phi-4-reasoning/QAIRT/htp_sc8480xp.json) | +| [qwen2.5-vl-3B-Instruct](Qwen-Qwen2.5-VL-3B-Instruct/builtin/optimize.py) | [gpt-oss-20b](gpt-oss-20b/int4_cuda_int4_qmoe/gpt-oss-20b.sh) | [microsoft-Phi-4-reasoning](microsoft-Phi-4-reasoning/QNN/config.json) | +| [qwen3.5-0.8B-Instruct](Qwen-Qwen3.5-0.8B/builtin/optimize.py) | [intel-bert-base-uncased-mrpc (ov)](intel-bert-base-uncased-mrpc/aitk/bert_ov.json) | [microsoft-Phi-4-reasoning](microsoft-Phi-4-reasoning/aitk/phi4_ov_config.json) | +| [qwen3.5-2B](Qwen-Qwen3.5-2B/builtin/optimize.py) | [intel-bert-base-uncased-mrpc-inc-quant](intel-bert-base-uncased-mrpc/oci/gpu/cuda/inc_quant.json) | [microsoft-resnet-50](microsoft-resnet-50/aitk/resnet_context_ov_static.json) | +| [qwen3.5-35B-A3B-MoE](Qwen-Qwen3.5-35B-A3B/builtin/optimize.py) | [intel-bert-base-uncased-mrpc-ptq](intel-bert-base-uncased-mrpc/oci/gpu/cuda/ptq.json) | [microsoft-resnet-50](microsoft-resnet-50/aitk/resnet_qdq_amd.json) | +| [qwen3.5-4B](Qwen-Qwen3.5-4B/builtin/optimize.py) | [intel-bert-base-uncased-mrpc](intel-bert-base-uncased-mrpc/QNN/config_gpu_fp32.json) | [microsoft-resnet-50](microsoft-resnet-50/aitk/resnet_qdq_qnn.json) | +| [qwen3.5-9B](Qwen-Qwen3.5-9B/builtin/optimize.py) | [intel-bert-base-uncased-mrpc](intel-bert-base-uncased-mrpc/aitk/bert_dml.json) | [microsoft-table-transformer-detection](microsoft-table-transformer-detection/QNN/ttd_config.json) | +| [qwen3vl-2B-Instruct](Qwen-Qwen3-VL-2B-Instruct/builtin/optimize.py) | [intel-bert-base-uncased-mrpc](intel-bert-base-uncased-mrpc/aitk/bert_migraphx.json) | [mistralai-Mistral-7B-Instruct-v0.2](mistralai-Mistral-7B-Instruct-v0.2/aitk/Mistral_7B_Instruct_v0.2_npu_context_ov_dy.json) | +| [qwen3vl-4B-Instruct](Qwen-Qwen3-VL-4B-Instruct/builtin/optimize.py) | [intel-bert-base-uncased-mrpc](intel-bert-base-uncased-mrpc/aitk/bert_qnn_gpu.json) | [mistralai-Mistral-7B-Instruct-v0.2](mistralai-Mistral-7B-Instruct-v0.2/aitk/Mistral_7B_Instruct_v0.2_vitis_ai_config.json) | +| [qwen3vl-8B-Instruct](Qwen-Qwen3-VL-8B-Instruct/builtin/optimize.py) | [intel-bert-base-uncased-mrpc](intel-bert-base-uncased-mrpc/aitk/bert_trtrtx.json) | [openai-clip-vit-base-patch16](openai-clip-vit-base-patch16/aitk/openai_clip_ov.json) | +| [sam2.1-hiera-small](sam2.1-hiera-small/OpenVINO/sam21_mask_decoder_ov.json) | [laion-CLIP-ViT-B-32-laion2B-s34B-b79K](laion-CLIP-ViT-B-32-laion2B-s34B-b79K/QNN/config_gpu_fp32.json) | [openai-clip-vit-base-patch16](openai-clip-vit-base-patch16/aitk/openai_clip_qdq_amd.json) | +| [sam2.1-hiera-small](sam2.1-hiera-small/OpenVINO/sam21_vision_encoder_ov.json) | [laion-CLIP-ViT-B-32-laion2B-s34B-b79K](laion-CLIP-ViT-B-32-laion2B-s34B-b79K/aitk/laion_clip_dml.json) | [openai-clip-vit-base-patch16](openai-clip-vit-base-patch16/aitk/openai_clip_qnn.json) | +| [sam2.1-hiera-small](sam2.1-hiera-small/aitk/sam2_ov_workflow.json) | [laion-CLIP-ViT-B-32-laion2B-s34B-b79K](laion-CLIP-ViT-B-32-laion2B-s34B-b79K/aitk/laion_clip_migraphx.json) | [openai-clip-vit-base-patch32](openai-clip-vit-base-patch32/aitk/openai_clip_ov.json) | +| [sd-legacy-stable-diffusion-v1-5](sd-legacy-stable-diffusion-v1-5/aitk/sd_ov_workflow.json) | [laion-CLIP-ViT-B-32-laion2B-s34B-b79K](laion-CLIP-ViT-B-32-laion2B-s34B-b79K/aitk/laion_clip_ov.json) | [openai-clip-vit-base-patch32](openai-clip-vit-base-patch32/aitk/openai_clip_qdq_amd.json) | +| [sd2-community-stable-diffusion-2-1](sd2-community-stable-diffusion-2-1/aitk/sd_ov_workflow.json) | [laion-CLIP-ViT-B-32-laion2B-s34B-b79K](laion-CLIP-ViT-B-32-laion2B-s34B-b79K/aitk/laion_clip_qnn_gpu.json) | [openai-clip-vit-base-patch32](openai-clip-vit-base-patch32/aitk/openai_clip_qnn.json) | +| [sshleifer-tiny-gpt2-sparsegpt](sshleifer-tiny-gpt2/olive/sparsegpt.json) | [laion-CLIP-ViT-B-32-laion2B-s34B-b79K](laion-CLIP-ViT-B-32-laion2B-s34B-b79K/aitk/laion_clip_trtrtx.json) | [openai-clip-vit-large-patch14](openai-clip-vit-large-patch14/aitk/openai_clip_ov.json) | +| [stable-diffusion-v1-4-safety-checker](compvis-stable-diffusion-v1-4/olive/config_safety_checker.json) | [meta-llama-Llama-3.1-8B-Instruct](meta-llama-Llama-3.1-8B-Instruct/aitk/llama3_1_dml_config.json) | [openai-clip-vit-large-patch14](openai-clip-vit-large-patch14/aitk/openai_clip_qdq_amd.json) | +| [stable-diffusion-v1-4-text-encoder](compvis-stable-diffusion-v1-4/olive/config_text_encoder.json) | [meta-llama-Llama-3.1-8B-Instruct](meta-llama-Llama-3.1-8B-Instruct/aitk/llama3_1_ov_gpu_config.json) | [openai-clip-vit-large-patch14](openai-clip-vit-large-patch14/aitk/openai_clip_qnn.json) | +| [stable-diffusion-v1-4-unet](compvis-stable-diffusion-v1-4/olive/config_unet.json) | [meta-llama-Llama-3.1-8B-Instruct](meta-llama-Llama-3.1-8B-Instruct/aitk/llama3_1_qnn_gpu.json) | [openai-whisper-large-v3-turbo-vitisai](openai-whisper-large-v3-turbo/VitisAI/run_whisper.py) | +| [stable-diffusion-v1-4-vae-decoder](compvis-stable-diffusion-v1-4/olive/config_vae_decoder.json) | [meta-llama-Llama-3.1-8B-Instruct](meta-llama-Llama-3.1-8B-Instruct/aitk/llama3_1_trtrtx_config.json) | [openai-whisper-large-v3-turbo](openai-whisper-large-v3-turbo/OpenVINO/whisper_large_v3_turbo_default_ov_npu.json) | +| [stable-diffusion-v1-4-vae-encoder](compvis-stable-diffusion-v1-4/olive/config_vae_encoder.json) | [meta-llama-Llama-3.2-1B-Instruct-dora](meta-llama-Llama-3.2-1B-Instruct/olive/dora.json) | [openai-whisper-large-v3-turbo](openai-whisper-large-v3-turbo/aitk/ov_npu_workflow.json) | +| [stable-diffusion-v1-5-safety-checker](sd-legacy-stable-diffusion-v1-5/VitisAI/config_safety_checker.json) | [meta-llama-Llama-3.2-1B-Instruct-hqq](meta-llama-Llama-3.2-1B-Instruct/olive/hqq.json) | [openai-whisper-large-v3-turbo](openai-whisper-large-v3-turbo/aitk/qnn_workflow.json) | +| [stable-diffusion-v1-5-text-encoder](sd-legacy-stable-diffusion-v1-5/VitisAI/config_text_encoder.json) | [meta-llama-Llama-3.2-1B-Instruct-lmeval-onnx](meta-llama-Llama-3.2-1B-Instruct/olive/lmeval_onnx.json) | [openai-whisper-large-v3-turbo](openai-whisper-large-v3-turbo/olive/whisper_large_v3_turbo_decoder_fp32.json) | +| [stable-diffusion-v1-5-unet](sd-legacy-stable-diffusion-v1-5/VitisAI/config_unet.json) | [meta-llama-Llama-3.2-1B-Instruct-lmeval](meta-llama-Llama-3.2-1B-Instruct/olive/lmeval.json) | [openai-whisper-large-v3-turbo](openai-whisper-large-v3-turbo/olive/whisper_large_v3_turbo_decoder_qdq.json) | +| [stable-diffusion-v1-5-vae-decoder](sd-legacy-stable-diffusion-v1-5/VitisAI/config_vae_decoder.json) | [meta-llama-Llama-3.2-1B-Instruct-loha](meta-llama-Llama-3.2-1B-Instruct/olive/loha.json) | [openai-whisper-large-v3-turbo](openai-whisper-large-v3-turbo/olive/whisper_large_v3_turbo_decoder_qdq_ctx.json) | +| [stable-diffusion-v1-5-vae-encoder](sd-legacy-stable-diffusion-v1-5/VitisAI/config_vae_encoder.json) | [meta-llama-Llama-3.2-1B-Instruct-lokr](meta-llama-Llama-3.2-1B-Instruct/olive/lokr.json) | [openai-whisper-large-v3-turbo](openai-whisper-large-v3-turbo/olive/whisper_large_v3_turbo_encoder_fp32.json) | +| [stable-diffusion-v1-5](sd-legacy-stable-diffusion-v1-5/olive/optimize.sh) | [meta-llama-Llama-3.2-1B-Instruct-mixed](meta-llama-Llama-3.2-1B-Instruct/olive/mixed.json) | [openai-whisper-large-v3-turbo](openai-whisper-large-v3-turbo/olive/whisper_large_v3_turbo_encoder_qdq.json) | +| [stable-diffusion-xl-base-1.0](stabilityai-stable-diffusion-xl-base-1.0/olive/stable_diffusion_xl.py) | [meta-llama-Llama-3.2-1B-Instruct-qlora](meta-llama-Llama-3.2-1B-Instruct/olive/qlora.json) | [openai-whisper-large-v3-turbo](openai-whisper-large-v3-turbo/olive/whisper_large_v3_turbo_encoder_qdq_ctx.json) | +| [timm-mobilenetv3_small_100.lamb_in1k](timm-mobilenetv3_small_100.lamb_in1k/olive/config.json) | [meta-llama-Llama-3.2-1B-Instruct](meta-llama-Llama-3.2-1B-Instruct/QNN/config_gpu.json) | [openai-whisper-medium-vitisai](openai-whisper-medium/VitisAI/run_whisper.py) | +| [translategemma-4b-it](google-translategemma-4b-it/builtin/optimize.py) | [meta-llama-Llama-3.2-1B-Instruct](meta-llama-Llama-3.2-1B-Instruct/QNN/config_gpu_ctxbin.json) | [openai-whisper-small-vitisai](openai-whisper-small/VitisAI/run_whisper.py) | +| [videochat-flash-qwen2_5-7b-internvideo2-1b](OpenGVLab-VideoChat-Flash-Qwen2_5-7B_InternVideo2-1B/builtin/optimize.py) | [meta-llama-Llama-3.2-1B-Instruct](meta-llama-Llama-3.2-1B-Instruct/aitk/llama3_2_dml_config.json) | [qwen2.5-7b-instruct](Qwen-Qwen2.5-7B-Instruct/QNN/config.json) | +| | [meta-llama-Llama-3.2-1B-Instruct](meta-llama-Llama-3.2-1B-Instruct/aitk/llama3_2_ov_gpu_config.json) | [sam-vit-base](sam-vit-base/QNN/sam_mask_box_decoder_qnn_fp16.json) | +| | [meta-llama-Llama-3.2-1B-Instruct](meta-llama-Llama-3.2-1B-Instruct/aitk/llama3_2_qnn_gpu_config.json) | [sam-vit-base](sam-vit-base/QNN/sam_mask_decoder_qnn_fp16_ctx.json) | +| | [meta-llama-Llama-3.2-1B-Instruct](meta-llama-Llama-3.2-1B-Instruct/aitk/llama3_2_trtrtx_config.json) | [sam-vit-base](sam-vit-base/QNN/sam_mask_point_decoder_qnn_fp16.json) | +| | [microsoft-Phi-3-mini-128k-instruct](microsoft-Phi-3-mini-128k-instruct/aitk/phi3_ov_config.json) | [sam-vit-base](sam-vit-base/QNN/sam_vision_encoder_qnn_w8a8_ctx.json) | +| | [microsoft-Phi-3-mini-128k-instruct](microsoft-Phi-3-mini-128k-instruct/aitk/phi3_trtrtx.json) | [sam-vit-base](sam-vit-base/aitk/sam_qnn_workflow.json) | +| | [microsoft-Phi-3-mini-4k-instruct](microsoft-Phi-3-mini-4k-instruct/aitk/phi3_ov_config.json) | [sam2.1-hiera-small](sam2.1-hiera-small/OpenVINO/sam21_mask_decoder_ov.json) | +| | [microsoft-Phi-3-mini-4k-instruct](microsoft-Phi-3-mini-4k-instruct/aitk/phi3_trtrtx.json) | [sam2.1-hiera-small](sam2.1-hiera-small/OpenVINO/sam21_vision_encoder_ov.json) | | | [microsoft-Phi-3.5-mini-instruct](microsoft-Phi-3.5-mini-instruct/QNN/config_gpu.json) | [sam2.1-hiera-small](sam2.1-hiera-small/QNN/sam21_mask_decoder_qnn_ctx.json) | | | [microsoft-Phi-3.5-mini-instruct](microsoft-Phi-3.5-mini-instruct/QNN/config_gpu_ctxbin.json) | [sam2.1-hiera-small](sam2.1-hiera-small/QNN/sam21_vision_encoder_qnn_ctx.json) | -| | [microsoft-Phi-3.5-mini-instruct](microsoft-Phi-3.5-mini-instruct/aitk/phi3_5_dml_config.json) | [sam2.1-hiera-small](sam2.1-hiera-small/aitk/sam2_qnn_workflow.json) | -| | [microsoft-Phi-3.5-mini-instruct](microsoft-Phi-3.5-mini-instruct/aitk/phi3_5_ov_gpu_config.json) | [sd-legacy-stable-diffusion-v1-5](sd-legacy-stable-diffusion-v1-5/aitk/sd_ov_npu_workflow.json) | -| | [microsoft-Phi-3.5-mini-instruct](microsoft-Phi-3.5-mini-instruct/aitk/phi3_5_qnn_gpu_config.json) | [sd-legacy-stable-diffusion-v1-5](sd-legacy-stable-diffusion-v1-5/aitk/sd_qnn_workflow.json) | -| | [microsoft-Phi-3.5-mini-instruct](microsoft-Phi-3.5-mini-instruct/aitk/phi3_5_trtrtx_config.json) | [sd2-community-stable-diffusion-2-1](sd2-community-stable-diffusion-2-1/aitk/sd_ov_npu_workflow.json) | -| | [microsoft-Phi-3.5-mini-instruct](microsoft-Phi-3.5-mini-instruct/olive/mixed.json) | [sd2-community-stable-diffusion-2-1](sd2-community-stable-diffusion-2-1/aitk/sd_qnn_workflow.json) | -| | [microsoft-Phi-4-mini-instruct-mixed-tied](microsoft-Phi-4-mini-instruct/olive/mixed-tied.json) | [stable-diffusion-v1-4-safety-checker](compvis-stable-diffusion-v1-4/olive/config_safety_checker.json) | -| | [microsoft-Phi-4-mini-instruct-mixed](microsoft-Phi-4-mini-instruct/olive/mixed.json) | [stable-diffusion-v1-4-text-encoder](compvis-stable-diffusion-v1-4/olive/config_text_encoder.json) | -| | [microsoft-Phi-4-mini-instruct](microsoft-Phi-4-mini-instruct/OpenVINO/Phi-4-mini-instruct-gpu-context-dy.json) | [stable-diffusion-v1-4-unet](compvis-stable-diffusion-v1-4/olive/config_unet.json) | -| | [microsoft-Phi-4-mini-instruct](microsoft-Phi-4-mini-instruct/OpenVINO/Phi_4_mini_instruct_context_ov_dynamic_sym_gs128_bkp_int8_sym.json) | [stable-diffusion-v1-4-vae-decoder](compvis-stable-diffusion-v1-4/olive/config_vae_decoder.json) | -| | [microsoft-Phi-4-mini-instruct](microsoft-Phi-4-mini-instruct/aitk/phi4_ov_config.json) | [stable-diffusion-v1-4-vae-encoder](compvis-stable-diffusion-v1-4/olive/config_vae_encoder.json) | -| | [microsoft-Phi-4-mini-instruct_nvmo_ptq_mixed_precision_awq_lite](microsoft-Phi-4-mini-instruct/NvTensorRtRtx/microsoft-Phi-4-mini-instruct_nvmo_ptq_mixed_precision_awq_lite.json) | [stable-diffusion-v1-5](sd-legacy-stable-diffusion-v1-5/olive/optimize.sh) | -| | [microsoft-Phi-4-mini-reasoning](microsoft-Phi-4-mini-reasoning/OpenVINO/Phi-4-mini-reasoning_context_ov_dynamic_sym_gs128_bkp_int8_sym.json) | [stable-diffusion-xl-base-1.0](stabilityai-stable-diffusion-xl-base-1.0/olive/stable_diffusion_xl.py) | -| | [microsoft-Phi-4-mini-reasoning](microsoft-Phi-4-mini-reasoning/OpenVINO/Phi_4_mini_instruct_context_ov_dynamic_sym_gs128_bkp_int8_sym.json) | [timm-mobilenetv3_small_100.lamb_in1k](timm-mobilenetv3_small_100.lamb_in1k/QNN/mobilenet_qnn_ep.json) | -| | [microsoft-Phi-4-mini-reasoning](microsoft-Phi-4-mini-reasoning/aitk/phi4_ov_gpu_config.json) | [timm-mobilenetv3_small_100.lamb_in1k](timm-mobilenetv3_small_100.lamb_in1k/VitisAI/config.json) | -| | [microsoft-Phi-4-reasoning-plus](microsoft-Phi-4-reasoning-plus/OpenVINO/Phi-4-Phi-4-reasoning-plus_context_ov_dynamic_sym_gs128_bkp_int8_sym.json) | | -| | [microsoft-Phi-4-reasoning](microsoft-Phi-4-reasoning/OpenVINO/Phi-4-reasoning_context_ov_dynamic_sym_gs128_bkp_int8_sym.json) | | -| | [microsoft-Phi-4](microsoft-Phi-4/OpenVINO/phi_4_gpu_context_dy.json) | | -| | [microsoft-Phi-4](microsoft-Phi-4/aitk/phi4_ov_config.json) | | -| | [microsoft-Phi-4](microsoft-Phi-4/aitk/phi4_trtrtx.json) | | +| | [microsoft-Phi-3.5-mini-instruct](microsoft-Phi-3.5-mini-instruct/aitk/phi3_5_dml_config.json) | [sam2.1-hiera-small](sam2.1-hiera-small/aitk/sam2_ov_workflow.json) | +| | [microsoft-Phi-3.5-mini-instruct](microsoft-Phi-3.5-mini-instruct/aitk/phi3_5_ov_gpu_config.json) | [sam2.1-hiera-small](sam2.1-hiera-small/aitk/sam2_qnn_workflow.json) | +| | [microsoft-Phi-3.5-mini-instruct](microsoft-Phi-3.5-mini-instruct/aitk/phi3_5_qnn_gpu_config.json) | [sd-legacy-stable-diffusion-v1-5](sd-legacy-stable-diffusion-v1-5/aitk/sd_ov_npu_workflow.json) | +| | [microsoft-Phi-3.5-mini-instruct](microsoft-Phi-3.5-mini-instruct/aitk/phi3_5_trtrtx_config.json) | [sd-legacy-stable-diffusion-v1-5](sd-legacy-stable-diffusion-v1-5/aitk/sd_qnn_workflow.json) | +| | [microsoft-Phi-3.5-mini-instruct](microsoft-Phi-3.5-mini-instruct/olive/mixed.json) | [sd2-community-stable-diffusion-2-1](sd2-community-stable-diffusion-2-1/aitk/sd_ov_npu_workflow.json) | +| | [microsoft-Phi-4-mini-instruct-mixed-tied](microsoft-Phi-4-mini-instruct/olive/mixed-tied.json) | [sd2-community-stable-diffusion-2-1](sd2-community-stable-diffusion-2-1/aitk/sd_qnn_workflow.json) | +| | [microsoft-Phi-4-mini-instruct-mixed](microsoft-Phi-4-mini-instruct/olive/mixed.json) | [stable-diffusion-v1-4-safety-checker](compvis-stable-diffusion-v1-4/olive/config_safety_checker.json) | +| | [microsoft-Phi-4-mini-instruct](microsoft-Phi-4-mini-instruct/OpenVINO/Phi-4-mini-instruct-gpu-context-dy.json) | [stable-diffusion-v1-4-text-encoder](compvis-stable-diffusion-v1-4/olive/config_text_encoder.json) | +| | [microsoft-Phi-4-mini-instruct](microsoft-Phi-4-mini-instruct/OpenVINO/Phi_4_mini_instruct_context_ov_dynamic_sym_gs128_bkp_int8_sym.json) | [stable-diffusion-v1-4-unet](compvis-stable-diffusion-v1-4/olive/config_unet.json) | +| | [microsoft-Phi-4-mini-instruct](microsoft-Phi-4-mini-instruct/aitk/phi4_ov_config.json) | [stable-diffusion-v1-4-vae-decoder](compvis-stable-diffusion-v1-4/olive/config_vae_decoder.json) | +| | [microsoft-Phi-4-mini-instruct](microsoft-Phi-4-mini-instruct/aitk/phi4_trtrtx.json) | [stable-diffusion-v1-4-vae-encoder](compvis-stable-diffusion-v1-4/olive/config_vae_encoder.json) | +| | [microsoft-Phi-4-mini-instruct_nvmo_ptq_mixed_precision_awq_lite](microsoft-Phi-4-mini-instruct/NvTensorRtRtx/microsoft-Phi-4-mini-instruct_nvmo_ptq_mixed_precision_awq_lite.json) | [stable-diffusion-v1-5-safety-checker](sd-legacy-stable-diffusion-v1-5/VitisAI/config_safety_checker.json) | +| | [microsoft-Phi-4-mini-reasoning](microsoft-Phi-4-mini-reasoning/OpenVINO/Phi-4-mini-reasoning_context_ov_dynamic_sym_gs128_bkp_int8_sym.json) | [stable-diffusion-v1-5-text-encoder](sd-legacy-stable-diffusion-v1-5/VitisAI/config_text_encoder.json) | +| | [microsoft-Phi-4-mini-reasoning](microsoft-Phi-4-mini-reasoning/OpenVINO/Phi_4_mini_instruct_context_ov_dynamic_sym_gs128_bkp_int8_sym.json) | [stable-diffusion-v1-5-unet](sd-legacy-stable-diffusion-v1-5/VitisAI/config_unet.json) | +| | [microsoft-Phi-4-mini-reasoning](microsoft-Phi-4-mini-reasoning/aitk/phi4_ov_gpu_config.json) | [stable-diffusion-v1-5-vae-decoder](sd-legacy-stable-diffusion-v1-5/VitisAI/config_vae_decoder.json) | +| | [microsoft-Phi-4-reasoning-plus](microsoft-Phi-4-reasoning-plus/OpenVINO/Phi-4-Phi-4-reasoning-plus_context_ov_dynamic_sym_gs128_bkp_int8_sym.json) | [stable-diffusion-v1-5-vae-encoder](sd-legacy-stable-diffusion-v1-5/VitisAI/config_vae_encoder.json) | +| | [microsoft-Phi-4-reasoning](microsoft-Phi-4-reasoning/OpenVINO/Phi-4-reasoning_context_ov_dynamic_sym_gs128_bkp_int8_sym.json) | [stable-diffusion-v1-5](sd-legacy-stable-diffusion-v1-5/olive/optimize.sh) | +| | [microsoft-Phi-4](microsoft-Phi-4/OpenVINO/phi_4_gpu_context_dy.json) | [stable-diffusion-xl-base-1.0](stabilityai-stable-diffusion-xl-base-1.0/olive/stable_diffusion_xl.py) | +| | [microsoft-Phi-4](microsoft-Phi-4/aitk/phi4_ov_config.json) | [timm-mobilenetv3_small_100.lamb_in1k](timm-mobilenetv3_small_100.lamb_in1k/QNN/mobilenet_qnn_ep.json) | +| | [microsoft-Phi-4](microsoft-Phi-4/aitk/phi4_trtrtx.json) | [timm-mobilenetv3_small_100.lamb_in1k](timm-mobilenetv3_small_100.lamb_in1k/VitisAI/config.json) | | | [microsoft-resnet-50](microsoft-resnet-50/aitk/resnet_context_ov_static.json) | | | | [microsoft-resnet-50](microsoft-resnet-50/aitk/resnet_dml.json) | | | | [microsoft-resnet-50](microsoft-resnet-50/aitk/resnet_migraphx.json) | | @@ -265,87 +269,90 @@ Below are list of available recipes grouped by different criteria. Click the lin | CPU | CUDA | Dml | MIGraphX | NvTensorRTRTX | OpenVINO | QNN | VitisAI | WebGpu | | :---: | :---: | :---: | :---: | :---: | :---: | :---: | :---: | :---: | -| [alibaba-nlp-gte-large-en-v1.5](alibaba-nlp-gte-large-en-v1.5/quant/config.json) | [Qwen-Qwen2.5-1.5B-Instruct-mixed](Qwen-Qwen2.5-1.5B-Instruct/olive/mixed.json) | [Qwen-Qwen2.5-1.5B-Instruct](Qwen-Qwen2.5-1.5B-Instruct/aitk/qwen2_5_dml_config.json) | [google-bert-bert-base-multilingual-cased](google-bert-bert-base-multilingual-cased/aitk/bert-base-multilingual-cased_migraphx.json) | [DeepSeek-R1-Distill-Llama-8B_Model_Builder_INT4](deepseek-ai-DeepSeek-R1-Distill-Llama-8B/NvTensorRtRtx/DeepSeek-R1-Distill-Llama-8B_model_builder_int4.json) | [OFA-Sys-chinese-clip-vit-base-patch16](OFA-Sys-chinese-clip-vit-base-patch16/aitk/openai_clip_ov.json) | [Qwen-Qwen2.5-1.5B-Instruct](Qwen-Qwen2.5-1.5B-Instruct/QNN/config.json) | [Qwen-Qwen2.5-0.5B-Instruct](Qwen-Qwen2.5-0.5B-Instruct/aitk/qwen2_5_vitis_ai_config.json) | [openai-whisper-base-webgpu-int8](openai-whisper-base/webgpu/whisper-base_webgpu_int8.json) | -| [facebook-opt-125m-splicegpt](facebook-opt-125m/olive/slicegpt.json) | [deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B](deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B/olive/mixed.json) | [deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B](deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B/aitk/deepseek_dml_config.json) | [google-vit-base-patch16-224](google-vit-base-patch16-224/aitk/vit-base-patch16-224_migraphx.json) | [DeepSeek-R1-Distill-Qwen-1.5B_Model_Builder_FP16](deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B/NvTensorRtRtx/DeepSeek-R1-Distill-Qwen-1.5B_model_builder_fp16.json) | [Qwen-Qwen2.5-0.5B-Instruct](Qwen-Qwen2.5-0.5B-Instruct/aitk/qwen2_5_ov_config.json) | [Qwen-Qwen2.5-1.5B-Instruct](Qwen-Qwen2.5-1.5B-Instruct/QNN/config_gpu.json) | [Qwen-Qwen2.5-1.5B-Instruct](Qwen-Qwen2.5-1.5B-Instruct/aitk/qwen2_5_vitis_ai_config.json) | [openai-whisper-base.en-webgpu-int8](openai-whisper-base.en/webgpu/whisper-base.en_webgpu_int8.json) | -| [gemma-3-1b-it_model_builder_cpu_FP32](google-gemma/olive/gemma-3-1b-it_model_builder_cpu_fp32.json) | [facebook-opt-125m-splicegpt](facebook-opt-125m/olive/slicegpt.json) | [google-bert-bert-base-multilingual-cased](google-bert-bert-base-multilingual-cased/aitk/bert-base-multilingual-cased_dml.json) | [intel-bert-base-uncased-mrpc](intel-bert-base-uncased-mrpc/aitk/bert_migraphx.json) | [DeepSeek-R1-Distill-Qwen-14B_NVMO_INT4_AWQ](deepseek-ai-DeepSeek-R1-Distill-Qwen-14B/NvTensorRtRtx/DeepSeek-R1-Distill-Qwen-14B_nvmo_int4_awq.json) | [Qwen-Qwen2.5-0.5B-Instruct](Qwen-Qwen2.5-0.5B-Instruct/aitk/qwen2_5_ov_npu_config.json) | [Qwen-Qwen2.5-1.5B-Instruct](Qwen-Qwen2.5-1.5B-Instruct/QNN/config_gpu_ctxbin.json) | [Qwen-Qwen2.5-7B-Instruct](Qwen-Qwen2.5-7B-Instruct/aitk/qwen2_5_vitis_ai_config.json) | [openai-whisper-large-v2-webgpu-int8](openai-whisper-large-v2/webgpu/whisper-large-v2_webgpu_int8.json) | -| [google-gemma](google-gemma/olive/README.md) | [google-gemma](google-gemma/olive/README.md) | [google-vit-base-patch16-224](google-vit-base-patch16-224/aitk/vit-base-patch16-224_dml.json) | [laion-CLIP-ViT-B-32-laion2B-s34B-b79K](laion-CLIP-ViT-B-32-laion2B-s34B-b79K/aitk/laion_clip_migraphx.json) | [DeepSeek-R1-Distill-Qwen-7B_NVMO_INT4_RTN](deepseek-ai-DeepSeek-R1-Distill-Qwen-7B/NvTensorRtRtx/DeepSeek-R1-Distill-Qwen-7B_nvmo_int4_rtn.json) | [Qwen-Qwen2.5-0.5B](Qwen-Qwen2.5-0.5B/aitk/qwen2_5_ov_config.json) | [Qwen-Qwen2.5-1.5B-Instruct](Qwen-Qwen2.5-1.5B-Instruct/aitk/qwen2_5_qnn_config.json) | [Qwen-Qwen2.5-Coder-0.5B-Instruct](Qwen-Qwen2.5-Coder-0.5B-Instruct/aitk/qwen2_5_vitis_ai_config.json) | [openai-whisper-large-v3-turbo-webgpu-int8](openai-whisper-large-v3-turbo/webgpu/whisper-large-v3-turbo_webgpu_int8.json) | -| [gpt-oss-20b](gpt-oss-20b/int4_cuda_int4_qmoe/gpt-oss-20b.sh) | [gpt-oss-20b](gpt-oss-20b/int4_cuda_int4_qmoe/gpt-oss-20b.sh) | [intel-bert-base-uncased-mrpc](intel-bert-base-uncased-mrpc/aitk/bert_dml.json) | [microsoft-resnet-50](microsoft-resnet-50/aitk/resnet_migraphx.json) | [Llama-3.2-1B-Instruct_Model_Builder_FP16](meta-llama-Llama-3.2-1B-Instruct/NvTensorRtRtx/Llama-3.2-1B-Instruct_model_builder_fp16.json) | [Qwen-Qwen2.5-1.5B-Instruct](Qwen-Qwen2.5-1.5B-Instruct/aitk/qwen2_5_ov_config.json) | [Qwen-Qwen2.5-1.5B-Instruct](Qwen-Qwen2.5-1.5B-Instruct/aitk/qwen2_5_qnn_gpu_config.json) | [Qwen-Qwen2.5-Coder-1.5B-Instruct](Qwen-Qwen2.5-Coder-1.5B-Instruct/aitk/qwen2_5_vitis_ai_config.json) | [openai-whisper-large-v3-webgpu-int8](openai-whisper-large-v3/webgpu/whisper-large-v3_webgpu_int8.json) | -| [intel-bert-base-uncased-mrpc-inc-smooth-quant](intel-bert-base-uncased-mrpc/oci/cpu/inc_smooth_quant.json) | [intel-bert-base-uncased-mrpc-inc-quant](intel-bert-base-uncased-mrpc/oci/gpu/cuda/inc_quant.json) | [laion-CLIP-ViT-B-32-laion2B-s34B-b79K](laion-CLIP-ViT-B-32-laion2B-s34B-b79K/aitk/laion_clip_dml.json) | [openai-clip-vit-base-patch16](openai-clip-vit-base-patch16/aitk/openai_clip_migraphx.json) | [Llama3.1-8B-Instruct_Model_Builder_INT4](meta-llama-Llama-3.1-8B-Instruct/NvTensorRtRtx/Llama-3.1-8B-Instruct_model_builder_int4.json) | [Qwen-Qwen2.5-1.5B-Instruct](Qwen-Qwen2.5-1.5B-Instruct/aitk/qwen2_5_ov_gpu_config.json) | [Qwen-Qwen2.5-7B-Instruct](Qwen-Qwen2.5-7B-Instruct/aitk/qwen2_5_qnn_config.json) | [Qwen-Qwen2.5-Coder-7B-Instruct](Qwen-Qwen2.5-Coder-7B-Instruct/aitk/qwen2_5_vitis_ai_config.json) | [openai-whisper-large-webgpu-int8](openai-whisper-large/webgpu/whisper-large_webgpu_int8.json) | -| [intel-bert-base-uncased-mrpc-ptq](intel-bert-base-uncased-mrpc/oci/cpu/ptq.json) | [intel-bert-base-uncased-mrpc-ptq](intel-bert-base-uncased-mrpc/oci/gpu/cuda/ptq.json) | [meta-llama-Llama-3.1-8B-Instruct](meta-llama-Llama-3.1-8B-Instruct/aitk/llama3_1_dml_config.json) | [openai-clip-vit-base-patch32](openai-clip-vit-base-patch32/aitk/openai_clip_migraphx.json) | [Mistral-7B-Instruct-v0.2_Model_Builder_INT4](mistralai-Mistral-7B-Instruct-v0.2/NvTensorRtRtx/Mistral-7B-Instruct-v0.2_model_builder_int4.json) | [Qwen-Qwen2.5-14B-Instruct](Qwen-Qwen2.5-14B-Instruct/aitk/qwen2_5_ov_config.json) | [deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B](deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B/QNN/config_gpu.json) | [deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B](deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B/aitk/deepseek_vitis_ai_config.json) | [openai-whisper-medium-webgpu-int8](openai-whisper-medium/webgpu/whisper-medium_webgpu_int8.json) | -| [meta-llama-Llama-3.2-1B-Instruct-dora](meta-llama-Llama-3.2-1B-Instruct/olive/dora.json) | [meta-llama-Llama-3.2-1B-Instruct-dora](meta-llama-Llama-3.2-1B-Instruct/olive/dora.json) | [meta-llama-Llama-3.2-1B-Instruct](meta-llama-Llama-3.2-1B-Instruct/aitk/llama3_2_dml_config.json) | [openai-clip-vit-large-patch14](openai-clip-vit-large-patch14/aitk/openai_clip_migraphx.json) | [Phi-3-mini-128k-instruct_NVMO_INT4_RTN](microsoft-Phi-3-mini-128k-instruct/NvTensorRtRtx/Phi-3-mini-128k-instruct_nvmo_int4_rtn.json) | [Qwen-Qwen2.5-14B-Instruct](Qwen-Qwen2.5-14B-Instruct/aitk/qwen2_5_ov_npu_config.json) | [deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B](deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B/QNN/config_gpu_ctxbin.json) | [deepseek-ai-DeepSeek-R1-Distill-Qwen-7B](deepseek-ai-DeepSeek-R1-Distill-Qwen-7B/aitk/deepseek_vitis_ai_config.json) | [openai-whisper-medium.en-webgpu-int8](openai-whisper-medium.en/webgpu/whisper-medium.en_webgpu_int8.json) | -| [meta-llama-Llama-3.2-1B-Instruct-hqq](meta-llama-Llama-3.2-1B-Instruct/olive/hqq.json) | [meta-llama-Llama-3.2-1B-Instruct-hqq](meta-llama-Llama-3.2-1B-Instruct/olive/hqq.json) | [microsoft-Phi-3.5-mini-instruct](microsoft-Phi-3.5-mini-instruct/aitk/phi3_5_dml_config.json) | | [Phi-3-mini-4k-instruct_Model_Builder_INT4](microsoft-Phi-3-mini-4k-instruct/NvTensorRtRtx/Phi-3-mini-4k-instruct_model_builder_int4.json) | [Qwen-Qwen2.5-3B-Instruct](Qwen-Qwen2.5-3B-Instruct/aitk/qwen2_5_ov_config.json) | [deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B](deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B/aitk/deepseek_qnn_config.json) | [google-bert-bert-base-multilingual-cased](google-bert-bert-base-multilingual-cased/aitk/bert-base-multilingual-cased_qdq_amd.json) | [openai-whisper-small-webgpu-int8](openai-whisper-small/webgpu/whisper-small_webgpu_int8.json) | -| [meta-llama-Llama-3.2-1B-Instruct-lmeval-onnx](meta-llama-Llama-3.2-1B-Instruct/olive/lmeval_onnx.json) | [meta-llama-Llama-3.2-1B-Instruct-lmeval-onnx](meta-llama-Llama-3.2-1B-Instruct/olive/lmeval_onnx.json) | [microsoft-resnet-50](microsoft-resnet-50/aitk/resnet_dml.json) | | [Phi3.5_Mini_Instruct_Model_Builder_INT4](microsoft-Phi-3.5-mini-instruct/NvTensorRtRtx/Phi-3.5-mini-instruct_model_builder_int4.json) | [Qwen-Qwen2.5-3B-Instruct](Qwen-Qwen2.5-3B-Instruct/aitk/qwen2_5_ov_npu_config.json) | [deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B](deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B/aitk/deepseek_qnn_gpu_config.json) | [google-vit-base-patch16-224](google-vit-base-patch16-224/aitk/vit-base-patch16-224_qdq_amd.json) | [openai-whisper-small.en-webgpu-int8](openai-whisper-small.en/webgpu/whisper-small.en_webgpu_int8.json) | -| [meta-llama-Llama-3.2-1B-Instruct-lmeval](meta-llama-Llama-3.2-1B-Instruct/olive/lmeval.json) | [meta-llama-Llama-3.2-1B-Instruct-lmeval](meta-llama-Llama-3.2-1B-Instruct/olive/lmeval.json) | [openai-clip-vit-base-patch16](openai-clip-vit-base-patch16/aitk/openai_clip_dml.json) | | [Qwen-Qwen2.5-0.5B-Instruct](Qwen-Qwen2.5-0.5B-Instruct/aitk/qwen2_5_trtrtx.json) | [Qwen-Qwen2.5-7B-Instruct](Qwen-Qwen2.5-7B-Instruct/aitk/qwen2_5_ov_config.json) | [google-bert-bert-base-multilingual-cased](google-bert-bert-base-multilingual-cased/QNN/config_gpu_fp32.json) | [intel-bert-base-uncased-mrpc (AMD)](intel-bert-base-uncased-mrpc/aitk/bert_qdq_amd.json) | [openai-whisper-tiny-webgpu-int8](openai-whisper-tiny/webgpu/whisper-tiny_webgpu_int8.json) | -| [meta-llama-Llama-3.2-1B-Instruct-loha](meta-llama-Llama-3.2-1B-Instruct/olive/loha.json) | [meta-llama-Llama-3.2-1B-Instruct-loha](meta-llama-Llama-3.2-1B-Instruct/olive/loha.json) | [openai-clip-vit-base-patch32](openai-clip-vit-base-patch32/aitk/openai_clip_dml.json) | | [Qwen-Qwen2.5-1.5B-Instruct](Qwen-Qwen2.5-1.5B-Instruct/aitk/qwen2_5_trtrtx_config.json) | [Qwen-Qwen2.5-7B-Instruct](Qwen-Qwen2.5-7B-Instruct/aitk/qwen2_5_ov_npu_config.json) | [google-bert-bert-base-multilingual-cased](google-bert-bert-base-multilingual-cased/aitk/bert-base-multilingual-cased_qdq_qnn.json) | [laion-CLIP-ViT-B-32-laion2B-s34B-b79K](laion-CLIP-ViT-B-32-laion2B-s34B-b79K/aitk/laion_clip_qdq_amd.json) | [openai-whisper-tiny.en-webgpu-int8](openai-whisper-tiny.en/webgpu/whisper-tiny.en_webgpu_int8.json) | -| [meta-llama-Llama-3.2-1B-Instruct-lokr](meta-llama-Llama-3.2-1B-Instruct/olive/lokr.json) | [meta-llama-Llama-3.2-1B-Instruct-lokr](meta-llama-Llama-3.2-1B-Instruct/olive/lokr.json) | [openai-clip-vit-large-patch14](openai-clip-vit-large-patch14/aitk/openai_clip_dml.json) | | [Qwen-Qwen2.5-14B-Instruct](Qwen-Qwen2.5-14B-Instruct/aitk/qwen2_5_trtrtx.json) | [Qwen-Qwen2.5-Coder-0.5B-Instruct](Qwen-Qwen2.5-Coder-0.5B-Instruct/aitk/qwen2_5_ov_config.json) | [google-bert-bert-base-multilingual-cased](google-bert-bert-base-multilingual-cased/aitk/bert-base-multilingual-cased_qnn_gpu.json) | [meta-llama-Llama-3.1-8B-Instruct](meta-llama-Llama-3.1-8B-Instruct/aitk/llama3_1_vitis_ai_config.json) | | -| [meta-llama-Llama-3.2-1B-Instruct-mixed](meta-llama-Llama-3.2-1B-Instruct/olive/mixed.json) | [meta-llama-Llama-3.2-1B-Instruct-mixed](meta-llama-Llama-3.2-1B-Instruct/olive/mixed.json) | | | [Qwen-Qwen2.5-7B-Instruct](Qwen-Qwen2.5-7B-Instruct/aitk/qwen2_5_trtrtx.json) | [Qwen-Qwen2.5-Coder-0.5B-Instruct](Qwen-Qwen2.5-Coder-0.5B-Instruct/aitk/qwen2_5_ov_npu_config.json) | [google-vit-base-patch16-224](google-vit-base-patch16-224/QNN/config_gpu_fp32.json) | [meta-llama-Llama-3.2-1B-Instruct](meta-llama-Llama-3.2-1B-Instruct/aitk/llama3_2_vitis_ai_config.json) | | -| [meta-llama-Llama-3.2-1B-Instruct-qlora](meta-llama-Llama-3.2-1B-Instruct/olive/qlora.json) | [meta-llama-Llama-3.2-1B-Instruct-qlora](meta-llama-Llama-3.2-1B-Instruct/olive/qlora.json) | | | [Qwen-Qwen2.5-Coder-0.5B-Instruct](Qwen-Qwen2.5-Coder-0.5B-Instruct/aitk/qwen2_5_trtrtx.json) | [Qwen-Qwen2.5-Coder-1.5B-Instruct](Qwen-Qwen2.5-Coder-1.5B-Instruct/aitk/qwen2_5_ov_config.json) | [google-vit-base-patch16-224](google-vit-base-patch16-224/QNN/vit_qnn_fp32_ctx.json) | [microsoft-Phi-3-mini-128k-instruct](microsoft-Phi-3-mini-128k-instruct/aitk/phi3_vitis_ai_config.json) | | -| [meta-llama-Meta-Llama-3-8B](meta-llama-Meta-Llama-3-8B/olive/config.json) | [microsoft-Phi-3.5-mini-instruct](microsoft-Phi-3.5-mini-instruct/olive/mixed.json) | | | [Qwen-Qwen2.5-Coder-1.5B-Instruct](Qwen-Qwen2.5-Coder-1.5B-Instruct/aitk/qwen2_5_trtrtx.json) | [Qwen-Qwen2.5-Coder-1.5B-Instruct](Qwen-Qwen2.5-Coder-1.5B-Instruct/aitk/qwen2_5_ov_npu_config.json) | [google-vit-base-patch16-224](google-vit-base-patch16-224/aitk/vit-base-patch16-224_qdq_qnn.json) | [microsoft-Phi-3-mini-4k-instruct](microsoft-Phi-3-mini-4k-instruct/aitk/phi3_vitis_ai_config.json) | | -| [microsoft-deberta-base-mnli](microsoft-deberta-base-mnli/aml/deberta.json) | [microsoft-Phi-4-mini-instruct-mixed-tied](microsoft-Phi-4-mini-instruct/olive/mixed-tied.json) | | | [Qwen-Qwen2.5-Coder-14B-Instruct](Qwen-Qwen2.5-Coder-14B-Instruct/aitk/qwen2_5_trtrtx.json) | [Qwen-Qwen2.5-Coder-14B-Instruct](Qwen-Qwen2.5-Coder-14B-Instruct/aitk/qwen2_5_ov_config.json) | [google-vit-base-patch16-224](google-vit-base-patch16-224/aitk/vit-base-patch16-224_qnn_gpu.json) | [microsoft-Phi-3.5-mini-instruct](microsoft-Phi-3.5-mini-instruct/aitk/phi3_5_vitis_ai_config.json) | | -| [moonshine-tiny](usefulSensors-moonshine-tiny/cli/optimize.sh) | [microsoft-Phi-4-mini-instruct-mixed](microsoft-Phi-4-mini-instruct/olive/mixed.json) | | | [Qwen-Qwen2.5-Coder-7B-Instruct](Qwen-Qwen2.5-Coder-7B-Instruct/aitk/qwen2_5_trtrtx.json) | [Qwen-Qwen2.5-Coder-14B-Instruct](Qwen-Qwen2.5-Coder-14B-Instruct/aitk/qwen2_5_ov_npu_config.json) | [intel-bert-base-uncased-mrpc](intel-bert-base-uncased-mrpc/QNN/config_gpu_fp32.json) | [microsoft-Phi-4-mini-instruct](microsoft-Phi-4-mini-instruct/aitk/phi4_vitis_ai_config.json) | | -| [openai-whisper-base-cpu-int8](openai-whisper-base/cpu/whisper-base_cpu_int8.json) | [mistral-7b](facebook-opt-125m/cli/optimize.sh) | | | [Qwen2.5-0.5B-Instruct_Model_Builder_FP16](Qwen-Qwen2.5-0.5B-Instruct/NvTensorRtRtx/Qwen2.5-0.5B-Instruct_model_builder_fp16.json) | [Qwen-Qwen2.5-Coder-3B-Instruct](Qwen-Qwen2.5-Coder-3B-Instruct/aitk/qwen2_5_ov_config.json) | [intel-bert-base-uncased-mrpc](intel-bert-base-uncased-mrpc/aitk/bert_qdq_qnn.json) | [microsoft-Phi-4-mini-reasoning](microsoft-Phi-4-mini-reasoning/aitk/phi4_vitis_ai_config.json) | | -| [openai-whisper-base.en-cpu-int8](openai-whisper-base.en/cpu/whisper-base.en_cpu_int8.json) | [mistral-7b](mistralai-Mistral-7B-v0.1/cli/optimize.sh) | | | [Qwen2.5-14B-Instruct_Model_Builder_INT4](Qwen-Qwen2.5-14B-Instruct/NvTensorRtRtx/Qwen2.5-14B-Instruct_model_builder_int4.json) | [Qwen-Qwen2.5-Coder-3B-Instruct](Qwen-Qwen2.5-Coder-3B-Instruct/aitk/qwen2_5_ov_npu_config.json) | [intel-bert-base-uncased-mrpc](intel-bert-base-uncased-mrpc/aitk/bert_qnn_gpu.json) | [microsoft-resnet-50](microsoft-resnet-50/aitk/resnet_qdq_amd.json) | | -| [openai-whisper-large-cpu-int8](openai-whisper-large/cpu/whisper-large_cpu_int8.json) | [mistral-7b](open-llama-3b/cli/optimize.sh) | | | [Qwen2.5-7B-Instruct_Model_Builder_INT4](Qwen-Qwen2.5-7B-Instruct/NvTensorRtRtx/Qwen2.5-7B-Instruct_model_builder_int4.json) | [Qwen-Qwen2.5-Coder-7B-Instruct](Qwen-Qwen2.5-Coder-7B-Instruct/aitk/qwen2_5_ov_config.json) | [laion-CLIP-ViT-B-32-laion2B-s34B-b79K](laion-CLIP-ViT-B-32-laion2B-s34B-b79K/QNN/config_gpu_fp32.json) | [mistralai-Mistral-7B-Instruct-v0.2](mistralai-Mistral-7B-Instruct-v0.2/aitk/Mistral_7B_Instruct_v0.2_vitis_ai_config.json) | | -| [openai-whisper-large-v2-cpu-int8](openai-whisper-large-v2/cpu/whisper-large-v2_cpu_int8.json) | [moonshine-tiny](usefulSensors-moonshine-tiny/cli/optimize.sh) | | | [Qwen2.5-Coder-0.5B-Instruct_Model_Builder_FP16](Qwen-Qwen2.5-Coder-0.5B-Instruct/NvTensorRtRtx/Qwen2.5-Coder-0.5B-Instruct_model_builder_fp16.json) | [Qwen-Qwen2.5-Coder-7B-Instruct](Qwen-Qwen2.5-Coder-7B-Instruct/aitk/qwen2_5_ov_npu_config.json) | [laion-CLIP-ViT-B-32-laion2B-s34B-b79K](laion-CLIP-ViT-B-32-laion2B-s34B-b79K/aitk/laion_clip_qnn.json) | [openai-clip-vit-base-patch16](openai-clip-vit-base-patch16/aitk/openai_clip_qdq_amd.json) | | -| [openai-whisper-large-v3-cpu-int8](openai-whisper-large-v3/cpu/whisper-large-v3_cpu_int8.json) | [openai-whisper-base-cuda-int8](openai-whisper-base/cuda/whisper-base_cuda_int8.json) | | | [Qwen2.5-Coder-1.5B-Instruct_Model_Builder_FP16](Qwen-Qwen2.5-Coder-1.5B-Instruct/NvTensorRtRtx/Qwen2.5-Coder-1.5B-Instruct_model_builder_fp16.json) | [deepseek-ai-DeepSeek-R1-Distill-Llama-8B](deepseek-ai-DeepSeek-R1-Distill-Llama-8B/aitk/deepseek_ov_config.json) | [laion-CLIP-ViT-B-32-laion2B-s34B-b79K](laion-CLIP-ViT-B-32-laion2B-s34B-b79K/aitk/laion_clip_qnn_gpu.json) | [openai-clip-vit-base-patch32](openai-clip-vit-base-patch32/aitk/openai_clip_qdq_amd.json) | | -| [openai-whisper-large-v3-turbo-cpu-int8](openai-whisper-large-v3-turbo/cpu/whisper-large-v3-turbo_cpu_int8.json) | [openai-whisper-base.en-cuda-int8](openai-whisper-base.en/cuda/whisper-base.en_cuda_int8.json) | | | [Qwen2.5-Coder-14B-Instruct_Model_Builder_INT4](Qwen-Qwen2.5-Coder-14B-Instruct/NvTensorRtRtx/Qwen2.5-Coder-14B-Instruct_model_builder_int4.json) | [deepseek-ai-DeepSeek-R1-Distill-Llama-8B](deepseek-ai-DeepSeek-R1-Distill-Llama-8B/aitk/deepseek_ov_npu_config.json) | [llama3.1-8b-instruct-x-elite](meta-llama-Llama-3.1-8B-Instruct/QNN/x_elite_config.json) | [openai-clip-vit-large-patch14](openai-clip-vit-large-patch14/aitk/openai_clip_qdq_amd.json) | | -| [openai-whisper-large-v3-turbo](openai-whisper-large-v3-turbo/olive/whisper_large_v3_turbo_decoder_fp32.json) | [openai-whisper-large-cuda-int8](openai-whisper-large/cuda/whisper-large_cuda_int8.json) | | | [Qwen2.5-Coder-7B-Instruct_Model_Builder_INT4](Qwen-Qwen2.5-Coder-7B-Instruct/NvTensorRtRtx/Qwen2.5-Coder-7B-Instruct_model_builder_int4.json) | [deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B](deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B/aitk/deepseek_ov_config.json) | [llama3.1-8b-instruct-x2-elite](meta-llama-Llama-3.1-8B-Instruct/QNN/x2_elite_config.json) | [openai-whisper-large-v3-turbo-vitisai](openai-whisper-large-v3-turbo/VitisAI/run_whisper.py) | | -| [openai-whisper-large-v3-turbo](openai-whisper-large-v3-turbo/olive/whisper_large_v3_turbo_decoder_qdq.json) | [openai-whisper-large-v2-cuda-int8](openai-whisper-large-v2/cuda/whisper-large-v2_cuda_int8.json) | | | [Qwen2.5_1.5B_Instruct_Model_Builder_FP16](Qwen-Qwen2.5-1.5B-Instruct/NvTensorRtRtx/Qwen2.5-1.5B-Instruct_model_builder_fp16.json) | [deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B](deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B/aitk/deepseek_ov_gpu_config.json) | [meta-llama-Llama-3.1-8B-Instruct](meta-llama-Llama-3.1-8B-Instruct/aitk/llama3_1_qnn_config.json) | [openai-whisper-medium-vitisai](openai-whisper-medium/VitisAI/run_whisper.py) | | -| [openai-whisper-large-v3-turbo](openai-whisper-large-v3-turbo/olive/whisper_large_v3_turbo_encoder_fp32.json) | [openai-whisper-large-v3-cuda-int8](openai-whisper-large-v3/cuda/whisper-large-v3_cuda_int8.json) | | | [deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B](deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B/aitk/deepseek_trtrtx_config.json) | [deepseek-ai-DeepSeek-R1-Distill-Qwen-14B](deepseek-ai-DeepSeek-R1-Distill-Qwen-14B/aitk/deepseek_ov_config.json) | [meta-llama-Llama-3.2-1B-Instruct](meta-llama-Llama-3.2-1B-Instruct/QNN/config_gpu.json) | [openai-whisper-small-vitisai](openai-whisper-small/VitisAI/run_whisper.py) | | -| [openai-whisper-large-v3-turbo](openai-whisper-large-v3-turbo/olive/whisper_large_v3_turbo_encoder_qdq.json) | [openai-whisper-large-v3-turbo-cuda-int8](openai-whisper-large-v3-turbo/cuda/whisper-large-v3-turbo_cuda_int8.json) | | | [deepseek-ai-DeepSeek-R1-Distill-Qwen-14B](deepseek-ai-DeepSeek-R1-Distill-Qwen-14B/aitk/deepseek_trtrtx.json) | [deepseek-ai-DeepSeek-R1-Distill-Qwen-14B](deepseek-ai-DeepSeek-R1-Distill-Qwen-14B/aitk/deepseek_ov_npu_config.json) | [meta-llama-Llama-3.2-1B-Instruct](meta-llama-Llama-3.2-1B-Instruct/QNN/config_gpu_ctxbin.json) | [timm-mobilenetv3_small_100.lamb_in1k](timm-mobilenetv3_small_100.lamb_in1k/VitisAI/config.json) | | -| [openai-whisper-medium-cpu-int8](openai-whisper-medium/cpu/whisper-medium_cpu_int8.json) | [openai-whisper-medium-cuda-int8](openai-whisper-medium/cuda/whisper-medium_cuda_int8.json) | | | [deepseek-ai-DeepSeek-R1-Distill-Qwen-7B](deepseek-ai-DeepSeek-R1-Distill-Qwen-7B/aitk/deepseek_trtrtx.json) | [deepseek-ai-DeepSeek-R1-Distill-Qwen-7B](deepseek-ai-DeepSeek-R1-Distill-Qwen-7B/aitk/deepseek_ov_config.json) | [meta-llama-Llama-3.2-1B-Instruct](meta-llama-Llama-3.2-1B-Instruct/aitk/llama3_2_qnn_config.json) | | | -| [openai-whisper-medium.en-cpu-int8](openai-whisper-medium.en/cpu/whisper-medium.en_cpu_int8.json) | [openai-whisper-medium.en-cuda-int8](openai-whisper-medium.en/cuda/whisper-medium.en_cuda_int8.json) | | | [google-bert-bert-base-multilingual-cased](google-bert-bert-base-multilingual-cased/aitk/bert-base-multilingual-cased_trtrtx.json) | [deepseek-ai-DeepSeek-R1-Distill-Qwen-7B](deepseek-ai-DeepSeek-R1-Distill-Qwen-7B/aitk/deepseek_ov_npu_config.json) | [meta-llama-Llama-3.2-1B-Instruct](meta-llama-Llama-3.2-1B-Instruct/aitk/llama3_2_qnn_gpu_config.json) | | | -| [openai-whisper-small-cpu-int8](openai-whisper-small/cpu/whisper-small_cpu_int8.json) | [openai-whisper-small-cuda-int8](openai-whisper-small/cuda/whisper-small_cuda_int8.json) | | | [google-vit-base-patch16-224](google-vit-base-patch16-224/aitk/vit-base-patch16-224_trtrtx.json) | [google-bert-bert-base-multilingual-cased](google-bert-bert-base-multilingual-cased/aitk/bert-base-multilingual-cased_context_ov_static.json) | [microsoft-Phi-3-mini-128k-instruct](microsoft-Phi-3-mini-128k-instruct/QNN/config.json) | | | -| [openai-whisper-small.en-cpu-int8](openai-whisper-small.en/cpu/whisper-small.en_cpu_int8.json) | [openai-whisper-small.en-cuda-int8](openai-whisper-small.en/cuda/whisper-small.en_cuda_int8.json) | | | [intel-bert-base-uncased-mrpc](intel-bert-base-uncased-mrpc/aitk/bert_trtrtx.json) | [google-gemma-3-1b-it](google-gemma-3-1b-it/OpenVINO/gemma_3_1b_it_context_ov_gpu_config.json) | [microsoft-Phi-3-mini-128k-instruct](microsoft-Phi-3-mini-128k-instruct/aitk/phi3_qnn.json) | | | -| [openai-whisper-tiny-cpu-int8](openai-whisper-tiny/cpu/whisper-tiny_cpu_int8.json) | [openai-whisper-tiny-cuda-int8](openai-whisper-tiny/cuda/whisper-tiny_cuda_int8.json) | | | [laion-CLIP-ViT-B-32-laion2B-s34B-b79K](laion-CLIP-ViT-B-32-laion2B-s34B-b79K/aitk/laion_clip_trtrtx.json) | [google-gemma-3-1b-it](google-gemma-3-1b-it/OpenVINO/gemma_3_1b_it_context_ov_npu_config.json) | [microsoft-Phi-3-mini-4k-instruct](microsoft-Phi-3-mini-4k-instruct/QNN/config.json) | | | -| [openai-whisper-tiny.en-cpu-int8](openai-whisper-tiny.en/cpu/whisper-tiny.en_cpu_int8.json) | [openai-whisper-tiny.en-cuda-int8](openai-whisper-tiny.en/cuda/whisper-tiny.en_cuda_int8.json) | | | [meta-llama-Llama-3.1-8B-Instruct](meta-llama-Llama-3.1-8B-Instruct/aitk/llama3_1_trtrtx_config.json) | [google-vit-base-patch16-224](google-vit-base-patch16-224/OpenVINO/vit_base_patch16_224_context_ov_static.json) | [microsoft-Phi-3-mini-4k-instruct](microsoft-Phi-3-mini-4k-instruct/aitk/phi3_qnn.json) | | | -| [qwen2.5-vl-3B-Instruct](Qwen-Qwen2.5-VL-3B-Instruct/builtin/optimize.py) | [qwen2.5-vl-3B-Instruct](Qwen-Qwen2.5-VL-3B-Instruct/builtin/optimize.py) | | | [meta-llama-Llama-3.2-1B-Instruct](meta-llama-Llama-3.2-1B-Instruct/aitk/llama3_2_trtrtx_config.json) | [google-vit-base-patch16-224](google-vit-base-patch16-224/aitk/vit_base_patch16_224_context_ov_static.json) | [microsoft-Phi-3.5-mini-instruct](microsoft-Phi-3.5-mini-instruct/QNN/config.json) | | | -| [qwen3.5-0.8B-Instruct](Qwen-Qwen3.5-0.8B/builtin/optimize.py) | [qwen3.5-0.8B-Instruct](Qwen-Qwen3.5-0.8B/builtin/optimize.py) | | | [microsoft-Phi-3-mini-128k-instruct](microsoft-Phi-3-mini-128k-instruct/aitk/phi3_trtrtx.json) | [intel-bert-base-uncased-mrpc (ov)](intel-bert-base-uncased-mrpc/aitk/bert_ov.json) | [microsoft-Phi-3.5-mini-instruct](microsoft-Phi-3.5-mini-instruct/QNN/config_fp16.json) | | | -| [qwen3.5-2B](Qwen-Qwen3.5-2B/builtin/optimize.py) | [qwen3.5-27B](Qwen-Qwen3.5-27B/builtin/optimize.py) | | | [microsoft-Phi-3-mini-4k-instruct](microsoft-Phi-3-mini-4k-instruct/aitk/phi3_trtrtx.json) | [laion-CLIP-ViT-B-32-laion2B-s34B-b79K](laion-CLIP-ViT-B-32-laion2B-s34B-b79K/aitk/laion_clip_ov.json) | [microsoft-Phi-3.5-mini-instruct](microsoft-Phi-3.5-mini-instruct/QNN/config_gpu.json) | | | -| [qwen3.5-4B](Qwen-Qwen3.5-4B/builtin/optimize.py) | [qwen3.5-2B](Qwen-Qwen3.5-2B/builtin/optimize.py) | | | [microsoft-Phi-3.5-mini-instruct](microsoft-Phi-3.5-mini-instruct/aitk/phi3_5_trtrtx_config.json) | [meta-llama-Llama-3.1-8B-Instruct](meta-llama-Llama-3.1-8B-Instruct/aitk/llama3_1_ov_config.json) | [microsoft-Phi-3.5-mini-instruct](microsoft-Phi-3.5-mini-instruct/QNN/config_gpu_ctxbin.json) | | | -| [qwen3.5-9B](Qwen-Qwen3.5-9B/builtin/optimize.py) | [qwen3.5-4B](Qwen-Qwen3.5-4B/builtin/optimize.py) | | | [microsoft-Phi-4-mini-instruct_nvmo_ptq_mixed_precision_awq_lite](microsoft-Phi-4-mini-instruct/NvTensorRtRtx/microsoft-Phi-4-mini-instruct_nvmo_ptq_mixed_precision_awq_lite.json) | [meta-llama-Llama-3.1-8B-Instruct](meta-llama-Llama-3.1-8B-Instruct/aitk/llama3_1_ov_gpu_config.json) | [microsoft-Phi-3.5-mini-instruct](microsoft-Phi-3.5-mini-instruct/aitk/phi3_5_qnn_config.json) | | | -| [qwen3vl-2B-Instruct](Qwen-Qwen3-VL-2B-Instruct/builtin/optimize.py) | [qwen3.5-9B](Qwen-Qwen3.5-9B/builtin/optimize.py) | | | [microsoft-Phi-4](microsoft-Phi-4/aitk/phi4_trtrtx.json) | [meta-llama-Llama-3.2-1B-Instruct](meta-llama-Llama-3.2-1B-Instruct/aitk/llama3_2_ov_config.json) | [microsoft-Phi-3.5-mini-instruct](microsoft-Phi-3.5-mini-instruct/aitk/phi3_5_qnn_gpu_config.json) | | | -| [qwen3vl-4B-Instruct](Qwen-Qwen3-VL-4B-Instruct/builtin/optimize.py) | [qwen3vl-2B-Instruct](Qwen-Qwen3-VL-2B-Instruct/builtin/optimize.py) | | | [microsoft-resnet-50](microsoft-resnet-50/aitk/resnet_trtrtx.json) | [meta-llama-Llama-3.2-1B-Instruct](meta-llama-Llama-3.2-1B-Instruct/aitk/llama3_2_ov_gpu_config.json) | [microsoft-Phi-4-mini-instruct](microsoft-Phi-4-mini-instruct/QNN/phi4_mini_qnn_docker.json) | | | -| [qwen3vl-8B-Instruct](Qwen-Qwen3-VL-8B-Instruct/builtin/optimize.py) | [qwen3vl-4B-Instruct](Qwen-Qwen3-VL-4B-Instruct/builtin/optimize.py) | | | [mistralai-Mistral-7B-Instruct-v0.2](mistralai-Mistral-7B-Instruct-v0.2/aitk/Mistral_7B_Instruct_v0.2_trtrtx.json) | [microsoft-Phi-3-mini-128k-instruct](microsoft-Phi-3-mini-128k-instruct/aitk/phi3_ov_config.json) | [microsoft-Phi-4-mini-instruct](microsoft-Phi-4-mini-instruct/aitk/phi4_qnn.json) | | | -| [sshleifer-tiny-gpt2-sparsegpt](sshleifer-tiny-gpt2/olive/sparsegpt.json) | [qwen3vl-8B-Instruct](Qwen-Qwen3-VL-8B-Instruct/builtin/optimize.py) | | | [openai-clip-vit-base-patch16](openai-clip-vit-base-patch16/aitk/openai_clip_trtrtx.json) | [microsoft-Phi-3-mini-128k-instruct](microsoft-Phi-3-mini-128k-instruct/aitk/phi3_ov_npu_config.json) | [microsoft-Phi-4-reasoning](microsoft-Phi-4-reasoning/QAIRT/htp_sc8380xp.json) | | | -| [stable-diffusion-v1-4-safety-checker](compvis-stable-diffusion-v1-4/olive/config_safety_checker.json) | [sshleifer-tiny-gpt2-sparsegpt](sshleifer-tiny-gpt2/olive/sparsegpt.json) | | | [openai-clip-vit-base-patch32](openai-clip-vit-base-patch32/aitk/openai_clip_trtrtx.json) | [microsoft-Phi-3-mini-4k-instruct](microsoft-Phi-3-mini-4k-instruct/aitk/phi3_ov_config.json) | [microsoft-Phi-4-reasoning](microsoft-Phi-4-reasoning/QAIRT/htp_sc8480xp.json) | | | -| [stable-diffusion-v1-4-text-encoder](compvis-stable-diffusion-v1-4/olive/config_text_encoder.json) | [stable-diffusion-v1-4-safety-checker](compvis-stable-diffusion-v1-4/olive/config_safety_checker.json) | | | [openai-clip-vit-large-patch14](openai-clip-vit-large-patch14/aitk/openai_clip_trtrtx.json) | [microsoft-Phi-3-mini-4k-instruct](microsoft-Phi-3-mini-4k-instruct/aitk/phi3_ov_npu_config.json) | [microsoft-Phi-4-reasoning](microsoft-Phi-4-reasoning/QNN/config.json) | | | -| [stable-diffusion-v1-4-unet](compvis-stable-diffusion-v1-4/olive/config_unet.json) | [stable-diffusion-v1-4-text-encoder](compvis-stable-diffusion-v1-4/olive/config_text_encoder.json) | | | [phi-4_Model_Builder_INT4](microsoft-Phi-4/NvTensorRtRtx/phi-4_model_builder_int4.json) | [microsoft-Phi-3.5-mini-instruct](microsoft-Phi-3.5-mini-instruct/aitk/phi3_5_ov_config.json) | [microsoft-resnet-50](microsoft-resnet-50/aitk/resnet_qdq_qnn.json) | | | -| [stable-diffusion-v1-4-vae-decoder](compvis-stable-diffusion-v1-4/olive/config_vae_decoder.json) | [stable-diffusion-v1-4-unet](compvis-stable-diffusion-v1-4/olive/config_unet.json) | | | | [microsoft-Phi-3.5-mini-instruct](microsoft-Phi-3.5-mini-instruct/aitk/phi3_5_ov_gpu_config.json) | [microsoft-resnet-50](microsoft-resnet-50/aitk/resnet_qnn_gpu.json) | | | -| [stable-diffusion-v1-4-vae-encoder](compvis-stable-diffusion-v1-4/olive/config_vae_encoder.json) | [stable-diffusion-v1-4-vae-decoder](compvis-stable-diffusion-v1-4/olive/config_vae_decoder.json) | | | | [microsoft-Phi-4-mini-instruct](microsoft-Phi-4-mini-instruct/OpenVINO/Phi-4-mini-instruct-gpu-context-dy.json) | [microsoft-table-transformer-detection](microsoft-table-transformer-detection/QNN/ttd_config.json) | | | -| [stable-diffusion-v1-5](sd-legacy-stable-diffusion-v1-5/olive/optimize.sh) | [stable-diffusion-v1-4-vae-encoder](compvis-stable-diffusion-v1-4/olive/config_vae_encoder.json) | | | | [microsoft-Phi-4-mini-instruct](microsoft-Phi-4-mini-instruct/OpenVINO/Phi_4_mini_instruct_context_ov_dynamic_sym_gs128_bkp_int8_sym.json) | [openai-clip-vit-base-patch16](openai-clip-vit-base-patch16/QNN/config_gpu_fp32.json) | | | -| [stable-diffusion-xl-base-1.0](stabilityai-stable-diffusion-xl-base-1.0/olive/stable_diffusion_xl.py) | [stable-diffusion-v1-5](sd-legacy-stable-diffusion-v1-5/olive/optimize.sh) | | | | [microsoft-Phi-4-mini-instruct](microsoft-Phi-4-mini-instruct/aitk/phi4_ov_config.json) | [openai-clip-vit-base-patch16](openai-clip-vit-base-patch16/aitk/openai_clip_qnn.json) | | | -| [timm-mobilenetv3_small_100.lamb_in1k](timm-mobilenetv3_small_100.lamb_in1k/olive/config.json) | [stable-diffusion-xl-base-1.0](stabilityai-stable-diffusion-xl-base-1.0/olive/stable_diffusion_xl.py) | | | | [microsoft-Phi-4-mini-instruct](microsoft-Phi-4-mini-instruct/aitk/phi4_ov_npu_config.json) | [openai-clip-vit-base-patch16](openai-clip-vit-base-patch16/aitk/openai_clip_qnn_gpu.json) | | | -| | | | | | [microsoft-Phi-4-mini-reasoning](microsoft-Phi-4-mini-reasoning/OpenVINO/Phi-4-mini-reasoning_context_ov_dynamic_sym_gs128_bkp_int8_sym.json) | [openai-clip-vit-base-patch32](openai-clip-vit-base-patch32/QNN/config_gpu_fp32.json) | | | -| | | | | | [microsoft-Phi-4-mini-reasoning](microsoft-Phi-4-mini-reasoning/OpenVINO/Phi_4_mini_instruct_context_ov_dynamic_sym_gs128_bkp_int8_sym.json) | [openai-clip-vit-base-patch32](openai-clip-vit-base-patch32/aitk/openai_clip_qnn.json) | | | -| | | | | | [microsoft-Phi-4-mini-reasoning](microsoft-Phi-4-mini-reasoning/aitk/phi4_ov_config.json) | [openai-clip-vit-base-patch32](openai-clip-vit-base-patch32/aitk/openai_clip_qnn_gpu.json) | | | -| | | | | | [microsoft-Phi-4-mini-reasoning](microsoft-Phi-4-mini-reasoning/aitk/phi4_ov_gpu_config.json) | [openai-clip-vit-large-patch14](openai-clip-vit-large-patch14/aitk/openai_clip_qnn.json) | | | -| | | | | | [microsoft-Phi-4-reasoning-plus](microsoft-Phi-4-reasoning-plus/OpenVINO/Phi-4-Phi-4-reasoning-plus_context_ov_dynamic_sym_gs128_bkp_int8_sym.json) | [openai-whisper-large-v3-turbo](openai-whisper-large-v3-turbo/aitk/qnn_workflow.json) | | | -| | | | | | [microsoft-Phi-4-reasoning-plus](microsoft-Phi-4-reasoning-plus/aitk/phi4_ov_config.json) | [openai-whisper-large-v3-turbo](openai-whisper-large-v3-turbo/olive/whisper_large_v3_turbo_decoder_fp32.json) | | | -| | | | | | [microsoft-Phi-4-reasoning](microsoft-Phi-4-reasoning/OpenVINO/Phi-4-reasoning_context_ov_dynamic_sym_gs128_bkp_int8_sym.json) | [openai-whisper-large-v3-turbo](openai-whisper-large-v3-turbo/olive/whisper_large_v3_turbo_decoder_qdq.json) | | | -| | | | | | [microsoft-Phi-4-reasoning](microsoft-Phi-4-reasoning/aitk/phi4_ov_config.json) | [openai-whisper-large-v3-turbo](openai-whisper-large-v3-turbo/olive/whisper_large_v3_turbo_decoder_qdq_ctx.json) | | | -| | | | | | [microsoft-Phi-4](microsoft-Phi-4/OpenVINO/phi_4_gpu_context_dy.json) | [openai-whisper-large-v3-turbo](openai-whisper-large-v3-turbo/olive/whisper_large_v3_turbo_encoder_fp32.json) | | | -| | | | | | [microsoft-Phi-4](microsoft-Phi-4/aitk/phi4_ov_config.json) | [openai-whisper-large-v3-turbo](openai-whisper-large-v3-turbo/olive/whisper_large_v3_turbo_encoder_qdq.json) | | | -| | | | | | [microsoft-resnet-50](microsoft-resnet-50/aitk/resnet_context_ov_static.json) | [openai-whisper-large-v3-turbo](openai-whisper-large-v3-turbo/olive/whisper_large_v3_turbo_encoder_qdq_ctx.json) | | | -| | | | | | [mistralai-Mistral-7B-Instruct-v0.2](mistralai-Mistral-7B-Instruct-v0.2/aitk/Mistral_7B_Instruct_v0.2_gpu_context_ov_dy.json) | [qwen2.5-7b-instruct](Qwen-Qwen2.5-7B-Instruct/QNN/config.json) | | | -| | | | | | [mistralai-Mistral-7B-Instruct-v0.2](mistralai-Mistral-7B-Instruct-v0.2/aitk/Mistral_7B_Instruct_v0.2_npu_context_ov_dy.json) | [sam-vit-base](sam-vit-base/QNN/sam_mask_box_decoder_qnn_fp16.json) | | | -| | | | | | [mistralai-Mistral-7B-Instruct-v0.3](mistralai-Mistral-7B-Instruct-v0.3/aitk/mistral-7b-instruct-v0.3-ov.json) | [sam-vit-base](sam-vit-base/QNN/sam_mask_decoder_qnn_fp16_ctx.json) | | | -| | | | | | [openai-clip-vit-base-patch16](openai-clip-vit-base-patch16/aitk/openai_clip_ov.json) | [sam-vit-base](sam-vit-base/QNN/sam_mask_point_decoder_qnn_fp16.json) | | | -| | | | | | [openai-clip-vit-base-patch32](openai-clip-vit-base-patch32/aitk/openai_clip_ov.json) | [sam-vit-base](sam-vit-base/QNN/sam_vision_encoder_qnn_w8a8_ctx.json) | | | -| | | | | | [openai-clip-vit-large-patch14](openai-clip-vit-large-patch14/aitk/openai_clip_ov.json) | [sam-vit-base](sam-vit-base/aitk/sam_qnn_workflow.json) | | | -| | | | | | [openai-whisper-large-v3-turbo](openai-whisper-large-v3-turbo/OpenVINO/whisper_large_v3_turbo_default_ov_npu.json) | [sam2.1-hiera-small](sam2.1-hiera-small/QNN/sam21_mask_decoder_qnn_ctx.json) | | | -| | | | | | [sd-legacy-stable-diffusion-v1-5](sd-legacy-stable-diffusion-v1-5/aitk/sd_ov_npu_workflow.json) | [sam2.1-hiera-small](sam2.1-hiera-small/QNN/sam21_vision_encoder_qnn_ctx.json) | | | -| | | | | | [sd-legacy-stable-diffusion-v1-5](sd-legacy-stable-diffusion-v1-5/aitk/sd_ov_workflow.json) | [sam2.1-hiera-small](sam2.1-hiera-small/aitk/sam2_qnn_workflow.json) | | | -| | | | | | [sd2-community-stable-diffusion-2-1](sd2-community-stable-diffusion-2-1/aitk/sd_ov_npu_workflow.json) | [sd-legacy-stable-diffusion-v1-5](sd-legacy-stable-diffusion-v1-5/aitk/sd_qnn_workflow.json) | | | -| | | | | | [sd2-community-stable-diffusion-2-1](sd2-community-stable-diffusion-2-1/aitk/sd_ov_workflow.json) | [sd2-community-stable-diffusion-2-1](sd2-community-stable-diffusion-2-1/aitk/sd_qnn_workflow.json) | | | -| | | | | | [stable-diffusion-v1-4-safety-checker](compvis-stable-diffusion-v1-4/olive/config_safety_checker.json) | [stable-diffusion-v1-4-safety-checker](compvis-stable-diffusion-v1-4/olive/config_safety_checker.json) | | | -| | | | | | [stable-diffusion-v1-4-text-encoder](compvis-stable-diffusion-v1-4/olive/config_text_encoder.json) | [stable-diffusion-v1-4-text-encoder](compvis-stable-diffusion-v1-4/olive/config_text_encoder.json) | | | -| | | | | | [stable-diffusion-v1-4-unet](compvis-stable-diffusion-v1-4/olive/config_unet.json) | [stable-diffusion-v1-4-unet](compvis-stable-diffusion-v1-4/olive/config_unet.json) | | | -| | | | | | [stable-diffusion-v1-4-vae-decoder](compvis-stable-diffusion-v1-4/olive/config_vae_decoder.json) | [stable-diffusion-v1-4-vae-decoder](compvis-stable-diffusion-v1-4/olive/config_vae_decoder.json) | | | -| | | | | | [stable-diffusion-v1-4-vae-encoder](compvis-stable-diffusion-v1-4/olive/config_vae_encoder.json) | [stable-diffusion-v1-4-vae-encoder](compvis-stable-diffusion-v1-4/olive/config_vae_encoder.json) | | | -| | | | | | [stable-diffusion-v1-5](sd-legacy-stable-diffusion-v1-5/olive/optimize.sh) | [stable-diffusion-v1-5](sd-legacy-stable-diffusion-v1-5/olive/optimize.sh) | | | -| | | | | | | [stable-diffusion-xl-base-1.0](stabilityai-stable-diffusion-xl-base-1.0/olive/stable_diffusion_xl.py) | | | -| | | | | | | [timm-mobilenetv3_small_100.lamb_in1k](timm-mobilenetv3_small_100.lamb_in1k/QNN/mobilenet_qnn_ep.json) | | | +| [alibaba-nlp-gte-large-en-v1.5](alibaba-nlp-gte-large-en-v1.5/quant/config.json) | [Qwen-Qwen2.5-1.5B-Instruct-mixed](Qwen-Qwen2.5-1.5B-Instruct/olive/mixed.json) | [Qwen-Qwen2.5-1.5B-Instruct](Qwen-Qwen2.5-1.5B-Instruct/aitk/qwen2_5_dml_config.json) | [google-bert-bert-base-multilingual-cased](google-bert-bert-base-multilingual-cased/aitk/bert-base-multilingual-cased_migraphx.json) | [DeepSeek-R1-Distill-Llama-8B_Model_Builder_INT4](deepseek-ai-DeepSeek-R1-Distill-Llama-8B/NvTensorRtRtx/DeepSeek-R1-Distill-Llama-8B_model_builder_int4.json) | [OFA-Sys-chinese-clip-vit-base-patch16](OFA-Sys-chinese-clip-vit-base-patch16/aitk/openai_clip_ov.json) | [Qwen-Qwen2.5-1.5B-Instruct](Qwen-Qwen2.5-1.5B-Instruct/QNN/config.json) | [Qwen-Qwen2.5-0.5B-Instruct](Qwen-Qwen2.5-0.5B-Instruct/aitk/qwen2_5_vitis_ai_config.json) | [ministral_3_3b](mistralai-Ministral-3-3B-Instruct-2512/builtin/optimize.py) | +| [facebook-opt-125m-splicegpt](facebook-opt-125m/olive/slicegpt.json) | [deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B](deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B/olive/mixed.json) | [deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B](deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B/aitk/deepseek_dml_config.json) | [google-vit-base-patch16-224](google-vit-base-patch16-224/aitk/vit-base-patch16-224_migraphx.json) | [DeepSeek-R1-Distill-Qwen-1.5B_Model_Builder_FP16](deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B/NvTensorRtRtx/DeepSeek-R1-Distill-Qwen-1.5B_model_builder_fp16.json) | [Qwen-Qwen2.5-0.5B-Instruct](Qwen-Qwen2.5-0.5B-Instruct/aitk/qwen2_5_ov_config.json) | [Qwen-Qwen2.5-1.5B-Instruct](Qwen-Qwen2.5-1.5B-Instruct/QNN/config_gpu.json) | [Qwen-Qwen2.5-1.5B-Instruct](Qwen-Qwen2.5-1.5B-Instruct/aitk/qwen2_5_vitis_ai_config.json) | [openai-whisper-base-webgpu-int8](openai-whisper-base/webgpu/whisper-base_webgpu_int8.json) | +| [flux2-klein-4b-text-encoder](Flux.2-Klein-4B/RyzenAI/config_text_encoder.json) | [facebook-opt-125m-splicegpt](facebook-opt-125m/olive/slicegpt.json) | [google-bert-bert-base-multilingual-cased](google-bert-bert-base-multilingual-cased/aitk/bert-base-multilingual-cased_dml.json) | [intel-bert-base-uncased-mrpc](intel-bert-base-uncased-mrpc/aitk/bert_migraphx.json) | [DeepSeek-R1-Distill-Qwen-14B_NVMO_INT4_AWQ](deepseek-ai-DeepSeek-R1-Distill-Qwen-14B/NvTensorRtRtx/DeepSeek-R1-Distill-Qwen-14B_nvmo_int4_awq.json) | [Qwen-Qwen2.5-0.5B-Instruct](Qwen-Qwen2.5-0.5B-Instruct/aitk/qwen2_5_ov_npu_config.json) | [Qwen-Qwen2.5-1.5B-Instruct](Qwen-Qwen2.5-1.5B-Instruct/QNN/config_gpu_ctxbin.json) | [Qwen-Qwen2.5-7B-Instruct](Qwen-Qwen2.5-7B-Instruct/aitk/qwen2_5_vitis_ai_config.json) | [openai-whisper-base.en-webgpu-int8](openai-whisper-base.en/webgpu/whisper-base.en_webgpu_int8.json) | +| [gemma-3-1b-it_model_builder_cpu_FP32](google-gemma/olive/gemma-3-1b-it_model_builder_cpu_fp32.json) | [gemma4-e2b-fp16-cuda](google-gemma-4-E2B-it/cuda/fp16/config.json) | [google-vit-base-patch16-224](google-vit-base-patch16-224/aitk/vit-base-patch16-224_dml.json) | [laion-CLIP-ViT-B-32-laion2B-s34B-b79K](laion-CLIP-ViT-B-32-laion2B-s34B-b79K/aitk/laion_clip_migraphx.json) | [DeepSeek-R1-Distill-Qwen-7B_NVMO_INT4_RTN](deepseek-ai-DeepSeek-R1-Distill-Qwen-7B/NvTensorRtRtx/DeepSeek-R1-Distill-Qwen-7B_nvmo_int4_rtn.json) | [Qwen-Qwen2.5-0.5B](Qwen-Qwen2.5-0.5B/aitk/qwen2_5_ov_config.json) | [Qwen-Qwen2.5-1.5B-Instruct](Qwen-Qwen2.5-1.5B-Instruct/aitk/qwen2_5_qnn_config.json) | [Qwen-Qwen2.5-Coder-0.5B-Instruct](Qwen-Qwen2.5-Coder-0.5B-Instruct/aitk/qwen2_5_vitis_ai_config.json) | [openai-whisper-large-v2-webgpu-int8](openai-whisper-large-v2/webgpu/whisper-large-v2_webgpu_int8.json) | +| [gemma4-e2b-fp32-cpu](google-gemma-4-E2B-it/cpu/fp32/config.json) | [gemma4-e2b-int4-kquant-cuda](google-gemma-4-E2B-it/cuda/int4/config.json) | [intel-bert-base-uncased-mrpc](intel-bert-base-uncased-mrpc/aitk/bert_dml.json) | [microsoft-resnet-50](microsoft-resnet-50/aitk/resnet_migraphx.json) | [Llama-3.2-1B-Instruct_Model_Builder_FP16](meta-llama-Llama-3.2-1B-Instruct/NvTensorRtRtx/Llama-3.2-1B-Instruct_model_builder_fp16.json) | [Qwen-Qwen2.5-1.5B-Instruct](Qwen-Qwen2.5-1.5B-Instruct/aitk/qwen2_5_ov_config.json) | [Qwen-Qwen2.5-1.5B-Instruct](Qwen-Qwen2.5-1.5B-Instruct/aitk/qwen2_5_qnn_gpu_config.json) | [Qwen-Qwen2.5-Coder-1.5B-Instruct](Qwen-Qwen2.5-Coder-1.5B-Instruct/aitk/qwen2_5_vitis_ai_config.json) | [openai-whisper-large-v3-turbo-webgpu-int8](openai-whisper-large-v3-turbo/webgpu/whisper-large-v3-turbo_webgpu_int8.json) | +| [gemma4-e2b-int4-kquant-cpu](google-gemma-4-E2B-it/cpu/int4/config.json) | [google-gemma](google-gemma/olive/README.md) | [laion-CLIP-ViT-B-32-laion2B-s34B-b79K](laion-CLIP-ViT-B-32-laion2B-s34B-b79K/aitk/laion_clip_dml.json) | [openai-clip-vit-base-patch16](openai-clip-vit-base-patch16/aitk/openai_clip_migraphx.json) | [Llama3.1-8B-Instruct_Model_Builder_INT4](meta-llama-Llama-3.1-8B-Instruct/NvTensorRtRtx/Llama-3.1-8B-Instruct_model_builder_int4.json) | [Qwen-Qwen2.5-1.5B-Instruct](Qwen-Qwen2.5-1.5B-Instruct/aitk/qwen2_5_ov_gpu_config.json) | [Qwen-Qwen2.5-7B-Instruct](Qwen-Qwen2.5-7B-Instruct/aitk/qwen2_5_qnn_config.json) | [Qwen-Qwen2.5-Coder-7B-Instruct](Qwen-Qwen2.5-Coder-7B-Instruct/aitk/qwen2_5_vitis_ai_config.json) | [openai-whisper-large-v3-webgpu-int8](openai-whisper-large-v3/webgpu/whisper-large-v3_webgpu_int8.json) | +| [google-gemma](google-gemma/olive/README.md) | [gpt-oss-20b](gpt-oss-20b/int4_cuda_int4_qmoe/gpt-oss-20b.sh) | [meta-llama-Llama-3.1-8B-Instruct](meta-llama-Llama-3.1-8B-Instruct/aitk/llama3_1_dml_config.json) | [openai-clip-vit-base-patch32](openai-clip-vit-base-patch32/aitk/openai_clip_migraphx.json) | [Mistral-7B-Instruct-v0.2_Model_Builder_INT4](mistralai-Mistral-7B-Instruct-v0.2/NvTensorRtRtx/Mistral-7B-Instruct-v0.2_model_builder_int4.json) | [Qwen-Qwen2.5-14B-Instruct](Qwen-Qwen2.5-14B-Instruct/aitk/qwen2_5_ov_config.json) | [deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B](deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B/QNN/config_gpu.json) | [deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B](deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B/aitk/deepseek_vitis_ai_config.json) | [openai-whisper-large-webgpu-int8](openai-whisper-large/webgpu/whisper-large_webgpu_int8.json) | +| [gpt-oss-20b](gpt-oss-20b/int4_cuda_int4_qmoe/gpt-oss-20b.sh) | [intel-bert-base-uncased-mrpc-inc-quant](intel-bert-base-uncased-mrpc/oci/gpu/cuda/inc_quant.json) | [meta-llama-Llama-3.2-1B-Instruct](meta-llama-Llama-3.2-1B-Instruct/aitk/llama3_2_dml_config.json) | [openai-clip-vit-large-patch14](openai-clip-vit-large-patch14/aitk/openai_clip_migraphx.json) | [Phi-3-mini-128k-instruct_NVMO_INT4_RTN](microsoft-Phi-3-mini-128k-instruct/NvTensorRtRtx/Phi-3-mini-128k-instruct_nvmo_int4_rtn.json) | [Qwen-Qwen2.5-14B-Instruct](Qwen-Qwen2.5-14B-Instruct/aitk/qwen2_5_ov_npu_config.json) | [deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B](deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B/QNN/config_gpu_ctxbin.json) | [deepseek-ai-DeepSeek-R1-Distill-Qwen-7B](deepseek-ai-DeepSeek-R1-Distill-Qwen-7B/aitk/deepseek_vitis_ai_config.json) | [openai-whisper-medium-webgpu-int8](openai-whisper-medium/webgpu/whisper-medium_webgpu_int8.json) | +| [intel-bert-base-uncased-mrpc-inc-smooth-quant](intel-bert-base-uncased-mrpc/oci/cpu/inc_smooth_quant.json) | [intel-bert-base-uncased-mrpc-ptq](intel-bert-base-uncased-mrpc/oci/gpu/cuda/ptq.json) | [microsoft-Phi-3.5-mini-instruct](microsoft-Phi-3.5-mini-instruct/aitk/phi3_5_dml_config.json) | | [Phi-3-mini-4k-instruct_Model_Builder_INT4](microsoft-Phi-3-mini-4k-instruct/NvTensorRtRtx/Phi-3-mini-4k-instruct_model_builder_int4.json) | [Qwen-Qwen2.5-3B-Instruct](Qwen-Qwen2.5-3B-Instruct/aitk/qwen2_5_ov_config.json) | [deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B](deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B/aitk/deepseek_qnn_config.json) | [google-bert-bert-base-multilingual-cased](google-bert-bert-base-multilingual-cased/aitk/bert-base-multilingual-cased_qdq_amd.json) | [openai-whisper-medium.en-webgpu-int8](openai-whisper-medium.en/webgpu/whisper-medium.en_webgpu_int8.json) | +| [intel-bert-base-uncased-mrpc-ptq](intel-bert-base-uncased-mrpc/oci/cpu/ptq.json) | [meta-llama-Llama-3.2-1B-Instruct-dora](meta-llama-Llama-3.2-1B-Instruct/olive/dora.json) | [microsoft-resnet-50](microsoft-resnet-50/aitk/resnet_dml.json) | | [Phi3.5_Mini_Instruct_Model_Builder_INT4](microsoft-Phi-3.5-mini-instruct/NvTensorRtRtx/Phi-3.5-mini-instruct_model_builder_int4.json) | [Qwen-Qwen2.5-3B-Instruct](Qwen-Qwen2.5-3B-Instruct/aitk/qwen2_5_ov_npu_config.json) | [deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B](deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B/aitk/deepseek_qnn_gpu_config.json) | [google-vit-base-patch16-224](google-vit-base-patch16-224/aitk/vit-base-patch16-224_qdq_amd.json) | [openai-whisper-small-webgpu-int8](openai-whisper-small/webgpu/whisper-small_webgpu_int8.json) | +| [meta-llama-Llama-3.2-1B-Instruct-dora](meta-llama-Llama-3.2-1B-Instruct/olive/dora.json) | [meta-llama-Llama-3.2-1B-Instruct-hqq](meta-llama-Llama-3.2-1B-Instruct/olive/hqq.json) | [openai-clip-vit-base-patch16](openai-clip-vit-base-patch16/aitk/openai_clip_dml.json) | | [Qwen-Qwen2.5-0.5B-Instruct](Qwen-Qwen2.5-0.5B-Instruct/aitk/qwen2_5_trtrtx.json) | [Qwen-Qwen2.5-7B-Instruct](Qwen-Qwen2.5-7B-Instruct/aitk/qwen2_5_ov_config.json) | [google-bert-bert-base-multilingual-cased](google-bert-bert-base-multilingual-cased/QNN/config_gpu_fp32.json) | [intel-bert-base-uncased-mrpc (AMD)](intel-bert-base-uncased-mrpc/aitk/bert_qdq_amd.json) | [openai-whisper-small.en-webgpu-int8](openai-whisper-small.en/webgpu/whisper-small.en_webgpu_int8.json) | +| [meta-llama-Llama-3.2-1B-Instruct-hqq](meta-llama-Llama-3.2-1B-Instruct/olive/hqq.json) | [meta-llama-Llama-3.2-1B-Instruct-lmeval-onnx](meta-llama-Llama-3.2-1B-Instruct/olive/lmeval_onnx.json) | [openai-clip-vit-base-patch32](openai-clip-vit-base-patch32/aitk/openai_clip_dml.json) | | [Qwen-Qwen2.5-1.5B-Instruct](Qwen-Qwen2.5-1.5B-Instruct/aitk/qwen2_5_trtrtx_config.json) | [Qwen-Qwen2.5-7B-Instruct](Qwen-Qwen2.5-7B-Instruct/aitk/qwen2_5_ov_npu_config.json) | [google-bert-bert-base-multilingual-cased](google-bert-bert-base-multilingual-cased/aitk/bert-base-multilingual-cased_qdq_qnn.json) | [laion-CLIP-ViT-B-32-laion2B-s34B-b79K](laion-CLIP-ViT-B-32-laion2B-s34B-b79K/aitk/laion_clip_qdq_amd.json) | [openai-whisper-tiny-webgpu-int8](openai-whisper-tiny/webgpu/whisper-tiny_webgpu_int8.json) | +| [meta-llama-Llama-3.2-1B-Instruct-lmeval-onnx](meta-llama-Llama-3.2-1B-Instruct/olive/lmeval_onnx.json) | [meta-llama-Llama-3.2-1B-Instruct-lmeval](meta-llama-Llama-3.2-1B-Instruct/olive/lmeval.json) | [openai-clip-vit-large-patch14](openai-clip-vit-large-patch14/aitk/openai_clip_dml.json) | | [Qwen-Qwen2.5-14B-Instruct](Qwen-Qwen2.5-14B-Instruct/aitk/qwen2_5_trtrtx.json) | [Qwen-Qwen2.5-Coder-0.5B-Instruct](Qwen-Qwen2.5-Coder-0.5B-Instruct/aitk/qwen2_5_ov_config.json) | [google-bert-bert-base-multilingual-cased](google-bert-bert-base-multilingual-cased/aitk/bert-base-multilingual-cased_qnn_gpu.json) | [meta-llama-Llama-3.1-8B-Instruct](meta-llama-Llama-3.1-8B-Instruct/aitk/llama3_1_vitis_ai_config.json) | [openai-whisper-tiny.en-webgpu-int8](openai-whisper-tiny.en/webgpu/whisper-tiny.en_webgpu_int8.json) | +| [meta-llama-Llama-3.2-1B-Instruct-lmeval](meta-llama-Llama-3.2-1B-Instruct/olive/lmeval.json) | [meta-llama-Llama-3.2-1B-Instruct-loha](meta-llama-Llama-3.2-1B-Instruct/olive/loha.json) | | | [Qwen-Qwen2.5-7B-Instruct](Qwen-Qwen2.5-7B-Instruct/aitk/qwen2_5_trtrtx.json) | [Qwen-Qwen2.5-Coder-0.5B-Instruct](Qwen-Qwen2.5-Coder-0.5B-Instruct/aitk/qwen2_5_ov_npu_config.json) | [google-vit-base-patch16-224](google-vit-base-patch16-224/QNN/config_gpu_fp32.json) | [meta-llama-Llama-3.2-1B-Instruct](meta-llama-Llama-3.2-1B-Instruct/aitk/llama3_2_vitis_ai_config.json) | | +| [meta-llama-Llama-3.2-1B-Instruct-loha](meta-llama-Llama-3.2-1B-Instruct/olive/loha.json) | [meta-llama-Llama-3.2-1B-Instruct-lokr](meta-llama-Llama-3.2-1B-Instruct/olive/lokr.json) | | | [Qwen-Qwen2.5-Coder-0.5B-Instruct](Qwen-Qwen2.5-Coder-0.5B-Instruct/aitk/qwen2_5_trtrtx.json) | [Qwen-Qwen2.5-Coder-1.5B-Instruct](Qwen-Qwen2.5-Coder-1.5B-Instruct/aitk/qwen2_5_ov_config.json) | [google-vit-base-patch16-224](google-vit-base-patch16-224/QNN/vit_qnn_fp32_ctx.json) | [microsoft-Phi-3-mini-128k-instruct](microsoft-Phi-3-mini-128k-instruct/aitk/phi3_vitis_ai_config.json) | | +| [meta-llama-Llama-3.2-1B-Instruct-lokr](meta-llama-Llama-3.2-1B-Instruct/olive/lokr.json) | [meta-llama-Llama-3.2-1B-Instruct-mixed](meta-llama-Llama-3.2-1B-Instruct/olive/mixed.json) | | | [Qwen-Qwen2.5-Coder-1.5B-Instruct](Qwen-Qwen2.5-Coder-1.5B-Instruct/aitk/qwen2_5_trtrtx.json) | [Qwen-Qwen2.5-Coder-1.5B-Instruct](Qwen-Qwen2.5-Coder-1.5B-Instruct/aitk/qwen2_5_ov_npu_config.json) | [google-vit-base-patch16-224](google-vit-base-patch16-224/aitk/vit-base-patch16-224_qdq_qnn.json) | [microsoft-Phi-3-mini-4k-instruct](microsoft-Phi-3-mini-4k-instruct/aitk/phi3_vitis_ai_config.json) | | +| [meta-llama-Llama-3.2-1B-Instruct-mixed](meta-llama-Llama-3.2-1B-Instruct/olive/mixed.json) | [meta-llama-Llama-3.2-1B-Instruct-qlora](meta-llama-Llama-3.2-1B-Instruct/olive/qlora.json) | | | [Qwen-Qwen2.5-Coder-14B-Instruct](Qwen-Qwen2.5-Coder-14B-Instruct/aitk/qwen2_5_trtrtx.json) | [Qwen-Qwen2.5-Coder-14B-Instruct](Qwen-Qwen2.5-Coder-14B-Instruct/aitk/qwen2_5_ov_config.json) | [google-vit-base-patch16-224](google-vit-base-patch16-224/aitk/vit-base-patch16-224_qnn_gpu.json) | [microsoft-Phi-3.5-mini-instruct](microsoft-Phi-3.5-mini-instruct/aitk/phi3_5_vitis_ai_config.json) | | +| [meta-llama-Llama-3.2-1B-Instruct-qlora](meta-llama-Llama-3.2-1B-Instruct/olive/qlora.json) | [microsoft-Phi-3.5-mini-instruct](microsoft-Phi-3.5-mini-instruct/olive/mixed.json) | | | [Qwen-Qwen2.5-Coder-7B-Instruct](Qwen-Qwen2.5-Coder-7B-Instruct/aitk/qwen2_5_trtrtx.json) | [Qwen-Qwen2.5-Coder-14B-Instruct](Qwen-Qwen2.5-Coder-14B-Instruct/aitk/qwen2_5_ov_npu_config.json) | [intel-bert-base-uncased-mrpc](intel-bert-base-uncased-mrpc/QNN/config_gpu_fp32.json) | [microsoft-Phi-4-mini-instruct](microsoft-Phi-4-mini-instruct/aitk/phi4_vitis_ai_config.json) | | +| [meta-llama-Meta-Llama-3-8B](meta-llama-Meta-Llama-3-8B/olive/config.json) | [microsoft-Phi-4-mini-instruct-mixed-tied](microsoft-Phi-4-mini-instruct/olive/mixed-tied.json) | | | [Qwen2.5-0.5B-Instruct_Model_Builder_FP16](Qwen-Qwen2.5-0.5B-Instruct/NvTensorRtRtx/Qwen2.5-0.5B-Instruct_model_builder_fp16.json) | [Qwen-Qwen2.5-Coder-3B-Instruct](Qwen-Qwen2.5-Coder-3B-Instruct/aitk/qwen2_5_ov_config.json) | [intel-bert-base-uncased-mrpc](intel-bert-base-uncased-mrpc/aitk/bert_qdq_qnn.json) | [microsoft-Phi-4-mini-reasoning](microsoft-Phi-4-mini-reasoning/aitk/phi4_vitis_ai_config.json) | | +| [microsoft-deberta-base-mnli](microsoft-deberta-base-mnli/aml/deberta.json) | [microsoft-Phi-4-mini-instruct-mixed](microsoft-Phi-4-mini-instruct/olive/mixed.json) | | | [Qwen2.5-14B-Instruct_Model_Builder_INT4](Qwen-Qwen2.5-14B-Instruct/NvTensorRtRtx/Qwen2.5-14B-Instruct_model_builder_int4.json) | [Qwen-Qwen2.5-Coder-3B-Instruct](Qwen-Qwen2.5-Coder-3B-Instruct/aitk/qwen2_5_ov_npu_config.json) | [intel-bert-base-uncased-mrpc](intel-bert-base-uncased-mrpc/aitk/bert_qnn_gpu.json) | [microsoft-resnet-50](microsoft-resnet-50/aitk/resnet_qdq_amd.json) | | +| [ministral_3_3b](mistralai-Ministral-3-3B-Instruct-2512/builtin/optimize.py) | [ministral_3_3b](mistralai-Ministral-3-3B-Instruct-2512/builtin/optimize.py) | | | [Qwen2.5-7B-Instruct_Model_Builder_INT4](Qwen-Qwen2.5-7B-Instruct/NvTensorRtRtx/Qwen2.5-7B-Instruct_model_builder_int4.json) | [Qwen-Qwen2.5-Coder-7B-Instruct](Qwen-Qwen2.5-Coder-7B-Instruct/aitk/qwen2_5_ov_config.json) | [laion-CLIP-ViT-B-32-laion2B-s34B-b79K](laion-CLIP-ViT-B-32-laion2B-s34B-b79K/QNN/config_gpu_fp32.json) | [mistralai-Mistral-7B-Instruct-v0.2](mistralai-Mistral-7B-Instruct-v0.2/aitk/Mistral_7B_Instruct_v0.2_vitis_ai_config.json) | | +| [moonshine-tiny](usefulSensors-moonshine-tiny/cli/optimize.sh) | [mistral-7b](facebook-opt-125m/cli/optimize.sh) | | | [Qwen2.5-Coder-0.5B-Instruct_Model_Builder_FP16](Qwen-Qwen2.5-Coder-0.5B-Instruct/NvTensorRtRtx/Qwen2.5-Coder-0.5B-Instruct_model_builder_fp16.json) | [Qwen-Qwen2.5-Coder-7B-Instruct](Qwen-Qwen2.5-Coder-7B-Instruct/aitk/qwen2_5_ov_npu_config.json) | [laion-CLIP-ViT-B-32-laion2B-s34B-b79K](laion-CLIP-ViT-B-32-laion2B-s34B-b79K/aitk/laion_clip_qnn.json) | [openai-clip-vit-base-patch16](openai-clip-vit-base-patch16/aitk/openai_clip_qdq_amd.json) | | +| [openai-whisper-base-cpu-int8](openai-whisper-base/cpu/whisper-base_cpu_int8.json) | [mistral-7b](mistralai-Mistral-7B-v0.1/cli/optimize.sh) | | | [Qwen2.5-Coder-1.5B-Instruct_Model_Builder_FP16](Qwen-Qwen2.5-Coder-1.5B-Instruct/NvTensorRtRtx/Qwen2.5-Coder-1.5B-Instruct_model_builder_fp16.json) | [deepseek-ai-DeepSeek-R1-Distill-Llama-8B](deepseek-ai-DeepSeek-R1-Distill-Llama-8B/aitk/deepseek_ov_config.json) | [laion-CLIP-ViT-B-32-laion2B-s34B-b79K](laion-CLIP-ViT-B-32-laion2B-s34B-b79K/aitk/laion_clip_qnn_gpu.json) | [openai-clip-vit-base-patch32](openai-clip-vit-base-patch32/aitk/openai_clip_qdq_amd.json) | | +| [openai-whisper-base.en-cpu-int8](openai-whisper-base.en/cpu/whisper-base.en_cpu_int8.json) | [mistral-7b](open-llama-3b/cli/optimize.sh) | | | [Qwen2.5-Coder-14B-Instruct_Model_Builder_INT4](Qwen-Qwen2.5-Coder-14B-Instruct/NvTensorRtRtx/Qwen2.5-Coder-14B-Instruct_model_builder_int4.json) | [deepseek-ai-DeepSeek-R1-Distill-Llama-8B](deepseek-ai-DeepSeek-R1-Distill-Llama-8B/aitk/deepseek_ov_npu_config.json) | [llama3.1-8b-instruct-x-elite](meta-llama-Llama-3.1-8B-Instruct/QNN/x_elite_config.json) | [openai-clip-vit-large-patch14](openai-clip-vit-large-patch14/aitk/openai_clip_qdq_amd.json) | | +| [openai-whisper-large-cpu-int8](openai-whisper-large/cpu/whisper-large_cpu_int8.json) | [moonshine-tiny](usefulSensors-moonshine-tiny/cli/optimize.sh) | | | [Qwen2.5-Coder-7B-Instruct_Model_Builder_INT4](Qwen-Qwen2.5-Coder-7B-Instruct/NvTensorRtRtx/Qwen2.5-Coder-7B-Instruct_model_builder_int4.json) | [deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B](deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B/aitk/deepseek_ov_config.json) | [llama3.1-8b-instruct-x2-elite](meta-llama-Llama-3.1-8B-Instruct/QNN/x2_elite_config.json) | [openai-whisper-large-v3-turbo-vitisai](openai-whisper-large-v3-turbo/VitisAI/run_whisper.py) | | +| [openai-whisper-large-v2-cpu-int8](openai-whisper-large-v2/cpu/whisper-large-v2_cpu_int8.json) | [openai-whisper-base-cuda-int8](openai-whisper-base/cuda/whisper-base_cuda_int8.json) | | | [Qwen2.5_1.5B_Instruct_Model_Builder_FP16](Qwen-Qwen2.5-1.5B-Instruct/NvTensorRtRtx/Qwen2.5-1.5B-Instruct_model_builder_fp16.json) | [deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B](deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B/aitk/deepseek_ov_gpu_config.json) | [meta-llama-Llama-3.1-8B-Instruct](meta-llama-Llama-3.1-8B-Instruct/aitk/llama3_1_qnn_config.json) | [openai-whisper-medium-vitisai](openai-whisper-medium/VitisAI/run_whisper.py) | | +| [openai-whisper-large-v3-cpu-int8](openai-whisper-large-v3/cpu/whisper-large-v3_cpu_int8.json) | [openai-whisper-base.en-cuda-int8](openai-whisper-base.en/cuda/whisper-base.en_cuda_int8.json) | | | [deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B](deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B/aitk/deepseek_trtrtx_config.json) | [deepseek-ai-DeepSeek-R1-Distill-Qwen-14B](deepseek-ai-DeepSeek-R1-Distill-Qwen-14B/aitk/deepseek_ov_config.json) | [meta-llama-Llama-3.1-8B-Instruct](meta-llama-Llama-3.1-8B-Instruct/aitk/llama3_1_qnn_gpu.json) | [openai-whisper-small-vitisai](openai-whisper-small/VitisAI/run_whisper.py) | | +| [openai-whisper-large-v3-turbo-cpu-int8](openai-whisper-large-v3-turbo/cpu/whisper-large-v3-turbo_cpu_int8.json) | [openai-whisper-large-cuda-int8](openai-whisper-large/cuda/whisper-large_cuda_int8.json) | | | [deepseek-ai-DeepSeek-R1-Distill-Qwen-14B](deepseek-ai-DeepSeek-R1-Distill-Qwen-14B/aitk/deepseek_trtrtx.json) | [deepseek-ai-DeepSeek-R1-Distill-Qwen-14B](deepseek-ai-DeepSeek-R1-Distill-Qwen-14B/aitk/deepseek_ov_npu_config.json) | [meta-llama-Llama-3.2-1B-Instruct](meta-llama-Llama-3.2-1B-Instruct/QNN/config_gpu.json) | [stable-diffusion-v1-5-safety-checker](sd-legacy-stable-diffusion-v1-5/VitisAI/config_safety_checker.json) | | +| [openai-whisper-large-v3-turbo](openai-whisper-large-v3-turbo/olive/whisper_large_v3_turbo_decoder_fp32.json) | [openai-whisper-large-v2-cuda-int8](openai-whisper-large-v2/cuda/whisper-large-v2_cuda_int8.json) | | | [deepseek-ai-DeepSeek-R1-Distill-Qwen-7B](deepseek-ai-DeepSeek-R1-Distill-Qwen-7B/aitk/deepseek_trtrtx.json) | [deepseek-ai-DeepSeek-R1-Distill-Qwen-7B](deepseek-ai-DeepSeek-R1-Distill-Qwen-7B/aitk/deepseek_ov_config.json) | [meta-llama-Llama-3.2-1B-Instruct](meta-llama-Llama-3.2-1B-Instruct/QNN/config_gpu_ctxbin.json) | [stable-diffusion-v1-5-text-encoder](sd-legacy-stable-diffusion-v1-5/VitisAI/config_text_encoder.json) | | +| [openai-whisper-large-v3-turbo](openai-whisper-large-v3-turbo/olive/whisper_large_v3_turbo_decoder_qdq.json) | [openai-whisper-large-v3-cuda-int8](openai-whisper-large-v3/cuda/whisper-large-v3_cuda_int8.json) | | | [google-bert-bert-base-multilingual-cased](google-bert-bert-base-multilingual-cased/aitk/bert-base-multilingual-cased_trtrtx.json) | [deepseek-ai-DeepSeek-R1-Distill-Qwen-7B](deepseek-ai-DeepSeek-R1-Distill-Qwen-7B/aitk/deepseek_ov_npu_config.json) | [meta-llama-Llama-3.2-1B-Instruct](meta-llama-Llama-3.2-1B-Instruct/aitk/llama3_2_qnn_config.json) | [stable-diffusion-v1-5-unet](sd-legacy-stable-diffusion-v1-5/VitisAI/config_unet.json) | | +| [openai-whisper-large-v3-turbo](openai-whisper-large-v3-turbo/olive/whisper_large_v3_turbo_encoder_fp32.json) | [openai-whisper-large-v3-turbo-cuda-int8](openai-whisper-large-v3-turbo/cuda/whisper-large-v3-turbo_cuda_int8.json) | | | [google-vit-base-patch16-224](google-vit-base-patch16-224/aitk/vit-base-patch16-224_trtrtx.json) | [google-bert-bert-base-multilingual-cased](google-bert-bert-base-multilingual-cased/aitk/bert-base-multilingual-cased_context_ov_static.json) | [meta-llama-Llama-3.2-1B-Instruct](meta-llama-Llama-3.2-1B-Instruct/aitk/llama3_2_qnn_gpu_config.json) | [stable-diffusion-v1-5-vae-decoder](sd-legacy-stable-diffusion-v1-5/VitisAI/config_vae_decoder.json) | | +| [openai-whisper-large-v3-turbo](openai-whisper-large-v3-turbo/olive/whisper_large_v3_turbo_encoder_qdq.json) | [openai-whisper-medium-cuda-int8](openai-whisper-medium/cuda/whisper-medium_cuda_int8.json) | | | [intel-bert-base-uncased-mrpc](intel-bert-base-uncased-mrpc/aitk/bert_trtrtx.json) | [google-gemma-3-1b-it](google-gemma-3-1b-it/OpenVINO/gemma_3_1b_it_context_ov_gpu_config.json) | [microsoft-Phi-3-mini-128k-instruct](microsoft-Phi-3-mini-128k-instruct/QNN/config.json) | [stable-diffusion-v1-5-vae-encoder](sd-legacy-stable-diffusion-v1-5/VitisAI/config_vae_encoder.json) | | +| [openai-whisper-medium-cpu-int8](openai-whisper-medium/cpu/whisper-medium_cpu_int8.json) | [openai-whisper-medium.en-cuda-int8](openai-whisper-medium.en/cuda/whisper-medium.en_cuda_int8.json) | | | [laion-CLIP-ViT-B-32-laion2B-s34B-b79K](laion-CLIP-ViT-B-32-laion2B-s34B-b79K/aitk/laion_clip_trtrtx.json) | [google-gemma-3-1b-it](google-gemma-3-1b-it/OpenVINO/gemma_3_1b_it_context_ov_npu_config.json) | [microsoft-Phi-3-mini-128k-instruct](microsoft-Phi-3-mini-128k-instruct/aitk/phi3_qnn.json) | [timm-mobilenetv3_small_100.lamb_in1k](timm-mobilenetv3_small_100.lamb_in1k/VitisAI/config.json) | | +| [openai-whisper-medium.en-cpu-int8](openai-whisper-medium.en/cpu/whisper-medium.en_cpu_int8.json) | [openai-whisper-small-cuda-int8](openai-whisper-small/cuda/whisper-small_cuda_int8.json) | | | [meta-llama-Llama-3.1-8B-Instruct](meta-llama-Llama-3.1-8B-Instruct/aitk/llama3_1_trtrtx_config.json) | [google-vit-base-patch16-224](google-vit-base-patch16-224/OpenVINO/vit_base_patch16_224_context_ov_static.json) | [microsoft-Phi-3-mini-4k-instruct](microsoft-Phi-3-mini-4k-instruct/QNN/config.json) | | | +| [openai-whisper-small-cpu-int8](openai-whisper-small/cpu/whisper-small_cpu_int8.json) | [openai-whisper-small.en-cuda-int8](openai-whisper-small.en/cuda/whisper-small.en_cuda_int8.json) | | | [meta-llama-Llama-3.2-1B-Instruct](meta-llama-Llama-3.2-1B-Instruct/aitk/llama3_2_trtrtx_config.json) | [google-vit-base-patch16-224](google-vit-base-patch16-224/aitk/vit_base_patch16_224_context_ov_static.json) | [microsoft-Phi-3-mini-4k-instruct](microsoft-Phi-3-mini-4k-instruct/aitk/phi3_qnn.json) | | | +| [openai-whisper-small.en-cpu-int8](openai-whisper-small.en/cpu/whisper-small.en_cpu_int8.json) | [openai-whisper-tiny-cuda-int8](openai-whisper-tiny/cuda/whisper-tiny_cuda_int8.json) | | | [microsoft-Phi-3-mini-128k-instruct](microsoft-Phi-3-mini-128k-instruct/aitk/phi3_trtrtx.json) | [intel-bert-base-uncased-mrpc (ov)](intel-bert-base-uncased-mrpc/aitk/bert_ov.json) | [microsoft-Phi-3.5-mini-instruct](microsoft-Phi-3.5-mini-instruct/QNN/config.json) | | | +| [openai-whisper-tiny-cpu-int8](openai-whisper-tiny/cpu/whisper-tiny_cpu_int8.json) | [openai-whisper-tiny.en-cuda-int8](openai-whisper-tiny.en/cuda/whisper-tiny.en_cuda_int8.json) | | | [microsoft-Phi-3-mini-4k-instruct](microsoft-Phi-3-mini-4k-instruct/aitk/phi3_trtrtx.json) | [laion-CLIP-ViT-B-32-laion2B-s34B-b79K](laion-CLIP-ViT-B-32-laion2B-s34B-b79K/aitk/laion_clip_ov.json) | [microsoft-Phi-3.5-mini-instruct](microsoft-Phi-3.5-mini-instruct/QNN/config_fp16.json) | | | +| [openai-whisper-tiny.en-cpu-int8](openai-whisper-tiny.en/cpu/whisper-tiny.en_cpu_int8.json) | [qwen2.5-vl-3B-Instruct](Qwen-Qwen2.5-VL-3B-Instruct/builtin/optimize.py) | | | [microsoft-Phi-3.5-mini-instruct](microsoft-Phi-3.5-mini-instruct/aitk/phi3_5_trtrtx_config.json) | [meta-llama-Llama-3.1-8B-Instruct](meta-llama-Llama-3.1-8B-Instruct/aitk/llama3_1_ov_config.json) | [microsoft-Phi-3.5-mini-instruct](microsoft-Phi-3.5-mini-instruct/QNN/config_gpu.json) | | | +| [qwen2.5-vl-3B-Instruct](Qwen-Qwen2.5-VL-3B-Instruct/builtin/optimize.py) | [qwen3.5-0.8B-Instruct](Qwen-Qwen3.5-0.8B/builtin/optimize.py) | | | [microsoft-Phi-4-mini-instruct](microsoft-Phi-4-mini-instruct/aitk/phi4_trtrtx.json) | [meta-llama-Llama-3.1-8B-Instruct](meta-llama-Llama-3.1-8B-Instruct/aitk/llama3_1_ov_gpu_config.json) | [microsoft-Phi-3.5-mini-instruct](microsoft-Phi-3.5-mini-instruct/QNN/config_gpu_ctxbin.json) | | | +| [qwen3.5-0.8B-Instruct](Qwen-Qwen3.5-0.8B/builtin/optimize.py) | [qwen3.5-27B](Qwen-Qwen3.5-27B/builtin/optimize.py) | | | [microsoft-Phi-4-mini-instruct_nvmo_ptq_mixed_precision_awq_lite](microsoft-Phi-4-mini-instruct/NvTensorRtRtx/microsoft-Phi-4-mini-instruct_nvmo_ptq_mixed_precision_awq_lite.json) | [meta-llama-Llama-3.2-1B-Instruct](meta-llama-Llama-3.2-1B-Instruct/aitk/llama3_2_ov_config.json) | [microsoft-Phi-3.5-mini-instruct](microsoft-Phi-3.5-mini-instruct/aitk/phi3_5_qnn_config.json) | | | +| [qwen3.5-2B](Qwen-Qwen3.5-2B/builtin/optimize.py) | [qwen3.5-2B](Qwen-Qwen3.5-2B/builtin/optimize.py) | | | [microsoft-Phi-4](microsoft-Phi-4/aitk/phi4_trtrtx.json) | [meta-llama-Llama-3.2-1B-Instruct](meta-llama-Llama-3.2-1B-Instruct/aitk/llama3_2_ov_gpu_config.json) | [microsoft-Phi-3.5-mini-instruct](microsoft-Phi-3.5-mini-instruct/aitk/phi3_5_qnn_gpu_config.json) | | | +| [qwen3.5-35B-A3B-MoE](Qwen-Qwen3.5-35B-A3B/builtin/optimize.py) | [qwen3.5-4B](Qwen-Qwen3.5-4B/builtin/optimize.py) | | | [microsoft-resnet-50](microsoft-resnet-50/aitk/resnet_trtrtx.json) | [microsoft-Phi-3-mini-128k-instruct](microsoft-Phi-3-mini-128k-instruct/aitk/phi3_ov_config.json) | [microsoft-Phi-4-mini-instruct](microsoft-Phi-4-mini-instruct/QNN/phi4_mini_qnn_docker.json) | | | +| [qwen3.5-4B](Qwen-Qwen3.5-4B/builtin/optimize.py) | [qwen3.5-9B](Qwen-Qwen3.5-9B/builtin/optimize.py) | | | [mistralai-Mistral-7B-Instruct-v0.2](mistralai-Mistral-7B-Instruct-v0.2/aitk/Mistral_7B_Instruct_v0.2_trtrtx.json) | [microsoft-Phi-3-mini-128k-instruct](microsoft-Phi-3-mini-128k-instruct/aitk/phi3_ov_npu_config.json) | [microsoft-Phi-4-mini-instruct](microsoft-Phi-4-mini-instruct/aitk/phi4_qnn.json) | | | +| [qwen3.5-9B](Qwen-Qwen3.5-9B/builtin/optimize.py) | [qwen3vl-2B-Instruct](Qwen-Qwen3-VL-2B-Instruct/builtin/optimize.py) | | | [openai-clip-vit-base-patch16](openai-clip-vit-base-patch16/aitk/openai_clip_trtrtx.json) | [microsoft-Phi-3-mini-4k-instruct](microsoft-Phi-3-mini-4k-instruct/aitk/phi3_ov_config.json) | [microsoft-Phi-4-reasoning](microsoft-Phi-4-reasoning/QAIRT/htp_sc8380xp.json) | | | +| [qwen3vl-2B-Instruct](Qwen-Qwen3-VL-2B-Instruct/builtin/optimize.py) | [qwen3vl-4B-Instruct](Qwen-Qwen3-VL-4B-Instruct/builtin/optimize.py) | | | [openai-clip-vit-base-patch32](openai-clip-vit-base-patch32/aitk/openai_clip_trtrtx.json) | [microsoft-Phi-3-mini-4k-instruct](microsoft-Phi-3-mini-4k-instruct/aitk/phi3_ov_npu_config.json) | [microsoft-Phi-4-reasoning](microsoft-Phi-4-reasoning/QAIRT/htp_sc8480xp.json) | | | +| [qwen3vl-4B-Instruct](Qwen-Qwen3-VL-4B-Instruct/builtin/optimize.py) | [qwen3vl-8B-Instruct](Qwen-Qwen3-VL-8B-Instruct/builtin/optimize.py) | | | [openai-clip-vit-large-patch14](openai-clip-vit-large-patch14/aitk/openai_clip_trtrtx.json) | [microsoft-Phi-3.5-mini-instruct](microsoft-Phi-3.5-mini-instruct/aitk/phi3_5_ov_config.json) | [microsoft-Phi-4-reasoning](microsoft-Phi-4-reasoning/QNN/config.json) | | | +| [qwen3vl-8B-Instruct](Qwen-Qwen3-VL-8B-Instruct/builtin/optimize.py) | [sshleifer-tiny-gpt2-sparsegpt](sshleifer-tiny-gpt2/olive/sparsegpt.json) | | | [phi-4_Model_Builder_INT4](microsoft-Phi-4/NvTensorRtRtx/phi-4_model_builder_int4.json) | [microsoft-Phi-3.5-mini-instruct](microsoft-Phi-3.5-mini-instruct/aitk/phi3_5_ov_gpu_config.json) | [microsoft-resnet-50](microsoft-resnet-50/aitk/resnet_qdq_qnn.json) | | | +| [sshleifer-tiny-gpt2-sparsegpt](sshleifer-tiny-gpt2/olive/sparsegpt.json) | [stable-diffusion-v1-4-safety-checker](compvis-stable-diffusion-v1-4/olive/config_safety_checker.json) | | | | [microsoft-Phi-4-mini-instruct](microsoft-Phi-4-mini-instruct/OpenVINO/Phi-4-mini-instruct-gpu-context-dy.json) | [microsoft-resnet-50](microsoft-resnet-50/aitk/resnet_qnn_gpu.json) | | | +| [stable-diffusion-v1-4-safety-checker](compvis-stable-diffusion-v1-4/olive/config_safety_checker.json) | [stable-diffusion-v1-4-text-encoder](compvis-stable-diffusion-v1-4/olive/config_text_encoder.json) | | | | [microsoft-Phi-4-mini-instruct](microsoft-Phi-4-mini-instruct/OpenVINO/Phi_4_mini_instruct_context_ov_dynamic_sym_gs128_bkp_int8_sym.json) | [microsoft-table-transformer-detection](microsoft-table-transformer-detection/QNN/ttd_config.json) | | | +| [stable-diffusion-v1-4-text-encoder](compvis-stable-diffusion-v1-4/olive/config_text_encoder.json) | [stable-diffusion-v1-4-unet](compvis-stable-diffusion-v1-4/olive/config_unet.json) | | | | [microsoft-Phi-4-mini-instruct](microsoft-Phi-4-mini-instruct/aitk/phi4_ov_config.json) | [openai-clip-vit-base-patch16](openai-clip-vit-base-patch16/QNN/config_gpu_fp32.json) | | | +| [stable-diffusion-v1-4-unet](compvis-stable-diffusion-v1-4/olive/config_unet.json) | [stable-diffusion-v1-4-vae-decoder](compvis-stable-diffusion-v1-4/olive/config_vae_decoder.json) | | | | [microsoft-Phi-4-mini-instruct](microsoft-Phi-4-mini-instruct/aitk/phi4_ov_npu_config.json) | [openai-clip-vit-base-patch16](openai-clip-vit-base-patch16/aitk/openai_clip_qnn.json) | | | +| [stable-diffusion-v1-4-vae-decoder](compvis-stable-diffusion-v1-4/olive/config_vae_decoder.json) | [stable-diffusion-v1-4-vae-encoder](compvis-stable-diffusion-v1-4/olive/config_vae_encoder.json) | | | | [microsoft-Phi-4-mini-reasoning](microsoft-Phi-4-mini-reasoning/OpenVINO/Phi-4-mini-reasoning_context_ov_dynamic_sym_gs128_bkp_int8_sym.json) | [openai-clip-vit-base-patch16](openai-clip-vit-base-patch16/aitk/openai_clip_qnn_gpu.json) | | | +| [stable-diffusion-v1-4-vae-encoder](compvis-stable-diffusion-v1-4/olive/config_vae_encoder.json) | [stable-diffusion-v1-5](sd-legacy-stable-diffusion-v1-5/olive/optimize.sh) | | | | [microsoft-Phi-4-mini-reasoning](microsoft-Phi-4-mini-reasoning/OpenVINO/Phi_4_mini_instruct_context_ov_dynamic_sym_gs128_bkp_int8_sym.json) | [openai-clip-vit-base-patch32](openai-clip-vit-base-patch32/QNN/config_gpu_fp32.json) | | | +| [stable-diffusion-v1-5-safety-checker](sd-legacy-stable-diffusion-v1-5/VitisAI/config_safety_checker.json) | [stable-diffusion-xl-base-1.0](stabilityai-stable-diffusion-xl-base-1.0/olive/stable_diffusion_xl.py) | | | | [microsoft-Phi-4-mini-reasoning](microsoft-Phi-4-mini-reasoning/aitk/phi4_ov_config.json) | [openai-clip-vit-base-patch32](openai-clip-vit-base-patch32/aitk/openai_clip_qnn.json) | | | +| [stable-diffusion-v1-5-text-encoder](sd-legacy-stable-diffusion-v1-5/VitisAI/config_text_encoder.json) | | | | | [microsoft-Phi-4-mini-reasoning](microsoft-Phi-4-mini-reasoning/aitk/phi4_ov_gpu_config.json) | [openai-clip-vit-base-patch32](openai-clip-vit-base-patch32/aitk/openai_clip_qnn_gpu.json) | | | +| [stable-diffusion-v1-5-unet](sd-legacy-stable-diffusion-v1-5/VitisAI/config_unet.json) | | | | | [microsoft-Phi-4-reasoning-plus](microsoft-Phi-4-reasoning-plus/OpenVINO/Phi-4-Phi-4-reasoning-plus_context_ov_dynamic_sym_gs128_bkp_int8_sym.json) | [openai-clip-vit-large-patch14](openai-clip-vit-large-patch14/aitk/openai_clip_qnn.json) | | | +| [stable-diffusion-v1-5-vae-decoder](sd-legacy-stable-diffusion-v1-5/VitisAI/config_vae_decoder.json) | | | | | [microsoft-Phi-4-reasoning-plus](microsoft-Phi-4-reasoning-plus/aitk/phi4_ov_config.json) | [openai-whisper-large-v3-turbo](openai-whisper-large-v3-turbo/aitk/qnn_workflow.json) | | | +| [stable-diffusion-v1-5-vae-encoder](sd-legacy-stable-diffusion-v1-5/VitisAI/config_vae_encoder.json) | | | | | [microsoft-Phi-4-reasoning](microsoft-Phi-4-reasoning/OpenVINO/Phi-4-reasoning_context_ov_dynamic_sym_gs128_bkp_int8_sym.json) | [openai-whisper-large-v3-turbo](openai-whisper-large-v3-turbo/olive/whisper_large_v3_turbo_decoder_fp32.json) | | | +| [stable-diffusion-v1-5](sd-legacy-stable-diffusion-v1-5/olive/optimize.sh) | | | | | [microsoft-Phi-4-reasoning](microsoft-Phi-4-reasoning/aitk/phi4_ov_config.json) | [openai-whisper-large-v3-turbo](openai-whisper-large-v3-turbo/olive/whisper_large_v3_turbo_decoder_qdq.json) | | | +| [stable-diffusion-xl-base-1.0](stabilityai-stable-diffusion-xl-base-1.0/olive/stable_diffusion_xl.py) | | | | | [microsoft-Phi-4](microsoft-Phi-4/OpenVINO/phi_4_gpu_context_dy.json) | [openai-whisper-large-v3-turbo](openai-whisper-large-v3-turbo/olive/whisper_large_v3_turbo_decoder_qdq_ctx.json) | | | +| [timm-mobilenetv3_small_100.lamb_in1k](timm-mobilenetv3_small_100.lamb_in1k/olive/config.json) | | | | | [microsoft-Phi-4](microsoft-Phi-4/aitk/phi4_ov_config.json) | [openai-whisper-large-v3-turbo](openai-whisper-large-v3-turbo/olive/whisper_large_v3_turbo_encoder_fp32.json) | | | +| [translategemma-4b-it](google-translategemma-4b-it/builtin/optimize.py) | | | | | [microsoft-resnet-50](microsoft-resnet-50/aitk/resnet_context_ov_static.json) | [openai-whisper-large-v3-turbo](openai-whisper-large-v3-turbo/olive/whisper_large_v3_turbo_encoder_qdq.json) | | | +| [videochat-flash-qwen2_5-7b-internvideo2-1b](OpenGVLab-VideoChat-Flash-Qwen2_5-7B_InternVideo2-1B/builtin/optimize.py) | | | | | [mistralai-Mistral-7B-Instruct-v0.2](mistralai-Mistral-7B-Instruct-v0.2/aitk/Mistral_7B_Instruct_v0.2_gpu_context_ov_dy.json) | [openai-whisper-large-v3-turbo](openai-whisper-large-v3-turbo/olive/whisper_large_v3_turbo_encoder_qdq_ctx.json) | | | +| | | | | | [mistralai-Mistral-7B-Instruct-v0.2](mistralai-Mistral-7B-Instruct-v0.2/aitk/Mistral_7B_Instruct_v0.2_npu_context_ov_dy.json) | [qwen2.5-7b-instruct](Qwen-Qwen2.5-7B-Instruct/QNN/config.json) | | | +| | | | | | [mistralai-Mistral-7B-Instruct-v0.3](mistralai-Mistral-7B-Instruct-v0.3/aitk/mistral-7b-instruct-v0.3-ov.json) | [sam-vit-base](sam-vit-base/QNN/sam_mask_box_decoder_qnn_fp16.json) | | | +| | | | | | [openai-clip-vit-base-patch16](openai-clip-vit-base-patch16/aitk/openai_clip_ov.json) | [sam-vit-base](sam-vit-base/QNN/sam_mask_decoder_qnn_fp16_ctx.json) | | | +| | | | | | [openai-clip-vit-base-patch32](openai-clip-vit-base-patch32/aitk/openai_clip_ov.json) | [sam-vit-base](sam-vit-base/QNN/sam_mask_point_decoder_qnn_fp16.json) | | | +| | | | | | [openai-clip-vit-large-patch14](openai-clip-vit-large-patch14/aitk/openai_clip_ov.json) | [sam-vit-base](sam-vit-base/QNN/sam_vision_encoder_qnn_w8a8_ctx.json) | | | +| | | | | | [openai-whisper-large-v3-turbo](openai-whisper-large-v3-turbo/OpenVINO/whisper_large_v3_turbo_default_ov_npu.json) | [sam-vit-base](sam-vit-base/aitk/sam_qnn_workflow.json) | | | +| | | | | | [openai-whisper-large-v3-turbo](openai-whisper-large-v3-turbo/aitk/ov_npu_workflow.json) | [sam2.1-hiera-small](sam2.1-hiera-small/QNN/sam21_mask_decoder_qnn_ctx.json) | | | +| | | | | | [openai-whisper-large-v3-turbo](openai-whisper-large-v3-turbo/aitk/ov_workflow.json) | [sam2.1-hiera-small](sam2.1-hiera-small/QNN/sam21_vision_encoder_qnn_ctx.json) | | | +| | | | | | [sam2.1-hiera-small](sam2.1-hiera-small/OpenVINO/sam21_mask_decoder_ov.json) | [sam2.1-hiera-small](sam2.1-hiera-small/aitk/sam2_qnn_workflow.json) | | | +| | | | | | [sam2.1-hiera-small](sam2.1-hiera-small/OpenVINO/sam21_vision_encoder_ov.json) | [sd-legacy-stable-diffusion-v1-5](sd-legacy-stable-diffusion-v1-5/aitk/sd_qnn_workflow.json) | | | +| | | | | | [sam2.1-hiera-small](sam2.1-hiera-small/aitk/sam2_ov_workflow.json) | [sd2-community-stable-diffusion-2-1](sd2-community-stable-diffusion-2-1/aitk/sd_qnn_workflow.json) | | | +| | | | | | [sd-legacy-stable-diffusion-v1-5](sd-legacy-stable-diffusion-v1-5/aitk/sd_ov_npu_workflow.json) | [stable-diffusion-v1-4-safety-checker](compvis-stable-diffusion-v1-4/olive/config_safety_checker.json) | | | +| | | | | | [sd-legacy-stable-diffusion-v1-5](sd-legacy-stable-diffusion-v1-5/aitk/sd_ov_workflow.json) | [stable-diffusion-v1-4-text-encoder](compvis-stable-diffusion-v1-4/olive/config_text_encoder.json) | | | +| | | | | | [sd2-community-stable-diffusion-2-1](sd2-community-stable-diffusion-2-1/aitk/sd_ov_npu_workflow.json) | [stable-diffusion-v1-4-unet](compvis-stable-diffusion-v1-4/olive/config_unet.json) | | | +| | | | | | [sd2-community-stable-diffusion-2-1](sd2-community-stable-diffusion-2-1/aitk/sd_ov_workflow.json) | [stable-diffusion-v1-4-vae-decoder](compvis-stable-diffusion-v1-4/olive/config_vae_decoder.json) | | | +| | | | | | [stable-diffusion-v1-4-safety-checker](compvis-stable-diffusion-v1-4/olive/config_safety_checker.json) | [stable-diffusion-v1-4-vae-encoder](compvis-stable-diffusion-v1-4/olive/config_vae_encoder.json) | | | +| | | | | | [stable-diffusion-v1-4-text-encoder](compvis-stable-diffusion-v1-4/olive/config_text_encoder.json) | [stable-diffusion-v1-5](sd-legacy-stable-diffusion-v1-5/olive/optimize.sh) | | | +| | | | | | [stable-diffusion-v1-4-unet](compvis-stable-diffusion-v1-4/olive/config_unet.json) | [stable-diffusion-xl-base-1.0](stabilityai-stable-diffusion-xl-base-1.0/olive/stable_diffusion_xl.py) | | | +| | | | | | [stable-diffusion-v1-4-vae-decoder](compvis-stable-diffusion-v1-4/olive/config_vae_decoder.json) | [timm-mobilenetv3_small_100.lamb_in1k](timm-mobilenetv3_small_100.lamb_in1k/QNN/mobilenet_qnn_ep.json) | | | +| | | | | | [stable-diffusion-v1-4-vae-encoder](compvis-stable-diffusion-v1-4/olive/config_vae_encoder.json) | | | | +| | | | | | [stable-diffusion-v1-5](sd-legacy-stable-diffusion-v1-5/olive/optimize.sh) | | | | From 8c08fb471ec52ccd099e18e835d40a346c3e7f73 Mon Sep 17 00:00:00 2001 From: xieofxie Date: Mon, 20 Jul 2026 13:57:40 +0800 Subject: [PATCH 02/10] aitk: add qwen3 qnn (#556) Co-authored-by: hualxie --- .aitk/configs/checks.json | 26 +- .aitk/configs/model_list.json | 20 +- .../requirements-NvidiaGPU-Qwen3.txt | 3 + .aitk/requirements/requirements-QNN-Qwen3.txt | 4 + .aitk/requirements/requirements-QNN.txt | 2 + .aitk/scripts/project_processor.py | 7 +- .aitk/scripts/sanitize/generator_amd.py | 4 +- .aitk/scripts/sanitize/generator_common.py | 11 + .aitk/scripts/sanitize/generator_dml.py | 4 +- .aitk/scripts/sanitize/generator_intel.py | 4 +- .aitk/scripts/sanitize/generator_qnn.py | 3 +- .aitk/scripts/sanitize/generator_trtrtx.py | 3 +- Qwen-Qwen3-0.6B/aitk/.gitignore | 5 + Qwen-Qwen3-0.6B/aitk/README.md | 17 ++ Qwen-Qwen3-0.6B/aitk/_copy.json.config | 18 ++ Qwen-Qwen3-0.6B/aitk/inference_sample.ipynb | 115 +++++++++ Qwen-Qwen3-0.6B/aitk/info.yml | 20 ++ Qwen-Qwen3-0.6B/aitk/model_project.config | 12 + Qwen-Qwen3-0.6B/aitk/qwen3_qnn_config.json | 157 ++++++++++++ .../aitk/qwen3_qnn_config.json.config | 226 ++++++++++++++++++ Qwen-Qwen3-0.6B/aitk/requirements.txt | 3 + Qwen-Qwen3-0.6B/aitk/winml.py | 58 +++++ 22 files changed, 701 insertions(+), 21 deletions(-) create mode 100644 .aitk/requirements/requirements-NvidiaGPU-Qwen3.txt create mode 100644 .aitk/requirements/requirements-QNN-Qwen3.txt create mode 100644 Qwen-Qwen3-0.6B/aitk/.gitignore create mode 100644 Qwen-Qwen3-0.6B/aitk/README.md create mode 100644 Qwen-Qwen3-0.6B/aitk/_copy.json.config create mode 100644 Qwen-Qwen3-0.6B/aitk/inference_sample.ipynb create mode 100644 Qwen-Qwen3-0.6B/aitk/info.yml create mode 100644 Qwen-Qwen3-0.6B/aitk/model_project.config create mode 100644 Qwen-Qwen3-0.6B/aitk/qwen3_qnn_config.json create mode 100644 Qwen-Qwen3-0.6B/aitk/qwen3_qnn_config.json.config create mode 100644 Qwen-Qwen3-0.6B/aitk/requirements.txt create mode 100644 Qwen-Qwen3-0.6B/aitk/winml.py diff --git a/.aitk/configs/checks.json b/.aitk/configs/checks.json index 05b7faaa7..033465e29 100644 --- a/.aitk/configs/checks.json +++ b/.aitk/configs/checks.json @@ -1,19 +1,19 @@ { - "configCheck": 174, - "copyCheck": 182, + "configCheck": 175, + "copyCheck": 184, "executePatchPyCheck": 0, - "executeRuntimeCheck": 106, + "executeRuntimeCheck": 107, "extensionCheck": 2, - "gitignoreCheck": 44, + "gitignoreCheck": 45, "inferenceModelCheck": 25, - "ipynbCheck": 52, - "licenseCheck": 41, - "modelProjectCheck": 46, - "oliveCheck": 84, - "oliveJsonCheck": 174, - "pathCheck": 1467, - "requirementsCheck": 44, + "ipynbCheck": 53, + "licenseCheck": 42, + "modelProjectCheck": 47, + "oliveCheck": 85, + "oliveJsonCheck": 175, + "pathCheck": 1480, + "requirementsCheck": 45, "templateCheck": 3, - "venvRequirementsCheck": 23, - "winmlCopyCheck": 38 + "venvRequirementsCheck": 25, + "winmlCopyCheck": 39 } diff --git a/.aitk/configs/model_list.json b/.aitk/configs/model_list.json index 05dd95af0..dd68e434b 100644 --- a/.aitk/configs/model_list.json +++ b/.aitk/configs/model_list.json @@ -485,7 +485,7 @@ "relativePath": "sam-vit-base/aitk", "version": 2, "pipeline_tags": [ - "fill-mask" + "mask-generation" ] }, { @@ -504,7 +504,7 @@ "relativePath": "sam2.1-hiera-small/aitk", "version": 3, "pipeline_tags": [ - "fill-mask" + "mask-generation" ] }, { @@ -840,6 +840,22 @@ "text-generation" ] }, + { + "displayName": "Qwen/Qwen3-0.6B", + "icon": "qwen", + "modelLink": "https://huggingface.co/Qwen/Qwen3-0.6B", + "id": "huggingface/Qwen/Qwen3-0.6B", + "runtimes": [ + "QNN" + ], + "architecture": "Transformer", + "status": "Hide", + "relativePath": "Qwen-Qwen3-0.6B/aitk", + "version": 1, + "pipeline_tags": [ + "text-generation" + ] + }, { "displayName": "sd2-community/stable-diffusion-2-1", "icon": "HuggingFace", diff --git a/.aitk/requirements/requirements-NvidiaGPU-Qwen3.txt b/.aitk/requirements/requirements-NvidiaGPU-Qwen3.txt new file mode 100644 index 000000000..e4280c116 --- /dev/null +++ b/.aitk/requirements/requirements-NvidiaGPU-Qwen3.txt @@ -0,0 +1,3 @@ +olive-ai==0.13.0 +onnxruntime-gpu==1.27.0 +transformers==4.57.3 diff --git a/.aitk/requirements/requirements-QNN-Qwen3.txt b/.aitk/requirements/requirements-QNN-Qwen3.txt new file mode 100644 index 000000000..f82b59efa --- /dev/null +++ b/.aitk/requirements/requirements-QNN-Qwen3.txt @@ -0,0 +1,4 @@ +# uvpip:uninstall onnxruntime-qnn;pre +# uvpip:install onnxruntime-qnn==2.1.0 --no-deps;post +olive-ai==0.13.0 +onnxruntime==1.24.4 diff --git a/.aitk/requirements/requirements-QNN.txt b/.aitk/requirements/requirements-QNN.txt index 91149a688..d16ae99ba 100644 --- a/.aitk/requirements/requirements-QNN.txt +++ b/.aitk/requirements/requirements-QNN.txt @@ -38,6 +38,8 @@ numpy==2.2.4 git+https://github.com/microsoft/Olive.git@b9c975f5651a4446283d9183348add9dcd4ed45f#egg=olive_ai onnx==1.17.0 onnx-ir==0.1.10 +# Uninstall onnxruntime from Qwen3 feature +# uvpip:uninstall onnxruntime onnxruntime-qnn;pre # uvpip:install onnxruntime-qnn==1.23.2 --extra-index-url https://aiinfra.pkgs.visualstudio.com/PublicPackages/_packaging/ORT-Nightly/pypi/simple --no-deps;post onnxscript==0.5.3 optuna==4.2.1 diff --git a/.aitk/scripts/project_processor.py b/.aitk/scripts/project_processor.py index f50b5a333..ac0072d88 100644 --- a/.aitk/scripts/project_processor.py +++ b/.aitk/scripts/project_processor.py @@ -223,7 +223,12 @@ def project_processor(): # model info modelInfo = convert_yaml_to_model_info(root_dir, yml_file, yaml_object) if GlobalVars.fillPipelineTags: - modelInfo.pipeline_tags = fetch_pipeline_tags(modelInfo.modelLink) + fetched_tags = fetch_pipeline_tags(modelInfo.modelLink) + if fetched_tags is None: + modelInfo.pipeline_tags = existing_pipeline_tags.get(modelInfo.id) + print(f"Warning: Could not fetch pipeline tags for {modelInfo.id}, using existing") + else: + modelInfo.pipeline_tags = fetched_tags else: modelInfo.pipeline_tags = existing_pipeline_tags.get(modelInfo.id) if modelInfo.id.lower() in all_ids: diff --git a/.aitk/scripts/sanitize/generator_amd.py b/.aitk/scripts/sanitize/generator_amd.py index bb278bb3c..a15503792 100644 --- a/.aitk/scripts/sanitize/generator_amd.py +++ b/.aitk/scripts/sanitize/generator_amd.py @@ -3,7 +3,7 @@ from typing import Optional from .constants import EPNames, OlivePassNames, OlivePropertyNames, PhaseTypeEnum -from .generator_common import create_model_parameter, set_optimization_path +from .generator_common import apply_runtime_feature_overrides, create_model_parameter, set_optimization_path from .model_info import ModelList from .model_parameter import ModelParameter, OptimizationPath, Section from .parameters import Parameter @@ -250,5 +250,7 @@ def generator_amd(id: str, recipe, folder: Path, modelList: ModelList): if quantize: parameter.sections.append(quantize) + apply_runtime_feature_overrides(aitk, parameter) + parameter.writeIfChanged() print(f"\tGenerated AMD configuration for {file}") diff --git a/.aitk/scripts/sanitize/generator_common.py b/.aitk/scripts/sanitize/generator_common.py index 03d9e75d9..b3b33778e 100644 --- a/.aitk/scripts/sanitize/generator_common.py +++ b/.aitk/scripts/sanitize/generator_common.py @@ -40,6 +40,17 @@ def create_model_parameter(aitk, name: str, configFile: Path): return parameter +def apply_runtime_feature_overrides(aitk: dict, parameter: ModelParameter): + """Apply explicit runtime feature overrides from info.yml aitk block. + + When set in info.yml, these replace any auto-generated values. + """ + for field in ("executeRuntimeFeatures", "evaluationRuntimeFeatures", "pyEnvRuntimeFeatures"): + value = aitk.get(field) + if value is not None: + setattr(parameter, field, value) + + def add_optimization_wa(optimizationPaths: list[OptimizationPath], k: str, v: dict) -> bool: if OlivePropertyNames.Precision in v: optimizationPaths.append( diff --git a/.aitk/scripts/sanitize/generator_dml.py b/.aitk/scripts/sanitize/generator_dml.py index b07ff104f..6fbf5903d 100644 --- a/.aitk/scripts/sanitize/generator_dml.py +++ b/.aitk/scripts/sanitize/generator_dml.py @@ -3,7 +3,7 @@ from typing import Optional from .constants import OlivePassNames, OlivePropertyNames, ParameterTypeEnum, PhaseTypeEnum -from .generator_common import create_model_parameter, set_optimization_path +from .generator_common import apply_runtime_feature_overrides, create_model_parameter, set_optimization_path from .model_info import ModelList from .model_parameter import ModelParameter, OptimizationPath, Section from .parameters import Parameter @@ -78,5 +78,7 @@ def generator_dml(id: str, recipe, folder: Path, modelList: ModelList): if quantize: parameter.sections.append(quantize) + apply_runtime_feature_overrides(aitk, parameter) + parameter.writeIfChanged() print(f"\tGenerated DML configuration for {file}") diff --git a/.aitk/scripts/sanitize/generator_intel.py b/.aitk/scripts/sanitize/generator_intel.py index 38b315c84..bdf820e66 100644 --- a/.aitk/scripts/sanitize/generator_intel.py +++ b/.aitk/scripts/sanitize/generator_intel.py @@ -3,7 +3,7 @@ from typing import Optional from .constants import OliveDeviceTypes, OlivePassNames, OlivePropertyNames, PhaseTypeEnum -from .generator_common import create_model_parameter, set_optimization_path +from .generator_common import apply_runtime_feature_overrides, create_model_parameter, set_optimization_path from .model_parameter import ModelParameter, OptimizationPath, Section from .parameters import Parameter from .utils import isLLM_by_id, open_ex @@ -89,5 +89,7 @@ def generator_intel(id: str, recipe, folder: Path): if quantize: parameter.sections.append(quantize) + apply_runtime_feature_overrides(aitk, parameter) + parameter.writeIfChanged() print(f"\tGenerated Intel configuration for {file}") diff --git a/.aitk/scripts/sanitize/generator_qnn.py b/.aitk/scripts/sanitize/generator_qnn.py index 0447028af..4b81f2a89 100644 --- a/.aitk/scripts/sanitize/generator_qnn.py +++ b/.aitk/scripts/sanitize/generator_qnn.py @@ -3,7 +3,7 @@ from .constants import OlivePassNames, OlivePropertyNames from .generator_amd import generate_quantization_config -from .generator_common import create_model_parameter, set_optimization_path +from .generator_common import apply_runtime_feature_overrides, create_model_parameter, set_optimization_path from .model_info import ModelList from .model_parameter import ModelParameter from .utils import isLLM_by_id, open_ex @@ -53,6 +53,7 @@ def generator_qnn(id: str, recipe, folder: Path, modelList: ModelList): parameter.sections.append(quantize) setup_features(content, parameter) + apply_runtime_feature_overrides(aitk, parameter) parameter.writeIfChanged() print(f"\tGenerated QNN configuration for {file}") diff --git a/.aitk/scripts/sanitize/generator_trtrtx.py b/.aitk/scripts/sanitize/generator_trtrtx.py index 0d82dc61d..bbecffb59 100644 --- a/.aitk/scripts/sanitize/generator_trtrtx.py +++ b/.aitk/scripts/sanitize/generator_trtrtx.py @@ -2,7 +2,7 @@ from pathlib import Path from .constants import OlivePassNames, OlivePropertyNames -from .generator_common import create_model_parameter, set_optimization_path +from .generator_common import apply_runtime_feature_overrides, create_model_parameter, set_optimization_path from .generator_dml import generate_quantization_config from .model_info import ModelList from .model_parameter import ModelParameter @@ -41,6 +41,7 @@ def generator_trtrtx(id: str, recipe, folder: Path, modelList: ModelList): parameter.isLLM = isLLM generate_additional_config(configFile, parameter) + apply_runtime_feature_overrides(aitk, parameter) quantize = generate_quantization_config(configFile, parameter) if quantize: diff --git a/Qwen-Qwen3-0.6B/aitk/.gitignore b/Qwen-Qwen3-0.6B/aitk/.gitignore new file mode 100644 index 000000000..46f598b81 --- /dev/null +++ b/Qwen-Qwen3-0.6B/aitk/.gitignore @@ -0,0 +1,5 @@ +__pycache__ +/cache +/history/*/* +!/history/*/history.config +!/history/*/olive_config.json diff --git a/Qwen-Qwen3-0.6B/aitk/README.md b/Qwen-Qwen3-0.6B/aitk/README.md new file mode 100644 index 000000000..c1e336c10 --- /dev/null +++ b/Qwen-Qwen3-0.6B/aitk/README.md @@ -0,0 +1,17 @@ +# Qwen3-0.6B Model Optimization + +This repository demonstrates the optimization of the [Qwen3-0.6B](https://huggingface.co/Qwen/Qwen3-0.6B) model. The optimization process is divided into these workflows: + +- PTQ + AOT for QNN NPU + +## PTQ + AOT for QNN NPU + +This workflow contains two steps: + +### Post-training Quantization (PTQ) + +This step uses GPTQModel library, MatMul-NBits-QDQ and Static Quantization etc. They are resource-intensive and require GPU acceleration. + +### Ahead of Time (AOT) Compilation + +This step compiles model using QNN Execution Provider in a separate Python environment with onnxruntime-qnn installed. Note that after compilation, the model must run on an EP with same or higher version of QAIRT SDK as the package (https://github.com/onnxruntime/onnxruntime-qnn/releases#release-v2.1.0). diff --git a/Qwen-Qwen3-0.6B/aitk/_copy.json.config b/Qwen-Qwen3-0.6B/aitk/_copy.json.config new file mode 100644 index 000000000..a5729934b --- /dev/null +++ b/Qwen-Qwen3-0.6B/aitk/_copy.json.config @@ -0,0 +1,18 @@ +{ + "copies": [ + { + "src": "../../deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B/aitk/inference_sample.ipynb", + "dst": "inference_sample.ipynb", + "replacements": [ + { + "find": "<|User|>{input}<|Assistant|>", + "replace": "<|im_start|>user\\\\n{input}<|im_end|>\\\\n<|im_start|>assistant\\\\n" + } + ] + }, + { + "src": "../../deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B/aitk/winml.py", + "dst": "winml.py" + } + ] +} diff --git a/Qwen-Qwen3-0.6B/aitk/inference_sample.ipynb b/Qwen-Qwen3-0.6B/aitk/inference_sample.ipynb new file mode 100644 index 000000000..2d4b94b3e --- /dev/null +++ b/Qwen-Qwen3-0.6B/aitk/inference_sample.ipynb @@ -0,0 +1,115 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "text = 'Who is Isaac Newton?'\n", + "ExecutionProvider=\"QNNExecutionProvider\"\n", + "model_folder = \"./model\"" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "from winml import register_execution_providers\n", + "register_execution_providers()" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "import onnxruntime_genai as og\n", + "import time\n", + "\n", + "# Load the base model and tokenizer\n", + "model = og.Model(model_folder)\n", + "tokenizer = og.Tokenizer(model)\n", + "tokenizer_stream = tokenizer.create_stream()\n", + "\n", + "# Set the max length to something sensible by default,\n", + "# since otherwise it will be set to the entire context length\n", + "search_options = {}\n", + "search_options[\"max_length\"] = 200\n", + "\n", + "chat_template = \"<|im_start|>user\\n{input}<|im_end|>\\n<|im_start|>assistant\\n\"\n", + "\n", + "# Generate prompt (prompt template + input)\n", + "prompt = f\"{chat_template.format(input=text)}\"\n", + "\n", + "# Encode the prompt using the tokenizer\n", + "input_tokens = tokenizer.encode(prompt)\n", + "\n", + "# Create params and generator\n", + "params = og.GeneratorParams(model)\n", + "params.set_search_options(**search_options)\n", + "generator = og.Generator(model, params)\n", + "\n", + "# Append input tokens to the generator\n", + "generator.append_tokens(input_tokens)\n", + "\n", + "print(\"\")\n", + "print(\"Output: \", end=\"\", flush=True)\n", + "\n", + "token_times = []\n", + "\n", + "# Stream the output\n", + "while True:\n", + " start_time = time.time()\n", + " if generator.is_done():\n", + " break\n", + " generator.generate_next_token()\n", + " new_token = generator.get_next_tokens()[0]\n", + " end_time = time.time()\n", + " \n", + " # Record the time for this token generation\n", + " token_time = end_time - start_time\n", + " token_times.append(token_time)\n", + "\n", + " print(tokenizer_stream.decode(new_token), end=\"\", flush=True)\n", + "\n", + "print()\n", + "\n", + "# Calculate and display timing statistics\n", + "if token_times:\n", + " total_tokens = len(token_times)\n", + " avg_time = sum(token_times) / total_tokens\n", + " \n", + " print(f\"Total tokens generated: {total_tokens}\")\n", + " print(f\"Average time per token: {avg_time:.4f} seconds\")\n", + " print(f\"Tokens per second: {total_tokens / sum(token_times):.2f}\")\n", + "\n", + "del generator\n" + ] + } + ], + "metadata": { + "kernelspec": { + "display_name": "Python 3", + "language": "python", + "name": "python3" + }, + "language_info": { + "codemirror_mode": { + "name": "ipython", + "version": 3 + }, + "file_extension": ".py", + "mimetype": "text/x-python", + "name": "python", + "nbconvert_exporter": "python", + "pygments_lexer": "ipython3", + "version": "3.11.9" + } + }, + "nbformat": 4, + "nbformat_minor": 2 +} diff --git a/Qwen-Qwen3-0.6B/aitk/info.yml b/Qwen-Qwen3-0.6B/aitk/info.yml new file mode 100644 index 000000000..adcffdc96 --- /dev/null +++ b/Qwen-Qwen3-0.6B/aitk/info.yml @@ -0,0 +1,20 @@ +keywords: + aitk +arch: qwen3 +recipes: + - file: "qwen3_qnn_config.json" + device: npu + ep: QNNExecutionProvider + aitk: + oliveFile: "QNN/config.json" + isGPURequired: true + executeRuntimeFeatures: + - GptqModel + - Qwen3 + pyEnvRuntimeFeatures: + - Qwen3 +aitk: + modelInfo: + id: "huggingface/Qwen/Qwen3-0.6B" + version: 1 + status: Hide diff --git a/Qwen-Qwen3-0.6B/aitk/model_project.config b/Qwen-Qwen3-0.6B/aitk/model_project.config new file mode 100644 index 000000000..8021e4620 --- /dev/null +++ b/Qwen-Qwen3-0.6B/aitk/model_project.config @@ -0,0 +1,12 @@ +{ + "workflows": [ + { + "file": "qwen3_qnn_config.json", + "templateName": "qwen3_qnn_config" + } + ], + "modelInfo": { + "id": "huggingface/Qwen/Qwen3-0.6B", + "version": 1 + } +} diff --git a/Qwen-Qwen3-0.6B/aitk/qwen3_qnn_config.json b/Qwen-Qwen3-0.6B/aitk/qwen3_qnn_config.json new file mode 100644 index 000000000..2887a14af --- /dev/null +++ b/Qwen-Qwen3-0.6B/aitk/qwen3_qnn_config.json @@ -0,0 +1,157 @@ +{ + "input_model": { + "type": "HfModel", + "model_path": "Qwen/Qwen3-0.6B" + }, + "systems": { + "target_system": { + "type": "PythonEnvironment", + "python_environment_path": "/path/to/qnn/env/bin", + "accelerators": [ + { + "device": "npu", + "execution_providers": [ + "QNNExecutionProvider" + ] + } + ] + } + }, + "data_configs": [ + { + "name": "wikitext2_train_joined", + "type": "HuggingfaceContainer", + "load_dataset_config": { + "data_name": "wikitext", + "subset": "wikitext-2-raw-v1", + "split": "train" + }, + "pre_process_data_config": { + "strategy": "join", + "add_special_tokens": true, + "max_seq_len": 4096, + "max_samples": 128 + } + }, + { + "name": "wikitext2_train_act", + "type": "HuggingfaceContainer", + "load_dataset_config": { + "data_name": "wikitext", + "subset": "wikitext-2-raw-v1", + "split": "train" + }, + "pre_process_data_config": { + "strategy": "line-by-line", + "add_special_tokens": false, + "max_seq_len": 1024, + "max_samples": 256 + } + } + ], + "passes": { + "q": { + "type": "QuaRot" + }, + "cs": { + "type": "CaptureSplitInfo", + "num_splits": 1, + "unique_embeds_lm_head_splits": true + }, + "g": { + "type": "GptqModel", + "bits": 8, + "sym": true, + "group_size": -1, + "lm_head": true, + "device": "cuda", + "data_config": "wikitext2_train_joined", + "dynamic": { + "+:.*lm_head*": { + "bits": 4, + "sym": true, + "group_size": 32, + "desc_act": false + } + } + }, + "mb": { + "type": "ModelBuilder", + "precision": "int4", + "int4_block_size": 32, + "int4_accuracy_level": 4, + "int4_op_types_to_quantize": [ + "Gather" + ] + }, + "mq": { + "type": "MatMulNBitsToQDQ", + "use_int4": false, + "add_zero_point": true, + "nodes_to_exclude": [ + "/lm_head/MatMulNBits" + ], + "save_as_external_data": true + }, + "gs": { + "type": "GraphSurgeries", + "surgeries": [ + { + "surgeon": "RemoveRopeMultiCache" + }, + { + "surgeon": "AttentionMaskToSequenceLengths" + }, + { + "surgeon": "RemoveGidxFromMatMulNBits" + }, + { + "surgeon": "SimplifiedLayerNormToL2Norm" + } + ], + "save_as_external_data": true + }, + "sq": { + "type": "OnnxStaticQuantization", + "data_config": "wikitext2_train_act", + "activation_type": "uint16", + "precision": "uint8", + "calibration_providers": [ + "CUDAExecutionProvider" + ], + "quant_preprocess": true, + "op_types_to_exclude": [ + "GatherBlockQuantized", + "GroupQueryAttention", + "MatMulNBits" + ], + "save_as_external_data": true + }, + "sp": { + "type": "SplitModel" + }, + "st": { + "type": "StaticLLM", + "batch_size": 1, + "context_length": 64 + }, + "cb": { + "type": "EPContextBinaryGenerator", + "provider_options": { + "htp_performance_mode": "burst", + "htp_graph_finalization_optimization_mode": "3", + "soc_model": "60" + }, + "weight_sharing": true + }, + "cp": { + "type": "ComposeOnnxModels" + } + }, + "target": "target_system", + "log_severity_level": 1, + "output_dir": "model/qwen3_0.6B", + "cache_dir": "cache", + "no_artifacts": true, + "evaluate_input_model": false +} diff --git a/Qwen-Qwen3-0.6B/aitk/qwen3_qnn_config.json.config b/Qwen-Qwen3-0.6B/aitk/qwen3_qnn_config.json.config new file mode 100644 index 000000000..4ac71a9ab --- /dev/null +++ b/Qwen-Qwen3-0.6B/aitk/qwen3_qnn_config.json.config @@ -0,0 +1,226 @@ +{ + "$schema": "https://github.com/microsoft/olive-recipes/raw/refs/heads/main/.aitk/configs/config_schema.json", + "name": "Convert to Qualcomm NPU", + "oliveFile": "QNN/config.json", + "isLLM": true, + "debugInfo": { + "autoGenerated": true, + "useModelBuilder": "mb" + }, + "isQNNLLM": true, + "isGPURequired": true, + "runtimeOverwrite": { + "autoGenerated": true, + "pyEnvPath": "systems.target_system.python_environment_path", + "executeEp": "CUDAExecutionProvider", + "evaluateUsedInExecute": true + }, + "executeRuntimeFeatures": [ + "GptqModel", + "Qwen3" + ], + "pyEnvRuntimeFeatures": [ + "Qwen3" + ], + "runtime": { + "autoGenerated": true, + "name": "Evaluate on", + "type": "enum", + "displayNames": [ + "Qualcomm NPU" + ], + "path": "systems.target_system.accelerators.0.execution_providers.0", + "values": [ + "QNNExecutionProvider" + ], + "readOnly": false + }, + "optimizationPaths": [ + { + "path": "passes.sq.precision", + "name": "WeightType" + }, + { + "path": "passes.sq.activation_type", + "name": "ActivationType" + } + ], + "optimizationDefault": "w8 a16", + "sections": [ + { + "autoGenerated": true, + "name": "Convert", + "phase": "Conversion", + "parameters": [], + "toggle": { + "autoGenerated": true, + "name": "Convert to ONNX format", + "type": "bool", + "path": "passes.mb", + "actions": [ + [], + [] + ], + "readOnly": true + } + }, + { + "autoGenerated": true, + "name": "Quantize", + "phase": "Quantization", + "parameters": [ + { + "autoGenerated": true, + "name": "Activation Type", + "tags": [ + "ActivationType" + ], + "description": "Quantization data type of activation. \u2018Int8\u2019 for signed 8-bit integer, \u2018UInt8\u2019 for unsigned 8-bit integer etc.", + "descriptionLink": "https://onnxruntime.ai/docs/performance/model-optimizations/quantization.html", + "type": "enum", + "displayNames": [ + "Int8", + "UInt8", + "Int16", + "UInt16" + ], + "displayType": "RadioGroup", + "path": "passes.sq.activation_type", + "values": [ + "int8", + "uint8", + "int16", + "uint16" + ], + "template": { + "path": "passes.sq.activation_type", + "template": "ActivationType" + } + }, + { + "autoGenerated": true, + "name": "Weight Type", + "tags": [ + "WeightType" + ], + "description": "Data type for quantizing weights. \u2018Int8\u2019 for signed 8-bit integer, \u2018UInt8\u2019 for unsigned 8-bit integer etc.", + "descriptionLink": "https://onnxruntime.ai/docs/performance/model-optimizations/quantization.html", + "type": "enum", + "displayNames": [ + "Int8", + "UInt8", + "Int16", + "UInt16" + ], + "displayType": "RadioGroup", + "path": "passes.sq.precision", + "values": [ + "int8", + "uint8", + "int16", + "uint16" + ], + "template": { + "path": "passes.sq.precision", + "template": "WeightType" + } + }, + { + "autoGenerated": true, + "name": "Quantization Dataset", + "tags": [ + "QuantizationDataset" + ], + "type": "enum", + "path": "data_configs[1].load_dataset_config.data_name", + "values": [ + "wikitext" + ], + "template": { + "path": "data_configs[1].load_dataset_config.data_name", + "values": [ + "wikitext" + ], + "template": "QuantizationDataset" + } + }, + { + "autoGenerated": true, + "name": "Quantization Dataset Subset", + "tags": [ + "QuantizationDatasetSubset", + "DependsOnDataset" + ], + "type": "enum", + "path": "data_configs[1].load_dataset_config.subset", + "values": [ + "wikitext-103-raw-v1", + "wikitext-103-v1", + "wikitext-2-raw-v1", + "wikitext-2-v1" + ], + "template": { + "path": "data_configs[1].load_dataset_config.subset", + "values": [ + "wikitext-103-raw-v1", + "wikitext-103-v1", + "wikitext-2-raw-v1", + "wikitext-2-v1" + ], + "template": "QuantizationDatasetSubset" + } + }, + { + "autoGenerated": true, + "name": "Quantization Dataset Split", + "tags": [ + "QuantizationDatasetSplit", + "DependsOnDataset" + ], + "type": "enum", + "path": "data_configs[1].load_dataset_config.split", + "values": [ + "train", + "validation", + "test" + ], + "template": { + "path": "data_configs[1].load_dataset_config.split", + "template": "QuantizationDatasetSplit" + } + }, + { + "autoGenerated": true, + "name": "Quantization Dataset Sequence Length", + "type": "int", + "path": "data_configs[1].pre_process_data_config.max_seq_len", + "template": { + "path": "data_configs[1].pre_process_data_config.max_seq_len", + "template": "QuantizationDatasetLength" + } + }, + { + "autoGenerated": true, + "name": "Quantization Dataset Size", + "type": "int", + "path": "data_configs[1].pre_process_data_config.max_samples", + "template": { + "path": "data_configs[1].pre_process_data_config.max_samples", + "template": "QuantizationDatasetSize" + } + } + ], + "toggle": { + "autoGenerated": true, + "name": "Quantize model", + "type": "bool", + "path": "passes.mb", + "actions": [ + [], + [] + ], + "readOnly": true + } + } + ] +} diff --git a/Qwen-Qwen3-0.6B/aitk/requirements.txt b/Qwen-Qwen3-0.6B/aitk/requirements.txt new file mode 100644 index 000000000..e1a8fe619 --- /dev/null +++ b/Qwen-Qwen3-0.6B/aitk/requirements.txt @@ -0,0 +1,3 @@ +# This file will be installed together with Foundry Toolkit runtime requirements +# For the full requirements, see Foundry Toolkit +datasets diff --git a/Qwen-Qwen3-0.6B/aitk/winml.py b/Qwen-Qwen3-0.6B/aitk/winml.py new file mode 100644 index 000000000..5a26031d6 --- /dev/null +++ b/Qwen-Qwen3-0.6B/aitk/winml.py @@ -0,0 +1,58 @@ +# https://learn.microsoft.com/en-us/windows/ai/new-windows-ml/initialize-execution-providers?tabs=python#production-app-example + + +def _get_ep_paths(ep: str | None = None) -> dict[str, str]: + import ctypes + import importlib.util + from pathlib import Path + + # Locate onnxruntime package path without importing it first + ort_spec = importlib.util.find_spec("onnxruntime") + assert ort_spec is not None and ort_spec.origin is not None + ort_package_path = Path(ort_spec.origin).parent + ort_capi_dir = ort_package_path / "capi" + ort_dll_path = ort_capi_dir / "onnxruntime.dll" + + # Load the onnxruntime DLL because "C:\Windows\System32\onnxruntime.dll" may be exist and loaded first + ctypes.WinDLL(str(ort_dll_path)) + + # remove the msvcp140.dll from the winrt-runtime package. + # So it does not cause issues with other libraries. + from importlib import metadata + + site_packages_path = Path(str(metadata.distribution("winrt-runtime").locate_file(""))) + dll_path = site_packages_path / "winrt" / "msvcp140.dll" + if dll_path.exists(): + dll_path.unlink() + + import winui3.microsoft.windows.ai.machinelearning as winml + from winui3.microsoft.windows.applicationmodel.dynamicdependency.bootstrap import InitializeOptions, initialize + + eps = {} + with initialize(options=InitializeOptions.ON_NO_MATCH_SHOW_UI): + catalog = winml.ExecutionProviderCatalog.get_default() + providers = catalog.find_all_providers() + for provider in providers: + if ep is not None and provider.name != ep: + continue + result = provider.ensure_ready_async().get() + if result.status == winml.ExecutionProviderReadyResultState.SUCCESS: + eps[provider.name] = provider.library_path + else: + print( + f"Execution provider '{provider.name}' is unavailable. Status: {result.status}; reason: {result.diagnostic_text}; error code: {result.extended_error.value}" + ) + return eps + + +def register_execution_providers(ep: str | None = None): + paths = _get_ep_paths(ep) + + import onnxruntime_genai as og + + for item in paths.items(): + try: + og.register_execution_provider_library(item[0], item[1]) # pyright: ignore[reportAttributeAccessIssue] + print(f"Successfully registered execution provider {item[0]} from {item[1]}") + except Exception as e: + print(f"Failed to register execution provider {item[0]} from {item[1]}: {e}") From e17f59e39396f7f42ba39dfd30c6489179757708 Mon Sep 17 00:00:00 2001 From: David Fan <30608893+jiafatom@users.noreply.github.com> Date: Mon, 20 Jul 2026 13:32:14 -0700 Subject: [PATCH 03/10] Add mixed quantize configs and eval for gemma4 E2B (#515) Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> --- google-gemma-4-E2B-it/README.md | 68 +++++++++++++++++- google-gemma-4-E2B-it/cpu/mixed/audio.json | 19 +++++ google-gemma-4-E2B-it/cpu/mixed/export.json | 14 ++++ google-gemma-4-E2B-it/cpu/mixed/text.json | 16 +++++ google-gemma-4-E2B-it/cpu/mixed/vision.json | 19 +++++ google-gemma-4-E2B-it/cuda/mixed/audio.json | 22 ++++++ google-gemma-4-E2B-it/cuda/mixed/export.json | 18 +++++ google-gemma-4-E2B-it/cuda/mixed/text.json | 20 ++++++ google-gemma-4-E2B-it/cuda/mixed/vision.json | 22 ++++++ google-gemma-4-E2B-it/eval.py | 2 +- google-gemma-4-E2B-it/eval/README.md | 65 +++++++++++++++++ google-gemma-4-E2B-it/eval/ai2d_cpu.json | 54 +++++++++++++++ google-gemma-4-E2B-it/eval/ai2d_cuda.json | 69 +++++++++++++++++++ .../eval/fleurs_asr_cpu.json | 56 +++++++++++++++ .../eval/fleurs_asr_cuda.json | 69 +++++++++++++++++++ google-gemma-4-E2B-it/eval/mmlu_cpu.json | 29 ++++++++ google-gemma-4-E2B-it/eval/mmlu_cuda.json | 42 +++++++++++ google-gemma-4-E2B-it/inference.py | 2 +- google-gemma-4-E2B-it/info.yml | 18 +++++ 19 files changed, 620 insertions(+), 4 deletions(-) create mode 100644 google-gemma-4-E2B-it/cpu/mixed/audio.json create mode 100644 google-gemma-4-E2B-it/cpu/mixed/export.json create mode 100644 google-gemma-4-E2B-it/cpu/mixed/text.json create mode 100644 google-gemma-4-E2B-it/cpu/mixed/vision.json create mode 100644 google-gemma-4-E2B-it/cuda/mixed/audio.json create mode 100644 google-gemma-4-E2B-it/cuda/mixed/export.json create mode 100644 google-gemma-4-E2B-it/cuda/mixed/text.json create mode 100644 google-gemma-4-E2B-it/cuda/mixed/vision.json create mode 100644 google-gemma-4-E2B-it/eval/README.md create mode 100644 google-gemma-4-E2B-it/eval/ai2d_cpu.json create mode 100644 google-gemma-4-E2B-it/eval/ai2d_cuda.json create mode 100644 google-gemma-4-E2B-it/eval/fleurs_asr_cpu.json create mode 100644 google-gemma-4-E2B-it/eval/fleurs_asr_cuda.json create mode 100644 google-gemma-4-E2B-it/eval/mmlu_cpu.json create mode 100644 google-gemma-4-E2B-it/eval/mmlu_cuda.json diff --git a/google-gemma-4-E2B-it/README.md b/google-gemma-4-E2B-it/README.md index 50ccd7c9d..7fa82a4a2 100644 --- a/google-gemma-4-E2B-it/README.md +++ b/google-gemma-4-E2B-it/README.md @@ -36,6 +36,40 @@ Install ONNX Runtime GenAI: | `cuda/fp16/config.json` | `MobiusBuilder(fp16)` | `cuda/fp16/models` | | `cuda/int4/config.json` | `MobiusBuilder(fp16)` → `OnnxKQuantQuantization(bits=4, block=32)` | `cuda/int4/models` | +### Mixed quantization (separate text / vision / audio) + +These recipes split the model into components with per-component +quantization — int4 for the text decoder, int8 for vision and audio +encoders — for better accuracy vs. latency trade-offs. + +| Recipe | Pipeline | Output dir | +|---|---|---| +| `cpu/mixed/export.json` | `MobiusBuilder(fp32)` — export all components | `cpu/mixed/models` | +| `cpu/mixed/text.json` | `OnnxKQuantQuantization(int4, block=32)` — quantize decoder | `cpu/mixed/models/decoder` | +| `cpu/mixed/vision.json` | `OnnxBlockWiseRtnQuantization(int8, block=128)` — quantize vision encoder | `cpu/mixed/models/vision_encoder` | +| `cpu/mixed/audio.json` | `OnnxBlockWiseRtnQuantization(int8, block=128)` — quantize audio encoder | `cpu/mixed/models/audio_encoder` | +| `cuda/mixed/export.json` | `MobiusBuilder(fp16)` — export all components | `cuda/mixed/models` | +| `cuda/mixed/text.json` | `OnnxKQuantQuantization(int4, block=32)` — quantize decoder | `cuda/mixed/models/decoder` | +| `cuda/mixed/vision.json` | `OnnxBlockWiseRtnQuantization(int8, block=32)` — quantize vision encoder | `cuda/mixed/models/vision_encoder` | +| `cuda/mixed/audio.json` | `OnnxBlockWiseRtnQuantization(int8, block=32)` — quantize audio encoder | `cuda/mixed/models/audio_encoder` | + +**Run order**: export first, then text, vision, and audio (the latter three +can run in parallel): + +```bash +# CPU mixed +olive run --config cpu/mixed/export.json +olive run --config cpu/mixed/text.json +olive run --config cpu/mixed/vision.json +olive run --config cpu/mixed/audio.json + +# CUDA mixed +olive run --config cuda/mixed/export.json +olive run --config cuda/mixed/text.json +olive run --config cuda/mixed/vision.json +olive run --config cuda/mixed/audio.json +``` + K-Quant (Q4_K_M) is significantly faster with GPU acceleration — install `cupy-cuda12x` for a 19–51× speedup during quantization. @@ -83,18 +117,48 @@ python inference.py --variant int4 --prompt "Hello" # CUDA INT4 python inference.py --device gpu --variant int4 --prompt "Explain quantum computing" +# CUDA mixed (int4 decoder + int8 vision/audio) +python inference.py --device gpu --variant mixed --prompt "Explain quantum computing" + # Interactive mode -python inference.py --device gpu --variant int4 --interactive +python inference.py --device gpu --variant mixed --interactive ``` ## Evaluation +### MMLU (text) + +Run through Olive with the `LMEvaluator` configs in `eval/` (mixed model): + ```bash -# MMLU Pro (default 100 samples), CPU +olive run --config eval/mmlu_cpu.json # CPU +olive run --config eval/mmlu_cuda.json # CUDA +``` + +Or use the standalone script (also supports `--task`, `--limit`, and other variants): + +```bash +# MMLU (5-shot, default 100 samples), CPU python eval.py # CUDA INT4 python eval.py --device gpu --variant int4 + +# CUDA mixed +python eval.py --device gpu --variant mixed +``` + +### Vision — AI2D (exact_match) + +> **Note**: The `olive run` eval configs require Olive to support nested model +> layouts in the evaluator's genai_config.json discovery. Until then, use +> a custom evaluation script. + +### Audio — FLEURS ASR (WER) + +> **Note**: The `olive run` audio eval configs require Olive to add gemma4 +> as a supported model type in the speech evaluator. Until then, use a custom +> evaluation script. ``` ## References diff --git a/google-gemma-4-E2B-it/cpu/mixed/audio.json b/google-gemma-4-E2B-it/cpu/mixed/audio.json new file mode 100644 index 000000000..d47b6c48f --- /dev/null +++ b/google-gemma-4-E2B-it/cpu/mixed/audio.json @@ -0,0 +1,19 @@ +{ + "input_model": { + "type": "ONNXModel", + "model_path": "cpu/mixed/models/audio_encoder/model.onnx" + }, + "passes": { + "int8": { + "type": "OnnxBlockWiseRtnQuantization", + "bits": 8, + "block_size": 128, + "is_symmetric": true, + "accuracy_level": 4, + "save_as_external_data": true, + "external_data_name": "model.onnx.data" + } + }, + "no_artifacts": true, + "output_dir": "cpu/mixed/models/audio_encoder" +} diff --git a/google-gemma-4-E2B-it/cpu/mixed/export.json b/google-gemma-4-E2B-it/cpu/mixed/export.json new file mode 100644 index 000000000..95d9688a2 --- /dev/null +++ b/google-gemma-4-E2B-it/cpu/mixed/export.json @@ -0,0 +1,14 @@ +{ + "input_model": { + "type": "HfModel", + "model_path": "google/gemma-4-E2B-it" + }, + "passes": { + "export": { + "type": "MobiusBuilder", + "precision": "fp32" + } + }, + "no_artifacts": true, + "output_dir": "cpu/mixed/models" +} diff --git a/google-gemma-4-E2B-it/cpu/mixed/text.json b/google-gemma-4-E2B-it/cpu/mixed/text.json new file mode 100644 index 000000000..ebf96a466 --- /dev/null +++ b/google-gemma-4-E2B-it/cpu/mixed/text.json @@ -0,0 +1,16 @@ +{ + "input_model": { + "type": "ONNXModel", + "model_path": "cpu/mixed/models/decoder/model.onnx" + }, + "passes": { + "int4_quantize": { + "type": "OnnxKQuantQuantization", + "bits": 4, + "block_size": 32, + "save_as_external_data": true + } + }, + "no_artifacts": true, + "output_dir": "cpu/mixed/models/decoder" +} diff --git a/google-gemma-4-E2B-it/cpu/mixed/vision.json b/google-gemma-4-E2B-it/cpu/mixed/vision.json new file mode 100644 index 000000000..995bfdd1f --- /dev/null +++ b/google-gemma-4-E2B-it/cpu/mixed/vision.json @@ -0,0 +1,19 @@ +{ + "input_model": { + "type": "ONNXModel", + "model_path": "cpu/mixed/models/vision_encoder/model.onnx" + }, + "passes": { + "int8": { + "type": "OnnxBlockWiseRtnQuantization", + "bits": 8, + "block_size": 128, + "is_symmetric": true, + "accuracy_level": 4, + "save_as_external_data": true, + "external_data_name": "model.onnx.data" + } + }, + "no_artifacts": true, + "output_dir": "cpu/mixed/models/vision_encoder" +} diff --git a/google-gemma-4-E2B-it/cuda/mixed/audio.json b/google-gemma-4-E2B-it/cuda/mixed/audio.json new file mode 100644 index 000000000..43bfe27ee --- /dev/null +++ b/google-gemma-4-E2B-it/cuda/mixed/audio.json @@ -0,0 +1,22 @@ +{ + "input_model": { + "type": "ONNXModel", + "model_path": "cuda/mixed/models/audio_encoder/model.onnx" + }, + "passes": { + "int8": { + "type": "OnnxBlockWiseRtnQuantization", + "bits": 8, + "block_size": 32, + "is_symmetric": false, + "save_as_external_data": true, + "external_data_name": "model.onnx.data" + } + }, + "target": { + "type": "LocalSystem", + "accelerators": [{ "device": "gpu", "execution_providers": ["CUDAExecutionProvider"] }] + }, + "no_artifacts": true, + "output_dir": "cuda/mixed/models/audio_encoder" +} diff --git a/google-gemma-4-E2B-it/cuda/mixed/export.json b/google-gemma-4-E2B-it/cuda/mixed/export.json new file mode 100644 index 000000000..57a5eee9b --- /dev/null +++ b/google-gemma-4-E2B-it/cuda/mixed/export.json @@ -0,0 +1,18 @@ +{ + "input_model": { + "type": "HfModel", + "model_path": "google/gemma-4-E2B-it" + }, + "passes": { + "export": { + "type": "MobiusBuilder", + "precision": "fp16" + } + }, + "target": { + "type": "LocalSystem", + "accelerators": [{ "device": "gpu", "execution_providers": ["CUDAExecutionProvider"] }] + }, + "no_artifacts": true, + "output_dir": "cuda/mixed/models" +} diff --git a/google-gemma-4-E2B-it/cuda/mixed/text.json b/google-gemma-4-E2B-it/cuda/mixed/text.json new file mode 100644 index 000000000..33eb4500b --- /dev/null +++ b/google-gemma-4-E2B-it/cuda/mixed/text.json @@ -0,0 +1,20 @@ +{ + "input_model": { + "type": "ONNXModel", + "model_path": "cuda/mixed/models/decoder/model.onnx" + }, + "passes": { + "int4_quantize": { + "type": "OnnxKQuantQuantization", + "bits": 4, + "block_size": 32, + "save_as_external_data": true + } + }, + "target": { + "type": "LocalSystem", + "accelerators": [{ "device": "gpu", "execution_providers": ["CUDAExecutionProvider"] }] + }, + "no_artifacts": true, + "output_dir": "cuda/mixed/models/decoder" +} diff --git a/google-gemma-4-E2B-it/cuda/mixed/vision.json b/google-gemma-4-E2B-it/cuda/mixed/vision.json new file mode 100644 index 000000000..c98afc698 --- /dev/null +++ b/google-gemma-4-E2B-it/cuda/mixed/vision.json @@ -0,0 +1,22 @@ +{ + "input_model": { + "type": "ONNXModel", + "model_path": "cuda/mixed/models/vision_encoder/model.onnx" + }, + "passes": { + "int8": { + "type": "OnnxBlockWiseRtnQuantization", + "bits": 8, + "block_size": 32, + "is_symmetric": false, + "save_as_external_data": true, + "external_data_name": "model.onnx.data" + } + }, + "target": { + "type": "LocalSystem", + "accelerators": [{ "device": "gpu", "execution_providers": ["CUDAExecutionProvider"] }] + }, + "no_artifacts": true, + "output_dir": "cuda/mixed/models/vision_encoder" +} diff --git a/google-gemma-4-E2B-it/eval.py b/google-gemma-4-E2B-it/eval.py index 244449b11..ac7d38647 100644 --- a/google-gemma-4-E2B-it/eval.py +++ b/google-gemma-4-E2B-it/eval.py @@ -49,7 +49,7 @@ def main(): ) parser.add_argument( "--variant", - choices=["fp32", "fp16", "int4"], + choices=["fp32", "fp16", "int4", "mixed"], default=None, help="Model variant (cpu defaults to fp32, gpu defaults to int4)", ) diff --git a/google-gemma-4-E2B-it/eval/README.md b/google-gemma-4-E2B-it/eval/README.md new file mode 100644 index 000000000..676974fc9 --- /dev/null +++ b/google-gemma-4-E2B-it/eval/README.md @@ -0,0 +1,65 @@ +# Gemma 4 E2B Evaluation + +Evaluate quantized ONNX Gemma 4 E2B models on text, vision and audio benchmarks using Olive's built-in metrics. + +## Benchmarks + +- **MMLU** — Multi-task language understanding (`mmlu`, 5-shot, accuracy) +- **AI2D** — Science diagram multiple-choice QA (`lmms-lab/ai2d`, test split, exact_match) +- **FLEURS ASR** — Speech transcription accuracy (`google/fleurs`, en_us test split, WER + RTFx) + +## Prerequisites + +Ensure you have the ONNX models built using the mixed configs (see `../cpu/mixed/` or `../cuda/mixed/`). + +## Usage + +```bash +# Text evaluation (MMLU, 5-shot) +olive run --config mmlu_cpu.json # CPU +olive run --config mmlu_cuda.json # CUDA + +# Vision evaluation +olive run --config ai2d_cpu.json # CPU +olive run --config ai2d_cuda.json # CUDA + +# Audio evaluation (FLEURS ASR) +olive run --config fleurs_asr_cpu.json # CPU +olive run --config fleurs_asr_cuda.json # CUDA +``` + +## Configs + +### Text (MMLU, 5-shot) + +| Config | Model Path | Device | +|--------|-----------|--------| +| `mmlu_cpu.json` | `../cpu/mixed/models` | CPU | +| `mmlu_cuda.json` | `../cuda/mixed/models` | GPU (CUDA) | + +These use Olive's `LMEvaluator` with the `ortgenai` backend. Gemma 4 exports with +`past_present_share_buffer` enabled, but the decoder requires it **disabled** for correct +KV-cache handling during evaluation, so the configs pass +`"model_args": { "past_present_share_buffer": false }`. Both `model_args` and the +`past_present_share_buffer` override require Olive with +[microsoft/Olive#2569](https://github.com/microsoft/Olive/pull/2569). +The configs run 5-shot (`"num_fewshot": 5`) with the chat template applied +(`"apply_chat_template": true`, `"fewshot_as_multiturn": true`). +The default `limit` is 100 samples; remove it to run the full benchmark. +The configs also set `"sample_log_num": 100`, which writes the per-question prediction +vs. target for the first 100 samples to `sample_logs/mmlu_samples.jsonl` +for inspection/debugging (set to `0` to disable). + +### Vision (AI2D) + +| Config | Model Path | Device | +|--------|-----------|--------| +| `ai2d_cpu.json` | `../cpu/mixed/models` | CPU | +| `ai2d_cuda.json` | `../cuda/mixed/models` | GPU (CUDA) | + +### Audio (FLEURS ASR) + +| Config | Model Path | Device | +|--------|-----------|--------| +| `fleurs_asr_cpu.json` | `../cpu/mixed/models/decoder` | CPU | +| `fleurs_asr_cuda.json` | `../cuda/mixed/models/decoder` | GPU (CUDA) | diff --git a/google-gemma-4-E2B-it/eval/ai2d_cpu.json b/google-gemma-4-E2B-it/eval/ai2d_cpu.json new file mode 100644 index 000000000..059a92b45 --- /dev/null +++ b/google-gemma-4-E2B-it/eval/ai2d_cpu.json @@ -0,0 +1,54 @@ +{ + "input_model": { + "type": "OnnxModel", + "model_path": "../cpu/mixed/models", + "onnx_file_name": "decoder/model.onnx" + }, + "data_configs": [ + { + "name": "ai2d_test", + "type": "HuggingfaceContainer", + "load_dataset_config": { + "type": "huggingface_dataset", + "params": { + "data_name": "lmms-lab/ai2d", + "split": "test" + } + }, + "pre_process_data_config": { + "type": "vision_vqa_pre_process", + "params": { + "task": "vision-vqa", + "image_col": "image", + "question_col": "question", + "answer_col": "answer", + "options_col": "options", + "system_prompt": "You are a concise multiple-choice answering assistant. When given a question with numbered options, respond with ONLY a single digit (1, 2, 3, or 4). Do not include any explanation, reasoning, or other text \u2014 just the digit.", + "limit": 100 + } + } + } + ], + "evaluators": { + "common_evaluator": { + "metrics": [ + { + "name": "vision_accuracy", + "type": "accuracy", + "data_config": "ai2d_test", + "sub_types": [ + { + "name": "exact_match", + "priority": 1, + "higher_is_better": true + } + ] + } + ] + } + }, + "engine": { + "evaluate_input_model": true, + "evaluator": "common_evaluator" + } +} diff --git a/google-gemma-4-E2B-it/eval/ai2d_cuda.json b/google-gemma-4-E2B-it/eval/ai2d_cuda.json new file mode 100644 index 000000000..2c023f7c4 --- /dev/null +++ b/google-gemma-4-E2B-it/eval/ai2d_cuda.json @@ -0,0 +1,69 @@ +{ + "input_model": { + "type": "OnnxModel", + "model_path": "../cuda/mixed/models", + "onnx_file_name": "decoder/model.onnx" + }, + "data_configs": [ + { + "name": "ai2d_test", + "type": "HuggingfaceContainer", + "load_dataset_config": { + "type": "huggingface_dataset", + "params": { + "data_name": "lmms-lab/ai2d", + "split": "test" + } + }, + "pre_process_data_config": { + "type": "vision_vqa_pre_process", + "params": { + "task": "vision-vqa", + "image_col": "image", + "question_col": "question", + "answer_col": "answer", + "options_col": "options", + "system_prompt": "You are a concise multiple-choice answering assistant. When given a question with numbered options, respond with ONLY a single digit (1, 2, 3, or 4). Do not include any explanation, reasoning, or other text \u2014 just the digit.", + "limit": 100 + } + } + } + ], + "evaluators": { + "common_evaluator": { + "metrics": [ + { + "name": "vision_accuracy", + "type": "accuracy", + "data_config": "ai2d_test", + "sample_log_num": 100, + "sample_log_dir": "sample_logs", + "sub_types": [ + { + "name": "exact_match", + "priority": 1, + "higher_is_better": true + } + ] + } + ] + } + }, + "engine": { + "evaluate_input_model": true, + "evaluator": "common_evaluator", + "host": "local_system", + "target": "local_system" + }, + "systems": { + "local_system": { + "type": "LocalSystem", + "accelerators": [ + { + "device": "gpu", + "execution_providers": ["CUDAExecutionProvider"] + } + ] + } + } +} diff --git a/google-gemma-4-E2B-it/eval/fleurs_asr_cpu.json b/google-gemma-4-E2B-it/eval/fleurs_asr_cpu.json new file mode 100644 index 000000000..aaaaab670 --- /dev/null +++ b/google-gemma-4-E2B-it/eval/fleurs_asr_cpu.json @@ -0,0 +1,56 @@ +{ + "input_model": { + "type": "OnnxModel", + "model_path": "../cpu/mixed/models/decoder" + }, + "data_configs": [ + { + "name": "fleurs_en_us", + "type": "HuggingfaceContainer", + "load_dataset_config": { + "type": "huggingface_dataset", + "params": { + "data_name": "google/fleurs", + "subset": "en_us", + "split": "test" + } + }, + "pre_process_data_config": { + "type": "speech_transcription_pre_process", + "params": { + "audio_col": "audio", + "text_col": "transcription", + "sample_rate": 16000, + "max_samples": 64 + } + } + } + ], + "evaluators": { + "common_evaluator": { + "metrics": [ + { + "name": "speech_accuracy", + "type": "accuracy", + "data_config": "fleurs_en_us", + "sub_types": [ + { + "name": "wer", + "priority": 1, + "higher_is_better": false + }, + { + "name": "rtfx", + "priority": 2, + "higher_is_better": true + } + ] + } + ] + } + }, + "engine": { + "evaluate_input_model": true, + "evaluator": "common_evaluator" + } +} diff --git a/google-gemma-4-E2B-it/eval/fleurs_asr_cuda.json b/google-gemma-4-E2B-it/eval/fleurs_asr_cuda.json new file mode 100644 index 000000000..d0417549c --- /dev/null +++ b/google-gemma-4-E2B-it/eval/fleurs_asr_cuda.json @@ -0,0 +1,69 @@ +{ + "input_model": { + "type": "OnnxModel", + "model_path": "../cuda/mixed/models/decoder" + }, + "data_configs": [ + { + "name": "fleurs_en_us", + "type": "HuggingfaceContainer", + "load_dataset_config": { + "type": "huggingface_dataset", + "params": { + "data_name": "google/fleurs", + "subset": "en_us", + "split": "test" + } + }, + "pre_process_data_config": { + "type": "speech_transcription_pre_process", + "params": { + "audio_col": "audio", + "text_col": "transcription", + "sample_rate": 16000, + "max_samples": 64 + } + } + } + ], + "evaluators": { + "common_evaluator": { + "metrics": [ + { + "name": "speech_accuracy", + "type": "accuracy", + "data_config": "fleurs_en_us", + "sub_types": [ + { + "name": "wer", + "priority": 1, + "higher_is_better": false + }, + { + "name": "rtfx", + "priority": 2, + "higher_is_better": true + } + ] + } + ] + } + }, + "engine": { + "evaluate_input_model": true, + "evaluator": "common_evaluator", + "host": "local_system", + "target": "local_system" + }, + "systems": { + "local_system": { + "type": "LocalSystem", + "accelerators": [ + { + "device": "gpu", + "execution_providers": ["CUDAExecutionProvider"] + } + ] + } + } +} diff --git a/google-gemma-4-E2B-it/eval/mmlu_cpu.json b/google-gemma-4-E2B-it/eval/mmlu_cpu.json new file mode 100644 index 000000000..9b0668976 --- /dev/null +++ b/google-gemma-4-E2B-it/eval/mmlu_cpu.json @@ -0,0 +1,29 @@ +{ + "input_model": { + "type": "OnnxModel", + "model_path": "../cpu/mixed/models", + "onnx_file_name": "decoder/model.onnx" + }, + "evaluators": { + "mmlu_evaluator": { + "type": "LMEvaluator", + "tasks": ["mmlu"], + "model_class": "ortgenai", + "batch_size": 1, + "max_length": 4096, + "limit": 100, + "bootstrap_iters": 0, + "sample_log_num": 100, + "num_fewshot": 5, + "apply_chat_template": true, + "fewshot_as_multiturn": true, + "model_args": { + "past_present_share_buffer": false + } + } + }, + "engine": { + "evaluate_input_model": true, + "evaluator": "mmlu_evaluator" + } +} diff --git a/google-gemma-4-E2B-it/eval/mmlu_cuda.json b/google-gemma-4-E2B-it/eval/mmlu_cuda.json new file mode 100644 index 000000000..e03f43721 --- /dev/null +++ b/google-gemma-4-E2B-it/eval/mmlu_cuda.json @@ -0,0 +1,42 @@ +{ + "input_model": { + "type": "OnnxModel", + "model_path": "../cuda/mixed/models", + "onnx_file_name": "decoder/model.onnx" + }, + "evaluators": { + "mmlu_evaluator": { + "type": "LMEvaluator", + "tasks": ["mmlu"], + "model_class": "ortgenai", + "batch_size": 1, + "max_length": 4096, + "limit": 100, + "bootstrap_iters": 0, + "sample_log_num": 100, + "num_fewshot": 5, + "apply_chat_template": true, + "fewshot_as_multiturn": true, + "model_args": { + "past_present_share_buffer": false + } + } + }, + "engine": { + "evaluate_input_model": true, + "evaluator": "mmlu_evaluator", + "host": "local_system", + "target": "local_system" + }, + "systems": { + "local_system": { + "type": "LocalSystem", + "accelerators": [ + { + "device": "gpu", + "execution_providers": ["CUDAExecutionProvider"] + } + ] + } + } +} diff --git a/google-gemma-4-E2B-it/inference.py b/google-gemma-4-E2B-it/inference.py index 18eb60dc9..8d3c16ff7 100644 --- a/google-gemma-4-E2B-it/inference.py +++ b/google-gemma-4-E2B-it/inference.py @@ -115,7 +115,7 @@ def interactive_mode(model: og.Model, tokenizer: og.Tokenizer, max_length: int): def main(): parser = argparse.ArgumentParser(description="Gemma 4 ORT GenAI Inference") parser.add_argument("--device", choices=["cpu", "gpu"], default="cpu") - parser.add_argument("--variant", choices=["fp32", "fp16", "int4"], default=None) + parser.add_argument("--variant", choices=["fp32", "fp16", "int4", "mixed"], default=None) parser.add_argument("--model-path", default=None, help="Override model directory") parser.add_argument("--prompt", type=str, default=None, help="Text prompt") parser.add_argument("--system-prompt", type=str, default=None, help="System prompt") diff --git a/google-gemma-4-E2B-it/info.yml b/google-gemma-4-E2B-it/info.yml index 8a317ea57..7b99142fe 100644 --- a/google-gemma-4-E2B-it/info.yml +++ b/google-gemma-4-E2B-it/info.yml @@ -24,6 +24,15 @@ recipes: devices: - cpu eps: CPUExecutionProvider + - name: gemma4-e2b-mixed-cpu + file: + - cpu/mixed/export.json + - cpu/mixed/text.json + - cpu/mixed/vision.json + - cpu/mixed/audio.json + devices: + - cpu + eps: CPUExecutionProvider - name: gemma4-e2b-fp16-cuda file: cuda/fp16/config.json devices: @@ -34,3 +43,12 @@ recipes: devices: - gpu eps: CUDAExecutionProvider + - name: gemma4-e2b-mixed-cuda + file: + - cuda/mixed/export.json + - cuda/mixed/text.json + - cuda/mixed/vision.json + - cuda/mixed/audio.json + devices: + - gpu + eps: CUDAExecutionProvider From ebd60df3bb94412fc94d77a8e16ebdff09fee86c Mon Sep 17 00:00:00 2001 From: Anatol Liu Date: Tue, 21 Jul 2026 14:46:42 -0400 Subject: [PATCH 04/10] nvidia-nemotron-asr-streaming-multilingual-0.6b: Export FP16 encoder when specifying NvTensorRtRtx execution provider (#530) --- .../{src => NvTensorRtRtx}/.gitignore | 0 .../NvTensorRtRtx/README.md | 46 +++ .../{src => NvTensorRtRtx}/__init__.py | 0 .../NvTensorRtRtx/info.yaml | 19 ++ .../nemotron_decoder_fp16_trtrtx.json | 67 ++++ .../nemotron_encoder_fp16_trtrtx.json | 52 +++ .../nemotron_joint_fp16_trtrtx.json | 57 ++++ .../nemotron_model_load.py | 53 ++- .../NvTensorRtRtx/optimize.py | 314 +++++++++++++++++ .../{src => NvTensorRtRtx}/requirements.txt | 0 .../README.md | 13 +- .../cpu/.gitignore | 14 + .../cpu/README.md | 64 ++++ .../cpu/__init__.py | 0 .../cpu/info.yaml | 19 ++ .../nemotron_decoder_fp32_cpu.json | 2 +- .../nemotron_encoder_int4_cpu.json | 2 +- .../nemotron_encoder_int8_cpu.json | 2 +- .../{src => cpu}/nemotron_joint_fp32_cpu.json | 2 +- .../cpu/nemotron_model_load.py | 310 +++++++++++++++++ .../{src => cpu}/optimize.py | 81 +++-- .../cpu/requirements.txt | 11 + .../scripts/README.md | 44 +-- .../scripts/export_tokenizer.py | 4 +- .../scripts/test_optimize.py | 320 ++++++++++++++++++ .../src/README.md | 93 ----- .../src/info.yaml | 22 -- 27 files changed, 1406 insertions(+), 205 deletions(-) rename nvidia-nemotron-asr-streaming-multilingual-0.6b/{src => NvTensorRtRtx}/.gitignore (100%) create mode 100644 nvidia-nemotron-asr-streaming-multilingual-0.6b/NvTensorRtRtx/README.md rename nvidia-nemotron-asr-streaming-multilingual-0.6b/{src => NvTensorRtRtx}/__init__.py (100%) create mode 100644 nvidia-nemotron-asr-streaming-multilingual-0.6b/NvTensorRtRtx/info.yaml create mode 100644 nvidia-nemotron-asr-streaming-multilingual-0.6b/NvTensorRtRtx/nemotron_decoder_fp16_trtrtx.json create mode 100644 nvidia-nemotron-asr-streaming-multilingual-0.6b/NvTensorRtRtx/nemotron_encoder_fp16_trtrtx.json create mode 100644 nvidia-nemotron-asr-streaming-multilingual-0.6b/NvTensorRtRtx/nemotron_joint_fp16_trtrtx.json rename nvidia-nemotron-asr-streaming-multilingual-0.6b/{src => NvTensorRtRtx}/nemotron_model_load.py (83%) create mode 100644 nvidia-nemotron-asr-streaming-multilingual-0.6b/NvTensorRtRtx/optimize.py rename nvidia-nemotron-asr-streaming-multilingual-0.6b/{src => NvTensorRtRtx}/requirements.txt (100%) create mode 100644 nvidia-nemotron-asr-streaming-multilingual-0.6b/cpu/.gitignore create mode 100644 nvidia-nemotron-asr-streaming-multilingual-0.6b/cpu/README.md create mode 100644 nvidia-nemotron-asr-streaming-multilingual-0.6b/cpu/__init__.py create mode 100644 nvidia-nemotron-asr-streaming-multilingual-0.6b/cpu/info.yaml rename nvidia-nemotron-asr-streaming-multilingual-0.6b/{src => cpu}/nemotron_decoder_fp32_cpu.json (96%) rename nvidia-nemotron-asr-streaming-multilingual-0.6b/{src => cpu}/nemotron_encoder_int4_cpu.json (96%) rename nvidia-nemotron-asr-streaming-multilingual-0.6b/{src => cpu}/nemotron_encoder_int8_cpu.json (96%) rename nvidia-nemotron-asr-streaming-multilingual-0.6b/{src => cpu}/nemotron_joint_fp32_cpu.json (95%) create mode 100644 nvidia-nemotron-asr-streaming-multilingual-0.6b/cpu/nemotron_model_load.py rename nvidia-nemotron-asr-streaming-multilingual-0.6b/{src => cpu}/optimize.py (85%) create mode 100644 nvidia-nemotron-asr-streaming-multilingual-0.6b/cpu/requirements.txt create mode 100644 nvidia-nemotron-asr-streaming-multilingual-0.6b/scripts/test_optimize.py delete mode 100644 nvidia-nemotron-asr-streaming-multilingual-0.6b/src/README.md delete mode 100644 nvidia-nemotron-asr-streaming-multilingual-0.6b/src/info.yaml diff --git a/nvidia-nemotron-asr-streaming-multilingual-0.6b/src/.gitignore b/nvidia-nemotron-asr-streaming-multilingual-0.6b/NvTensorRtRtx/.gitignore similarity index 100% rename from nvidia-nemotron-asr-streaming-multilingual-0.6b/src/.gitignore rename to nvidia-nemotron-asr-streaming-multilingual-0.6b/NvTensorRtRtx/.gitignore diff --git a/nvidia-nemotron-asr-streaming-multilingual-0.6b/NvTensorRtRtx/README.md b/nvidia-nemotron-asr-streaming-multilingual-0.6b/NvTensorRtRtx/README.md new file mode 100644 index 000000000..cd08c7254 --- /dev/null +++ b/nvidia-nemotron-asr-streaming-multilingual-0.6b/NvTensorRtRtx/README.md @@ -0,0 +1,46 @@ +# Nemotron 3.5 ASR Streaming Multilingual 0.6B — TRT-RTX + +This recipe exports +**nvidia/NVIDIA-Nemotron-3.5-ASR-Streaming-Multilingual-0.6b** for the +NvTensorRtRtx execution provider. The encoder, decoder, and joint network use +homogeneous FP16 inputs, outputs, and internal compute at ONNX opset 23. + +Silero VAD is omitted because it is not supported by TRT-RTX. + +## Setup + +From the repository root: + +```bash +python -m venv .venv +source .venv/bin/activate +pip install -r nvidia-nemotron-asr-streaming-multilingual-0.6b/NvTensorRtRtx/requirements.txt +``` + +## Run + +From the recipe directory: + +```bash +cd nvidia-nemotron-asr-streaming-multilingual-0.6b + +python NvTensorRtRtx/optimize.py + +# Custom output directory +python NvTensorRtRtx/optimize.py --output-dir build/multilingual_onnx_fp16 +``` + +Individual Olive configurations can also be run directly: + +```bash +python -m olive run --config NvTensorRtRtx/nemotron_encoder_fp16_trtrtx.json +python -m olive run --config NvTensorRtRtx/nemotron_decoder_fp16_trtrtx.json +python -m olive run --config NvTensorRtRtx/nemotron_joint_fp16_trtrtx.json +``` + +## Output + +The default output directory is `NvTensorRtRtx/build/onnx_models_fp16/`. +It contains FP16 `encoder.onnx`, `decoder.onnx`, and `joint.onnx` +models, tokenizer files, `genai_config.json`, and +`audio_processor_config.json`. diff --git a/nvidia-nemotron-asr-streaming-multilingual-0.6b/src/__init__.py b/nvidia-nemotron-asr-streaming-multilingual-0.6b/NvTensorRtRtx/__init__.py similarity index 100% rename from nvidia-nemotron-asr-streaming-multilingual-0.6b/src/__init__.py rename to nvidia-nemotron-asr-streaming-multilingual-0.6b/NvTensorRtRtx/__init__.py diff --git a/nvidia-nemotron-asr-streaming-multilingual-0.6b/NvTensorRtRtx/info.yaml b/nvidia-nemotron-asr-streaming-multilingual-0.6b/NvTensorRtRtx/info.yaml new file mode 100644 index 000000000..548485638 --- /dev/null +++ b/nvidia-nemotron-asr-streaming-multilingual-0.6b/NvTensorRtRtx/info.yaml @@ -0,0 +1,19 @@ +name: nvidia-nemotron-asr-streaming-multilingual-0.6b +provider: NVIDIA +model_id: nvidia/NVIDIA-Nemotron-3.5-ASR-Streaming-Multilingual-0.6b +task: automatic-speech-recognition +framework: ONNX Runtime +execution_provider: NvTensorRTRTXExecutionProvider +summary: > + TRT-RTX recipe for exporting the multilingual NVIDIA Nemotron 3.5 ASR + Streaming RNNT encoder, decoder, and joint models with homogeneous FP16 + inputs, outputs, and internal compute at ONNX opset 23. + +artifacts: + - NvTensorRtRtx/nemotron_encoder_fp16_trtrtx.json + - NvTensorRtRtx/nemotron_decoder_fp16_trtrtx.json + - NvTensorRtRtx/nemotron_joint_fp16_trtrtx.json + - NvTensorRtRtx/optimize.py + +scripts: + - scripts/export_tokenizer.py diff --git a/nvidia-nemotron-asr-streaming-multilingual-0.6b/NvTensorRtRtx/nemotron_decoder_fp16_trtrtx.json b/nvidia-nemotron-asr-streaming-multilingual-0.6b/NvTensorRtRtx/nemotron_decoder_fp16_trtrtx.json new file mode 100644 index 000000000..4d059ed40 --- /dev/null +++ b/nvidia-nemotron-asr-streaming-multilingual-0.6b/NvTensorRtRtx/nemotron_decoder_fp16_trtrtx.json @@ -0,0 +1,67 @@ +{ + "input_model": { + "type": "PyTorchModel", + "model_path": "nvidia/NVIDIA-Nemotron-3.5-ASR-Streaming-Multilingual-0.6b", + "model_loader": "decoder_fp16_model_loader", + "model_script": "NvTensorRtRtx/nemotron_model_load.py", + "io_config": { + "input_names": [ + "targets", + "h_in", + "c_in" + ], + "output_names": [ + "decoder_output", + "h_out", + "c_out" + ], + "dynamic_axes": { + "targets": { + "0": "batch", + "1": "target_len" + }, + "h_in": { + "1": "batch" + }, + "c_in": { + "1": "batch" + }, + "decoder_output": { + "0": "batch", + "2": "target_len" + }, + "h_out": { + "1": "batch" + }, + "c_out": { + "1": "batch" + } + } + }, + "dummy_inputs_func": "decoder_fp16_dummy_inputs" + }, + "systems": { + "local_system": { + "type": "LocalSystem", + "accelerators": [ + { + "device": "gpu", + "execution_providers": [ + "NvTensorRTRTXExecutionProvider" + ] + } + ] + } + }, + "passes": { + "convert": { + "type": "OnnxConversion", + "target_opset": 23, + "save_as_external_data": true, + "external_data_name": "decoder.onnx.data" + } + }, + "target": "local_system", + "output_dir": "build/onnx_models_trtrtx_fp16/decoder.onnx", + "no_artifacts": true +} diff --git a/nvidia-nemotron-asr-streaming-multilingual-0.6b/NvTensorRtRtx/nemotron_encoder_fp16_trtrtx.json b/nvidia-nemotron-asr-streaming-multilingual-0.6b/NvTensorRtRtx/nemotron_encoder_fp16_trtrtx.json new file mode 100644 index 000000000..66ae5c898 --- /dev/null +++ b/nvidia-nemotron-asr-streaming-multilingual-0.6b/NvTensorRtRtx/nemotron_encoder_fp16_trtrtx.json @@ -0,0 +1,52 @@ +{ + "input_model": { + "type": "PyTorchModel", + "model_path": "nvidia/NVIDIA-Nemotron-3.5-ASR-Streaming-Multilingual-0.6b", + "model_loader": "encoder_fp16_model_loader", + "model_script": "NvTensorRtRtx/nemotron_model_load.py", + "io_config": { + "input_names": [ + "audio_signal", + "length", + "cache_last_channel", + "cache_last_time", + "cache_last_channel_len", + "lang_id" + ], + "output_names": [ + "outputs", + "encoded_lengths", + "cache_last_channel_next", + "cache_last_time_next", + "cache_last_channel_len_next" + ] + }, + "dummy_inputs_func": "encoder_fp16_dummy_inputs" + }, + "systems": { + "local_system": { + "type": "LocalSystem", + "accelerators": [ + { + "device": "gpu", + "execution_providers": [ + "NvTensorRTRTXExecutionProvider" + ] + } + ] + } + }, + "passes": { + "convert": { + "type": "OnnxConversion", + "target_opset": 23, + "dynamic": false, + "use_dynamo_exporter": true, + "save_as_external_data": true, + "external_data_name": "encoder.onnx.data" + } + }, + "target": "local_system", + "output_dir": "build/onnx_models_trtrtx_fp16/encoder.onnx", + "no_artifacts": true +} diff --git a/nvidia-nemotron-asr-streaming-multilingual-0.6b/NvTensorRtRtx/nemotron_joint_fp16_trtrtx.json b/nvidia-nemotron-asr-streaming-multilingual-0.6b/NvTensorRtRtx/nemotron_joint_fp16_trtrtx.json new file mode 100644 index 000000000..7912663ac --- /dev/null +++ b/nvidia-nemotron-asr-streaming-multilingual-0.6b/NvTensorRtRtx/nemotron_joint_fp16_trtrtx.json @@ -0,0 +1,57 @@ +{ + "input_model": { + "type": "PyTorchModel", + "model_path": "nvidia/NVIDIA-Nemotron-3.5-ASR-Streaming-Multilingual-0.6b", + "model_loader": "joint_fp16_model_loader", + "model_script": "NvTensorRtRtx/nemotron_model_load.py", + "io_config": { + "input_names": [ + "encoder_output", + "decoder_output" + ], + "output_names": [ + "joint_output" + ], + "dynamic_axes": { + "encoder_output": { + "0": "batch", + "1": "time" + }, + "decoder_output": { + "0": "batch", + "1": "target_len" + }, + "joint_output": { + "0": "batch", + "1": "time", + "2": "target_len" + } + } + }, + "dummy_inputs_func": "joint_fp16_dummy_inputs" + }, + "systems": { + "local_system": { + "type": "LocalSystem", + "accelerators": [ + { + "device": "gpu", + "execution_providers": [ + "NvTensorRTRTXExecutionProvider" + ] + } + ] + } + }, + "passes": { + "convert": { + "type": "OnnxConversion", + "target_opset": 23, + "save_as_external_data": true, + "external_data_name": "joint.onnx.data" + } + }, + "target": "local_system", + "output_dir": "build/onnx_models_trtrtx_fp16/joint.onnx", + "no_artifacts": true +} diff --git a/nvidia-nemotron-asr-streaming-multilingual-0.6b/src/nemotron_model_load.py b/nvidia-nemotron-asr-streaming-multilingual-0.6b/NvTensorRtRtx/nemotron_model_load.py similarity index 83% rename from nvidia-nemotron-asr-streaming-multilingual-0.6b/src/nemotron_model_load.py rename to nvidia-nemotron-asr-streaming-multilingual-0.6b/NvTensorRtRtx/nemotron_model_load.py index 6e11da441..7062b34eb 100644 --- a/nvidia-nemotron-asr-streaming-multilingual-0.6b/src/nemotron_model_load.py +++ b/nvidia-nemotron-asr-streaming-multilingual-0.6b/NvTensorRtRtx/nemotron_model_load.py @@ -72,6 +72,11 @@ def _get_streaming_shapes(): "chunk_encoded_frames": chunk_encoded_frames, } +def configure_encoder_for_sdpa(encoder): + for layer in getattr(encoder, "layers", []) or []: + attention = getattr(layer, "self_attn", None) + if attention is not None and hasattr(attention, "use_pytorch_sdpa"): + attention.use_pytorch_sdpa = True def _load_nemo_model(model_name=MODEL_NAME): """Load the NeMo ASR model (shared across loaders). @@ -103,7 +108,7 @@ def _load_nemo_model(model_name=MODEL_NAME): ) nemo_path = hf_hub_download(repo_id=model_name, filename=nemo_files[0]) - asr_model = nemo_asr.models.ASRModel.restore_from(nemo_path) + asr_model = nemo_asr.models.ASRModel.restore_from(nemo_path, map_location="cpu") asr_model = asr_model.cpu() asr_model.eval() return asr_model @@ -182,6 +187,14 @@ def encoder_dummy_inputs(model): ) +def encoder_fp16_dummy_inputs(model): + """Generate FP16 floating-point inputs for the TRT-RTX encoder export.""" + inputs = list(encoder_dummy_inputs(model)) + for index in (0, 2, 3): + inputs[index] = inputs[index].to(dtype=torch.float16) + return tuple(inputs) + + # --------------------------------------------------------------------------- # Decoder (stateful LSTM) # --------------------------------------------------------------------------- @@ -214,6 +227,23 @@ def decoder_model_loader(model_name): return wrapper +def encoder_fp16_model_loader(model_name): + """Load the streaming encoder wrapper in FP16 for NvTensorRtRtx export.""" + wrapper = encoder_model_loader(model_name) + configure_encoder_for_sdpa(wrapper.enc) + wrapper.to(dtype=torch.float16) + wrapper.eval() + return wrapper + + +def decoder_fp16_model_loader(model_name): + """Load the stateful decoder wrapper in FP16 for NvTensorRtRtx export.""" + wrapper = decoder_model_loader(model_name) + wrapper.to(dtype=torch.float16) + wrapper.eval() + return wrapper + + def decoder_dummy_inputs(model): """Generate dummy inputs for ONNX export of the stateful decoder.""" batch = 1 @@ -224,6 +254,14 @@ def decoder_dummy_inputs(model): ) +def decoder_fp16_dummy_inputs(model): + """Generate INT64 targets and FP16 LSTM states for NvTensorRtRtx export.""" + inputs = list(decoder_dummy_inputs(model)) + inputs[1] = inputs[1].to(dtype=torch.float16) + inputs[2] = inputs[2].to(dtype=torch.float16) + return tuple(inputs) + + # --------------------------------------------------------------------------- # Joint network # --------------------------------------------------------------------------- @@ -250,6 +288,14 @@ def joint_model_loader(model_name): return wrapper +def joint_fp16_model_loader(model_name): + """Load the joint wrapper in FP16 for NvTensorRtRtx export.""" + wrapper = joint_model_loader(model_name) + wrapper.to(dtype=torch.float16) + wrapper.eval() + return wrapper + + def joint_dummy_inputs(model): """Generate dummy inputs for ONNX export of the joint network.""" batch = 1 @@ -257,3 +303,8 @@ def joint_dummy_inputs(model): torch.randn(batch, 1, D_MODEL), torch.randn(batch, 1, DECODER_HIDDEN), ) + + +def joint_fp16_dummy_inputs(model): + """Generate FP16 encoder and decoder inputs for NvTensorRtRtx export.""" + return tuple(value.to(dtype=torch.float16) for value in joint_dummy_inputs(model)) diff --git a/nvidia-nemotron-asr-streaming-multilingual-0.6b/NvTensorRtRtx/optimize.py b/nvidia-nemotron-asr-streaming-multilingual-0.6b/NvTensorRtRtx/optimize.py new file mode 100644 index 000000000..f44723b32 --- /dev/null +++ b/nvidia-nemotron-asr-streaming-multilingual-0.6b/NvTensorRtRtx/optimize.py @@ -0,0 +1,314 @@ +"""End-to-end TRT-RTX optimization pipeline for Nemotron Speech Streaming. + +The encoder, decoder, and joint network are exported with homogeneous FP16 +inputs, outputs, and internal compute at ONNX opset 23. The pipeline also +exports the tokenizer and generates runtime configuration files. Silero VAD +is omitted because it is not supported by TRT-RTX. + +Usage: + python NvTensorRtRtx/optimize.py + + # Or run individual Olive configs: + python -m olive run --config NvTensorRtRtx/nemotron_encoder_fp16_trtrtx.json + python -m olive run --config NvTensorRtRtx/nemotron_decoder_fp16_trtrtx.json + python -m olive run --config NvTensorRtRtx/nemotron_joint_fp16_trtrtx.json +""" + +import argparse +import json +import subprocess +import sys +import tempfile +from pathlib import Path + +# Ensure the recipe root is on sys.path so `from NvTensorRtRtx.nemotron_model_load import ...` works +# regardless of where the script is invoked from. +_SCRIPT_DIR = Path(__file__).resolve().parent +_RECIPE_ROOT = _SCRIPT_DIR.parent +if str(_RECIPE_ROOT) not in sys.path: + sys.path.insert(0, str(_RECIPE_ROOT)) + +_TOKENIZER_SCRIPT = _RECIPE_ROOT / "scripts" / "export_tokenizer.py" + +DEFAULT_OUTPUT_DIR = "build/onnx_models_fp16" + + +def _resolve(path: str) -> Path: + """Resolve a path relative to the NvTensorRtRtx/ directory.""" + p = Path(path) + return p if p.is_absolute() else _SCRIPT_DIR / p + + +def _run_olive_pipeline(config_name: str, output_dir: str, output_subdir: str, model_path: str = None): + """Run an Olive pipeline from a JSON config, overriding output_dir.""" + from olive import run as olive_run + + config_path = _SCRIPT_DIR / config_name + with open(config_path) as f: + config = json.load(f) + + config["output_dir"] = str(_resolve(output_dir) / output_subdir) + if model_path is not None: + config["input_model"]["model_path"] = model_path + + with tempfile.NamedTemporaryFile( + mode="w", suffix=".json", dir=str(_SCRIPT_DIR), delete=False + ) as tmp: + json.dump(config, tmp, indent=4) + tmp_path = tmp.name + + try: + olive_run(tmp_path) + finally: + Path(tmp_path).unlink(missing_ok=True) + + +def run_olive_pipelines(output_dir: str, model_path: str = None): + """Run the TRT-RTX FP16 Olive pipelines.""" + print("=== Stage 1: Olive Encoder (OnnxConversion → FP16, opset 23) ===") + _run_olive_pipeline("nemotron_encoder_fp16_trtrtx.json", output_dir, "encoder.onnx", model_path) + print() + + print("=== Stage 2: Olive Decoder (OnnxConversion, FP16, opset 23) ===") + _run_olive_pipeline("nemotron_decoder_fp16_trtrtx.json", output_dir, "decoder.onnx", model_path) + print() + + print("=== Stage 3: Olive Joint (OnnxConversion, FP16, opset 23) ===") + _run_olive_pipeline("nemotron_joint_fp16_trtrtx.json", output_dir, "joint.onnx", model_path) + print() + + +def run_tokenizer_export(model_name: str, output_dir: str): + """Export tokenizer files to the output directory.""" + print("=== Stage 4: Exporting tokenizer ===") + cmd = [ + sys.executable, + str(_TOKENIZER_SCRIPT), + "--model_name", model_name, + "--output_dir", str(_resolve(output_dir)), + ] + result = subprocess.run(cmd, cwd=str(_SCRIPT_DIR)) + if result.returncode != 0: + raise RuntimeError(f"Tokenizer export failed (exit code {result.returncode})") + print() + + +def generate_configs(model_name: str, output_dir: str, chunk_size: float, include_vad: bool = True): + """Generate genai_config.json and audio_processor_config.json. + + Loads the NeMo model to extract architecture parameters, then writes + the config files needed by onnxruntime-genai for inference. + """ + print("=== Stage 5: Generating config files ===") + from NvTensorRtRtx.nemotron_model_load import _load_nemo_model, get_att_context_size, D_MODEL, N_LAYERS, DECODER_HIDDEN, DECODER_LSTM_LAYERS + + asr_model = _load_nemo_model(model_name) + asr_model.eval() + + dst = _resolve(output_dir) + dst.mkdir(parents=True, exist_ok=True) + + encoder = asr_model.encoder + joint = asr_model.joint + + vocab_size = joint.num_classes_with_blank + blank_id = vocab_size - 1 + + preprocessor_cfg = asr_model.cfg.get('preprocessor', {}) + sample_rate = preprocessor_cfg.get('sample_rate', 16000) + n_mels = preprocessor_cfg.get('features', preprocessor_cfg.get('nfilt', 128)) + n_fft = preprocessor_cfg.get('n_fft', 512) + hop_length = preprocessor_cfg.get('hop_length', 160) + win_length = preprocessor_cfg.get('win_length', 400) + preemph = preprocessor_cfg.get('preemph', 0.97) + + subsampling_factor = getattr(encoder, 'subsampling_factor', 8) + att_context_size = get_att_context_size(chunk_size) + left_context = att_context_size[0] + + conv_context = 8 + if hasattr(encoder, 'layers') and len(encoder.layers) > 0: + layer = encoder.layers[0] + if hasattr(layer, 'conv') and hasattr(layer.conv, 'conv'): + conv = layer.conv.conv + if hasattr(conv, 'kernel_size'): + ks = conv.kernel_size[0] if isinstance(conv.kernel_size, tuple) else conv.kernel_size + conv_context = ks - 1 + + pre_encode_cache_size = getattr(encoder, 'pre_encode_cache_size', 9) + if isinstance(pre_encode_cache_size, (list, tuple)): + pre_encode_cache_size = pre_encode_cache_size[-1] + + chunk_samples = int(chunk_size * sample_rate) + max_symbols = asr_model.cfg.get('decoding', {}).get('greedy', {}).get('max_symbols', 10) + + genai_config = { + "model": { + "type": "nemotron_speech", + "vocab_size": vocab_size, + "num_mels": n_mels, + "fft_size": n_fft, + "hop_length": hop_length, + "win_length": win_length, + "preemph": preemph, + "log_eps": 5.96046448e-08, + "subsampling_factor": subsampling_factor, + "left_context": left_context, + "conv_context": conv_context, + "pre_encode_cache_size": pre_encode_cache_size, + "sample_rate": sample_rate, + "chunk_samples": chunk_samples, + "blank_id": blank_id, + "max_symbols_per_step": max_symbols, + "encoder": { + "filename": "encoder.onnx", + "hidden_size": D_MODEL, + "num_hidden_layers": N_LAYERS, + "inputs": { + "audio_features": "audio_signal", + "input_lengths": "length", + "cache_last_channel": "cache_last_channel", + "cache_last_time": "cache_last_time", + "cache_last_channel_len": "cache_last_channel_len", + "lang_id": "lang_id", + }, + "outputs": { + "encoder_outputs": "outputs", + "output_lengths": "encoded_lengths", + "cache_last_channel_next": "cache_last_channel_next", + "cache_last_time_next": "cache_last_time_next", + "cache_last_channel_len_next": "cache_last_channel_len_next", + }, + }, + "decoder": { + "filename": "decoder.onnx", + "hidden_size": DECODER_HIDDEN, + "num_hidden_layers": DECODER_LSTM_LAYERS, + "inputs": { + "targets": "targets", + "lstm_hidden_state": "h_in", + "lstm_cell_state": "c_in", + }, + "outputs": { + "outputs": "decoder_output", + "lstm_hidden_state": "h_out", + "lstm_cell_state": "c_out", + }, + }, + "joiner": { + "filename": "joint.onnx", + "inputs": { + "encoder_outputs": "encoder_output", + "decoder_outputs": "decoder_output", + }, + "outputs": { + "logits": "joint_output", + }, + }, + }, + } + if include_vad: + genai_config["model"]["vad"] = { + "filename": "silero_vad.onnx", + "threshold": 0.3, + "silence_duration_ms": 3360, + "prefix_padding_ms": 560, + } + + with open(dst / "genai_config.json", "w") as f: + json.dump(genai_config, f, indent=2) + print(f" [OK] genai_config.json") + + # Audio processor config + window_size = preprocessor_cfg.get('window_size', preprocessor_cfg.get('n_window_size', 0.025)) + window_stride = preprocessor_cfg.get('window_stride', preprocessor_cfg.get('n_window_stride', 0.01)) + if isinstance(window_size, float) and window_size < 1.0: + window_length_samples = int(window_size * sample_rate) + elif isinstance(window_size, int): + window_length_samples = window_size + else: + window_length_samples = 400 + if isinstance(window_stride, float) and window_stride < 1.0: + hop_length_samples = int(window_stride * sample_rate) + elif isinstance(window_stride, int): + hop_length_samples = window_stride + else: + hop_length_samples = 160 + + audio_config = { + "model_type": "speech_features", + "audio_params": { + "sample_rate": sample_rate, + "n_fft": n_fft, + "hop_length": hop_length_samples, + "n_mels": n_mels, + "window_length": window_length_samples, + "window_type": "hann", + "fmin": 0, + "fmax": sample_rate // 2, + "dither": preprocessor_cfg.get('dither', 0.0), + "preemphasis": preemph, + "log_zero_guard_type": "add", + "log_zero_guard_value": 1e-10, + "normalize": preprocessor_cfg.get('normalize', 'none'), + "center": True, + "mag_power": 2.0, + }, + } + + with open(dst / "audio_processor_config.json", "w") as f: + json.dump(audio_config, f, indent=2) + print(f" [OK] audio_processor_config.json") + print() + + +def main(): + from NvTensorRtRtx.nemotron_model_load import MODEL_NAME, CHUNK_SIZE + + parser = argparse.ArgumentParser( + description="Optimize Nemotron Speech Streaming for NvTensorRtRtx inference" + ) + parser.add_argument( + "--model-name", + default=MODEL_NAME, + help="HuggingFace model name or path to a local .nemo file", + ) + parser.add_argument( + "--output-dir", + default=DEFAULT_OUTPUT_DIR, + help=f"Output directory for optimized models (default: {DEFAULT_OUTPUT_DIR})", + ) + args = parser.parse_args() + + if not args.model_name.endswith(".nemo") and args.model_name != MODEL_NAME: + raise ValueError( + f"This recipe only supports '{MODEL_NAME}' (or a .nemo file with the same architecture). " + f"Got: '{args.model_name}'" + ) + + run_olive_pipelines(output_dir=args.output_dir, model_path=args.model_name) + run_tokenizer_export(model_name=args.model_name, output_dir=args.output_dir) + generate_configs( + model_name=args.model_name, + output_dir=args.output_dir, + chunk_size=CHUNK_SIZE, + include_vad=False, + ) + + print("=== Stage 6: Skipping Silero VAD for NvTensorRtRtx ===") + print(" VAD is omitted from genai_config.json for TRT-RTX compatibility.") + print() + + output_path = _resolve(args.output_dir) + if output_path.exists(): + files = sorted(f for f in output_path.iterdir() if f.is_file()) + total_mb = sum(f.stat().st_size for f in files) / (1024 * 1024) + print(f"=== Done! Optimized models → {output_path} ===") + print(f" Total size: {total_mb:.1f} MB") + for f in files: + tag = " ← FP16 (Olive, opset 23)" if f.name.startswith("encoder") else "" + print(f" {f.name} ({f.stat().st_size / (1024 * 1024):.1f} MB){tag}") + + +if __name__ == "__main__": + main() diff --git a/nvidia-nemotron-asr-streaming-multilingual-0.6b/src/requirements.txt b/nvidia-nemotron-asr-streaming-multilingual-0.6b/NvTensorRtRtx/requirements.txt similarity index 100% rename from nvidia-nemotron-asr-streaming-multilingual-0.6b/src/requirements.txt rename to nvidia-nemotron-asr-streaming-multilingual-0.6b/NvTensorRtRtx/requirements.txt diff --git a/nvidia-nemotron-asr-streaming-multilingual-0.6b/README.md b/nvidia-nemotron-asr-streaming-multilingual-0.6b/README.md index 0fde31d95..16771b342 100644 --- a/nvidia-nemotron-asr-streaming-multilingual-0.6b/README.md +++ b/nvidia-nemotron-asr-streaming-multilingual-0.6b/README.md @@ -7,13 +7,14 @@ a 0.6B-parameter streaming multilingual ASR model covering 100+ languages. Multilingual evaluation is in progress on **FLEURS**, **Common Voice**, **Multilingual LibriSpeech (MLS)**, and **VoxPopuli**. Per-language WER -numbers for the INT4 ONNX build will be published here once the matrix is -complete. +numbers will be published here once the matrix is complete. ## Recipes -- [`src/`](./src) — ONNX export + INT4 / INT8 quantization for CPU and CUDA - execution providers. +- [`cpu/`](./cpu) — INT4 or INT8 encoder with FP32 decoder and joint models for + the CPU execution provider. +- [`NvTensorRtRtx/`](./NvTensorRtRtx) — homogeneous FP16 opset-23 encoder, + decoder, and joint models for the NvTensorRtRtx execution provider. See the README inside each subfolder for setup and run instructions. @@ -21,11 +22,11 @@ See the README inside each subfolder for setup and run instructions. Streaming inference is supported via [`onnxruntime-genai`](https://github.com/microsoft/onnxruntime-genai), with a per-utterance `--language` flag (or `auto`) that selects the encoder -prompt token. Example: +prompt token. For example, after running the CPU recipe: ```bash python onnxruntime-genai/examples/python/nemotron_speech.py \ - --model_path src/build/onnx_models_int4 \ + --model_path cpu/build/onnx_models_int4 \ --audio_file path/to/audio.wav \ --language de \ -e cpu diff --git a/nvidia-nemotron-asr-streaming-multilingual-0.6b/cpu/.gitignore b/nvidia-nemotron-asr-streaming-multilingual-0.6b/cpu/.gitignore new file mode 100644 index 000000000..291419770 --- /dev/null +++ b/nvidia-nemotron-asr-streaming-multilingual-0.6b/cpu/.gitignore @@ -0,0 +1,14 @@ +# Generated model artifacts +build/ + +# Python bytecode +__pycache__/ +*.pyc + +# Olive cache +.olive-cache/ + +# Temp and log files +*.temp +*.bak +*.log diff --git a/nvidia-nemotron-asr-streaming-multilingual-0.6b/cpu/README.md b/nvidia-nemotron-asr-streaming-multilingual-0.6b/cpu/README.md new file mode 100644 index 000000000..55b466a5e --- /dev/null +++ b/nvidia-nemotron-asr-streaming-multilingual-0.6b/cpu/README.md @@ -0,0 +1,64 @@ +# Nemotron 3.5 ASR Streaming Multilingual 0.6B — CPU + +This recipe exports +**nvidia/NVIDIA-Nemotron-3.5-ASR-Streaming-Multilingual-0.6b** to ONNX for +CPU inference: + +- **Encoder**: INT4 k-quant by default, or INT8 dynamic quantization +- **Decoder**: FP32 +- **Joint**: FP32 + +## Setup + +From the repository root: + +```bash +python -m venv .venv +source .venv/bin/activate +pip install -r nvidia-nemotron-asr-streaming-multilingual-0.6b/cpu/requirements.txt +``` + +## Run + +From the recipe directory: + +```bash +cd nvidia-nemotron-asr-streaming-multilingual-0.6b + +# INT4 encoder (default) +python cpu/optimize.py + +# INT8 encoder +python cpu/optimize.py --encoder-precision int8 + +# Custom output directory +python cpu/optimize.py --output-dir build/multilingual_onnx_int4 +``` + +The pipeline exports the encoder, decoder, joint network, tokenizer, and +runtime configuration files, and downloads Silero VAD. + +Individual Olive configurations can also be run directly: + +```bash +python -m olive run --config cpu/nemotron_encoder_int4_cpu.json +python -m olive run --config cpu/nemotron_encoder_int8_cpu.json +python -m olive run --config cpu/nemotron_decoder_fp32_cpu.json +python -m olive run --config cpu/nemotron_joint_fp32_cpu.json +``` + +## Output + +The default output directory is `cpu/build/onnx_models_int4/`. It contains +`encoder.onnx`, `decoder.onnx`, `joint.onnx`, tokenizer files, +`genai_config.json`, `audio_processor_config.json`, and `silero_vad.onnx`. + +## Inference + +```bash +python onnxruntime-genai/examples/python/nemotron_speech.py \ + --model_path cpu/build/onnx_models_int4 \ + --audio_file path/to/audio.wav \ + --language de \ + -e cpu +``` diff --git a/nvidia-nemotron-asr-streaming-multilingual-0.6b/cpu/__init__.py b/nvidia-nemotron-asr-streaming-multilingual-0.6b/cpu/__init__.py new file mode 100644 index 000000000..e69de29bb diff --git a/nvidia-nemotron-asr-streaming-multilingual-0.6b/cpu/info.yaml b/nvidia-nemotron-asr-streaming-multilingual-0.6b/cpu/info.yaml new file mode 100644 index 000000000..b3a6c223e --- /dev/null +++ b/nvidia-nemotron-asr-streaming-multilingual-0.6b/cpu/info.yaml @@ -0,0 +1,19 @@ +name: nvidia-nemotron-asr-streaming-multilingual-0.6b +provider: NVIDIA +model_id: nvidia/NVIDIA-Nemotron-3.5-ASR-Streaming-Multilingual-0.6b +task: automatic-speech-recognition +framework: ONNX Runtime +execution_provider: CPUExecutionProvider +summary: > + CPU recipe for exporting the multilingual NVIDIA Nemotron 3.5 ASR + Streaming RNNT model and quantizing the encoder to INT4 or INT8. + +artifacts: + - cpu/nemotron_encoder_int4_cpu.json + - cpu/nemotron_encoder_int8_cpu.json + - cpu/nemotron_decoder_fp32_cpu.json + - cpu/nemotron_joint_fp32_cpu.json + - cpu/optimize.py + +scripts: + - scripts/export_tokenizer.py diff --git a/nvidia-nemotron-asr-streaming-multilingual-0.6b/src/nemotron_decoder_fp32_cpu.json b/nvidia-nemotron-asr-streaming-multilingual-0.6b/cpu/nemotron_decoder_fp32_cpu.json similarity index 96% rename from nvidia-nemotron-asr-streaming-multilingual-0.6b/src/nemotron_decoder_fp32_cpu.json rename to nvidia-nemotron-asr-streaming-multilingual-0.6b/cpu/nemotron_decoder_fp32_cpu.json index f6a1b51fb..d5defe3e6 100644 --- a/nvidia-nemotron-asr-streaming-multilingual-0.6b/src/nemotron_decoder_fp32_cpu.json +++ b/nvidia-nemotron-asr-streaming-multilingual-0.6b/cpu/nemotron_decoder_fp32_cpu.json @@ -3,7 +3,7 @@ "type": "PyTorchModel", "model_path": "nvidia/NVIDIA-Nemotron-3.5-ASR-Streaming-Multilingual-0.6b", "model_loader": "decoder_model_loader", - "model_script": "src/nemotron_model_load.py", + "model_script": "cpu/nemotron_model_load.py", "io_config": { "input_names": ["targets", "h_in", "c_in"], "output_names": ["decoder_output", "h_out", "c_out"], diff --git a/nvidia-nemotron-asr-streaming-multilingual-0.6b/src/nemotron_encoder_int4_cpu.json b/nvidia-nemotron-asr-streaming-multilingual-0.6b/cpu/nemotron_encoder_int4_cpu.json similarity index 96% rename from nvidia-nemotron-asr-streaming-multilingual-0.6b/src/nemotron_encoder_int4_cpu.json rename to nvidia-nemotron-asr-streaming-multilingual-0.6b/cpu/nemotron_encoder_int4_cpu.json index 947e24fc0..967e028bc 100644 --- a/nvidia-nemotron-asr-streaming-multilingual-0.6b/src/nemotron_encoder_int4_cpu.json +++ b/nvidia-nemotron-asr-streaming-multilingual-0.6b/cpu/nemotron_encoder_int4_cpu.json @@ -3,7 +3,7 @@ "type": "PyTorchModel", "model_path": "nvidia/NVIDIA-Nemotron-3.5-ASR-Streaming-Multilingual-0.6b", "model_loader": "encoder_model_loader", - "model_script": "src/nemotron_model_load.py", + "model_script": "cpu/nemotron_model_load.py", "io_config": { "input_names": [ "audio_signal", "length", diff --git a/nvidia-nemotron-asr-streaming-multilingual-0.6b/src/nemotron_encoder_int8_cpu.json b/nvidia-nemotron-asr-streaming-multilingual-0.6b/cpu/nemotron_encoder_int8_cpu.json similarity index 96% rename from nvidia-nemotron-asr-streaming-multilingual-0.6b/src/nemotron_encoder_int8_cpu.json rename to nvidia-nemotron-asr-streaming-multilingual-0.6b/cpu/nemotron_encoder_int8_cpu.json index 5afa4a9c9..05559c596 100644 --- a/nvidia-nemotron-asr-streaming-multilingual-0.6b/src/nemotron_encoder_int8_cpu.json +++ b/nvidia-nemotron-asr-streaming-multilingual-0.6b/cpu/nemotron_encoder_int8_cpu.json @@ -3,7 +3,7 @@ "type": "PyTorchModel", "model_path": "nvidia/NVIDIA-Nemotron-3.5-ASR-Streaming-Multilingual-0.6b", "model_loader": "encoder_model_loader", - "model_script": "src/nemotron_model_load.py", + "model_script": "cpu/nemotron_model_load.py", "io_config": { "input_names": [ "audio_signal", "length", diff --git a/nvidia-nemotron-asr-streaming-multilingual-0.6b/src/nemotron_joint_fp32_cpu.json b/nvidia-nemotron-asr-streaming-multilingual-0.6b/cpu/nemotron_joint_fp32_cpu.json similarity index 95% rename from nvidia-nemotron-asr-streaming-multilingual-0.6b/src/nemotron_joint_fp32_cpu.json rename to nvidia-nemotron-asr-streaming-multilingual-0.6b/cpu/nemotron_joint_fp32_cpu.json index 86d65c7b6..0948e2dc1 100644 --- a/nvidia-nemotron-asr-streaming-multilingual-0.6b/src/nemotron_joint_fp32_cpu.json +++ b/nvidia-nemotron-asr-streaming-multilingual-0.6b/cpu/nemotron_joint_fp32_cpu.json @@ -3,7 +3,7 @@ "type": "PyTorchModel", "model_path": "nvidia/NVIDIA-Nemotron-3.5-ASR-Streaming-Multilingual-0.6b", "model_loader": "joint_model_loader", - "model_script": "src/nemotron_model_load.py", + "model_script": "cpu/nemotron_model_load.py", "io_config": { "input_names": ["encoder_output", "decoder_output"], "output_names": ["joint_output"], diff --git a/nvidia-nemotron-asr-streaming-multilingual-0.6b/cpu/nemotron_model_load.py b/nvidia-nemotron-asr-streaming-multilingual-0.6b/cpu/nemotron_model_load.py new file mode 100644 index 000000000..7062b34eb --- /dev/null +++ b/nvidia-nemotron-asr-streaming-multilingual-0.6b/cpu/nemotron_model_load.py @@ -0,0 +1,310 @@ +# ------------------------------------------------------------------------- +# Copyright (c) Microsoft Corporation. All rights reserved. +# Licensed under the MIT License. +# -------------------------------------------------------------------------- +"""Model loaders and dummy input generators for Nemotron Speech Streaming components. + +Used by Olive's OnnxConversion pass via the ``model_script`` / ``model_loader`` +mechanism. Each component (encoder, decoder, joint) has its own loader and +dummy inputs function, referenced from separate Olive JSON configs. + +Streaming defaults (chunk_size=0.56s, left_chunks=10) are defined as constants +below and are the single source of truth for the export shapes. +""" + +import torch +import torch.nn as nn +import torch.nn.functional as F + +# --------------------------------------------------------------------------- +# Shared streaming constants +# --------------------------------------------------------------------------- +# CHUNK_SIZE is hardcoded because it determines the static ONNX input shapes +# at export time. The NeMo model supports multiple chunk sizes (0.08, 0.16, +# 0.56, 1.12s) at runtime, but once exported to ONNX with static shapes the +# encoder is locked to a single chunk size. 0.56s is the recommended default +# per NVIDIA's documentation (best latency/accuracy trade-off). The value is +# not available from a HuggingFace config — it lives inside the .nemo archive +# as encoder.att_context_size and requires loading the full model to read. +CHUNK_SIZE = 0.56 # seconds +MEL_FEATURES = 128 +SUBSAMPLING_FACTOR = 8 + +# Model architecture constants (0.6B multilingual model) +N_LAYERS = 24 +D_MODEL = 1024 +CONV_CONTEXT = 8 # conv_kernel_size(9) - 1 +DECODER_HIDDEN = 640 +DECODER_LSTM_LAYERS = 2 +NUM_PROMPTS = 128 # one-hot language-ID size + +# Streaming config — single source of truth. +# chunk_encoded_frames = int(CHUNK_SIZE * 100) // SUBSAMPLING_FACTOR = 7 +# last_channel_cache_size = LEFT_CHUNKS * chunk_encoded_frames = 70 +LEFT_CHUNKS = 10 + +MODEL_NAME = "nvidia/NVIDIA-Nemotron-3.5-ASR-Streaming-Multilingual-0.6b" + + +def get_att_context_size(chunk_size: float = CHUNK_SIZE, left_chunks: int = LEFT_CHUNKS): + """Compute attention context size for the streaming encoder. + + left_context = left_chunks * chunk_encoded_frames; + right_context is indexed by chunk_size. + """ + right_context = {0.08: 0, 0.16: 1, 0.56: 6, 1.12: 13}.get(chunk_size, 13) + chunk_encoded_frames = int(chunk_size * 100) // SUBSAMPLING_FACTOR + left_context = left_chunks * chunk_encoded_frames + return [left_context, right_context] + + +def _get_streaming_shapes(): + """Compute static streaming tensor shapes from the shared constants.""" + pre_encode_cache = 9 + chunk_mel_frames = int(CHUNK_SIZE * 100) # 56 for 0.56s + static_mel_frames = chunk_mel_frames + pre_encode_cache # 65 + chunk_encoded_frames = chunk_mel_frames // SUBSAMPLING_FACTOR # 7 + last_channel_cache_size = LEFT_CHUNKS * chunk_encoded_frames # 70 + + return { + "last_channel_cache_size": last_channel_cache_size, + "static_mel_frames": static_mel_frames, + "chunk_encoded_frames": chunk_encoded_frames, + } + +def configure_encoder_for_sdpa(encoder): + for layer in getattr(encoder, "layers", []) or []: + attention = getattr(layer, "self_attn", None) + if attention is not None and hasattr(attention, "use_pytorch_sdpa"): + attention.use_pytorch_sdpa = True + +def _load_nemo_model(model_name=MODEL_NAME): + """Load the NeMo ASR model (shared across loaders). + + For HF repo IDs we download the ``.nemo`` archive via ``hf_hub_download`` + and feed it to ``restore_from``. This avoids a bug in NeMo 2.6.2's + ``from_pretrained`` HF-cache integration where the cache directory is + incorrectly treated as already-extracted (because the repo also ships + README/safety markdown files alongside the archive), causing + ``model_config.yaml`` lookup to fail. + """ + import nemo.collections.asr as nemo_asr + + if model_name.endswith(".nemo"): + nemo_path = model_name + else: + from huggingface_hub import hf_hub_download, list_repo_files + + # Find the .nemo file in the repo (filename is not standardised). + files = list_repo_files(model_name) + nemo_files = [f for f in files if f.endswith(".nemo")] + if not nemo_files: + raise RuntimeError( + f"No .nemo archive found in HuggingFace repo {model_name!r}" + ) + if len(nemo_files) > 1: + raise RuntimeError( + f"Multiple .nemo archives found in {model_name!r}: {nemo_files}" + ) + nemo_path = hf_hub_download(repo_id=model_name, filename=nemo_files[0]) + + asr_model = nemo_asr.models.ASRModel.restore_from(nemo_path, map_location="cpu") + asr_model = asr_model.cpu() + asr_model.eval() + return asr_model + + +# --------------------------------------------------------------------------- +# Encoder +# --------------------------------------------------------------------------- + +class StreamingEncoderWrapper(nn.Module): + """Wrap the NeMo CacheAware encoder + prompt_kernel for streaming ONNX export. + + Takes a `lang_id` int64 input of shape [B]. The one-hot prompt tensor of + shape [B, T_out, NUM_PROMPTS] is built inside the graph, then concatenated + to the encoded features along the channel dim and projected back to + D_MODEL via `prompt_kernel` (Linear -> ReLU -> Linear). + """ + + def __init__(self, enc, prompt_kernel): + super().__init__() + self.enc = enc + self.prompt_kernel = prompt_kernel + + def forward(self, audio_signal, length, + cache_last_channel, cache_last_time, cache_last_channel_len, + lang_id): + audio_signal = audio_signal.transpose(1, 2) # [B, T, mel] -> [B, mel, T] + encoded, encoded_len, cache_ch_next, cache_tm_next, cache_len_next = \ + self.enc.forward_for_export( + audio_signal=audio_signal, + length=length, + cache_last_channel=cache_last_channel, + cache_last_time=cache_last_time, + cache_last_channel_len=cache_last_channel_len, + ) + encoded = encoded.transpose(1, 2) # [B, D, T] -> [B, T, D] + # Build one-hot prompt [B, T, NUM_PROMPTS] from lang_id [B]. + onehot = F.one_hot(lang_id, num_classes=NUM_PROMPTS).to(encoded.dtype) # [B, 128] + prompt = onehot.unsqueeze(1).expand(-1, encoded.shape[1], -1) # [B, T, 128] + concat = torch.cat([encoded, prompt], dim=-1) + encoded = self.prompt_kernel(concat).to(encoded.dtype) + return encoded, encoded_len, cache_ch_next, cache_tm_next, cache_len_next + + +def encoder_model_loader(model_name): + """Load the NeMo model and return the streaming encoder wrapper.""" + asr_model = _load_nemo_model(model_name) + encoder = asr_model.encoder + encoder.eval() + prompt_kernel = asr_model.prompt_kernel + prompt_kernel.eval() + + att_context_size = get_att_context_size() + if hasattr(encoder, "set_default_att_context_size"): + encoder.set_default_att_context_size(att_context_size) + + wrapper = StreamingEncoderWrapper(encoder, prompt_kernel) + wrapper.eval() + return wrapper + + +def encoder_dummy_inputs(model): + """Generate dummy inputs for ONNX export of the streaming encoder.""" + shapes = _get_streaming_shapes() + static_mel_frames = shapes["static_mel_frames"] + last_channel_cache_size = shapes["last_channel_cache_size"] + + batch = 1 + return ( + torch.randn(batch, static_mel_frames, MEL_FEATURES), + torch.tensor([static_mel_frames], dtype=torch.int64), + torch.zeros(batch, N_LAYERS, last_channel_cache_size, D_MODEL), + torch.zeros(batch, N_LAYERS, D_MODEL, CONV_CONTEXT), + torch.zeros(batch, dtype=torch.int64), + torch.zeros(batch, dtype=torch.int64), # lang_id [B] + ) + + +def encoder_fp16_dummy_inputs(model): + """Generate FP16 floating-point inputs for the TRT-RTX encoder export.""" + inputs = list(encoder_dummy_inputs(model)) + for index in (0, 2, 3): + inputs[index] = inputs[index].to(dtype=torch.float16) + return tuple(inputs) + + +# --------------------------------------------------------------------------- +# Decoder (stateful LSTM) +# --------------------------------------------------------------------------- + +class StatefulDecoderWrapper(nn.Module): + """Wrap the NeMo decoder to expose LSTM states as explicit I/O.""" + + def __init__(self, dec): + super().__init__() + self.decoder = dec + self.decoder._rnnt_export = True + + def forward(self, targets, h_in, c_in): + g, states = self.decoder.predict( + y=targets, state=(h_in, c_in), add_sos=False + ) + h_out, c_out = states + g = g.transpose(1, 2) # [B, 1, D] -> [B, D, 1] + return g, h_out, c_out + + +def decoder_model_loader(model_name): + """Load the NeMo model and return the stateful decoder wrapper.""" + asr_model = _load_nemo_model(model_name) + decoder = asr_model.decoder + decoder.eval() + + wrapper = StatefulDecoderWrapper(decoder) + wrapper.eval() + return wrapper + + +def encoder_fp16_model_loader(model_name): + """Load the streaming encoder wrapper in FP16 for NvTensorRtRtx export.""" + wrapper = encoder_model_loader(model_name) + configure_encoder_for_sdpa(wrapper.enc) + wrapper.to(dtype=torch.float16) + wrapper.eval() + return wrapper + + +def decoder_fp16_model_loader(model_name): + """Load the stateful decoder wrapper in FP16 for NvTensorRtRtx export.""" + wrapper = decoder_model_loader(model_name) + wrapper.to(dtype=torch.float16) + wrapper.eval() + return wrapper + + +def decoder_dummy_inputs(model): + """Generate dummy inputs for ONNX export of the stateful decoder.""" + batch = 1 + return ( + torch.zeros(batch, 1, dtype=torch.int64), + torch.zeros(DECODER_LSTM_LAYERS, batch, DECODER_HIDDEN, dtype=torch.float32), + torch.zeros(DECODER_LSTM_LAYERS, batch, DECODER_HIDDEN, dtype=torch.float32), + ) + + +def decoder_fp16_dummy_inputs(model): + """Generate INT64 targets and FP16 LSTM states for NvTensorRtRtx export.""" + inputs = list(decoder_dummy_inputs(model)) + inputs[1] = inputs[1].to(dtype=torch.float16) + inputs[2] = inputs[2].to(dtype=torch.float16) + return tuple(inputs) + + +# --------------------------------------------------------------------------- +# Joint network +# --------------------------------------------------------------------------- + +class JointWrapper(nn.Module): + """Wrap the NeMo RNNTJoint so torch.onnx.export can trace it.""" + + def __init__(self, j): + super().__init__() + self.joint = j + + def forward(self, encoder_output, decoder_output): + return self.joint.joint(encoder_output, decoder_output) + + +def joint_model_loader(model_name): + """Load the NeMo model and return the joint network wrapper.""" + asr_model = _load_nemo_model(model_name) + joint = asr_model.joint + joint.eval() + + wrapper = JointWrapper(joint) + wrapper.eval() + return wrapper + + +def joint_fp16_model_loader(model_name): + """Load the joint wrapper in FP16 for NvTensorRtRtx export.""" + wrapper = joint_model_loader(model_name) + wrapper.to(dtype=torch.float16) + wrapper.eval() + return wrapper + + +def joint_dummy_inputs(model): + """Generate dummy inputs for ONNX export of the joint network.""" + batch = 1 + return ( + torch.randn(batch, 1, D_MODEL), + torch.randn(batch, 1, DECODER_HIDDEN), + ) + + +def joint_fp16_dummy_inputs(model): + """Generate FP16 encoder and decoder inputs for NvTensorRtRtx export.""" + return tuple(value.to(dtype=torch.float16) for value in joint_dummy_inputs(model)) diff --git a/nvidia-nemotron-asr-streaming-multilingual-0.6b/src/optimize.py b/nvidia-nemotron-asr-streaming-multilingual-0.6b/cpu/optimize.py similarity index 85% rename from nvidia-nemotron-asr-streaming-multilingual-0.6b/src/optimize.py rename to nvidia-nemotron-asr-streaming-multilingual-0.6b/cpu/optimize.py index 616fd00fe..a3bcb3228 100644 --- a/nvidia-nemotron-asr-streaming-multilingual-0.6b/src/optimize.py +++ b/nvidia-nemotron-asr-streaming-multilingual-0.6b/cpu/optimize.py @@ -1,23 +1,17 @@ -"""End-to-end optimization pipeline for Nemotron Speech Streaming. +"""End-to-end CPU optimization pipeline for Nemotron Speech Streaming. -All model components (encoder, decoder, joint) are exported and optimized -using Olive's declarative pass system: - - - Encoder: OnnxConversion → OnnxKQuantQuantization - - Decoder: OnnxConversion (FP32) - - Joint: OnnxConversion (FP32) - -After the Olive pipelines, tokenizer and config files are generated and -Silero VAD is downloaded. +The encoder is exported and quantized to INT4 or INT8. The decoder and joint +network are exported in FP32. The pipeline also exports the tokenizer, +generates runtime configuration files, and downloads Silero VAD. Usage: - # Full pipeline - python src/optimize.py + python cpu/optimize.py + python cpu/optimize.py --encoder-precision int8 - # Or use Olive CLI directly for individual components: - python -m olive run --config src/nemotron_encoder_int4_cpu.json - python -m olive run --config src/nemotron_decoder_fp32_cpu.json - python -m olive run --config src/nemotron_joint_fp32_cpu.json + # Or run individual Olive configs: + python -m olive run --config cpu/nemotron_encoder_int4_cpu.json + python -m olive run --config cpu/nemotron_decoder_fp32_cpu.json + python -m olive run --config cpu/nemotron_joint_fp32_cpu.json """ import argparse @@ -28,7 +22,7 @@ import tempfile from pathlib import Path -# Ensure the recipe root is on sys.path so `from src.nemotron_model_load import ...` works +# Ensure the recipe root is on sys.path so `from cpu.nemotron_model_load import ...` works # regardless of where the script is invoked from. _SCRIPT_DIR = Path(__file__).resolve().parent _RECIPE_ROOT = _SCRIPT_DIR.parent @@ -40,8 +34,15 @@ DEFAULT_OUTPUT_DIR = "build/onnx_models_int4" +def resolve_encoder_config(encoder_precision: str) -> str: + """Select the encoder Olive config for the requested precision.""" + if encoder_precision == "int8": + return "nemotron_encoder_int8_cpu.json" + return "nemotron_encoder_int4_cpu.json" + + def _resolve(path: str) -> Path: - """Resolve a path relative to the src/ directory.""" + """Resolve a path relative to the cpu/ directory.""" p = Path(path) return p if p.is_absolute() else _SCRIPT_DIR / p @@ -71,14 +72,12 @@ def _run_olive_pipeline(config_name: str, output_dir: str, output_subdir: str, m def run_olive_pipelines(output_dir: str, model_path: str = None, encoder_precision: str = "int4"): - """Run all Olive pipelines: encoder (INT4 or INT8), decoder (FP32), joint (FP32).""" + """Run the CPU Olive pipelines.""" if encoder_precision == "int8": - encoder_config = "nemotron_encoder_int8_cpu.json" print("=== Stage 1: Olive Encoder (OnnxConversion → INT8 k-quant) ===") else: - encoder_config = "nemotron_encoder_int4_cpu.json" print("=== Stage 1: Olive Encoder (OnnxConversion → INT4 quant) ===") - _run_olive_pipeline(encoder_config, output_dir, "encoder.onnx", model_path) + _run_olive_pipeline(resolve_encoder_config(encoder_precision), output_dir, "encoder.onnx", model_path) print() print("=== Stage 2: Olive Decoder (OnnxConversion, FP32) ===") @@ -105,14 +104,14 @@ def run_tokenizer_export(model_name: str, output_dir: str): print() -def generate_configs(model_name: str, output_dir: str, chunk_size: float): +def generate_configs(model_name: str, output_dir: str, chunk_size: float, include_vad: bool = True): """Generate genai_config.json and audio_processor_config.json. Loads the NeMo model to extract architecture parameters, then writes the config files needed by onnxruntime-genai for inference. """ print("=== Stage 5: Generating config files ===") - from src.nemotron_model_load import _load_nemo_model, get_att_context_size, D_MODEL, N_LAYERS, DECODER_HIDDEN, DECODER_LSTM_LAYERS + from cpu.nemotron_model_load import _load_nemo_model, get_att_context_size, D_MODEL, N_LAYERS, DECODER_HIDDEN, DECODER_LSTM_LAYERS asr_model = _load_nemo_model(model_name) asr_model.eval() @@ -217,14 +216,15 @@ def generate_configs(model_name: str, output_dir: str, chunk_size: float): "logits": "joint_output", }, }, - "vad": { - "filename": "silero_vad.onnx", - "threshold": 0.3, - "silence_duration_ms": 3360, - "prefix_padding_ms": 560, - }, }, } + if include_vad: + genai_config["model"]["vad"] = { + "filename": "silero_vad.onnx", + "threshold": 0.3, + "silence_duration_ms": 3360, + "prefix_padding_ms": 560, + } with open(dst / "genai_config.json", "w") as f: json.dump(genai_config, f, indent=2) @@ -294,7 +294,7 @@ def download_silero_vad(output_dir: str): def main(): - from src.nemotron_model_load import MODEL_NAME, CHUNK_SIZE + from cpu.nemotron_model_load import MODEL_NAME, CHUNK_SIZE parser = argparse.ArgumentParser( description="Optimize Nemotron Speech Streaming for CPU inference" @@ -313,36 +313,29 @@ def main(): "--encoder-precision", choices=["int4", "int8"], default="int4", - help="Encoder precision: int4 (k-quant) or int8 (dynamic). Default: int4.", + help="Encoder precision. Default: int4.", ) args = parser.parse_args() - # Validate model name — the Olive configs and model_load.py constants are - # specific to the 0.6B model architecture. if not args.model_name.endswith(".nemo") and args.model_name != MODEL_NAME: raise ValueError( f"This recipe only supports '{MODEL_NAME}' (or a .nemo file with the same architecture). " f"Got: '{args.model_name}'" ) - # Stages 1-3: Run Olive pipelines for encoder, decoder, joint run_olive_pipelines( output_dir=args.output_dir, model_path=args.model_name, encoder_precision=args.encoder_precision, ) - - # Stage 4: Export tokenizer run_tokenizer_export(model_name=args.model_name, output_dir=args.output_dir) - - # Stage 5: Generate config files (chunk_size matches the hardcoded export shapes) generate_configs( model_name=args.model_name, output_dir=args.output_dir, chunk_size=CHUNK_SIZE, + include_vad=True, ) - # Stage 6: Download Silero VAD vad_dest = _resolve(args.output_dir) / "silero_vad.onnx" try: download_silero_vad(output_dir=args.output_dir) @@ -353,16 +346,18 @@ def main(): f" and place silero_vad.onnx at: {vad_dest}" ) - # Summary output_path = _resolve(args.output_dir) if output_path.exists(): files = sorted(f for f in output_path.iterdir() if f.is_file()) total_mb = sum(f.stat().st_size for f in files) / (1024 * 1024) print(f"=== Done! Optimized models → {output_path} ===") print(f" Total size: {total_mb:.1f} MB") - enc_label = {"int4": "INT4 k-quant (Olive)", "int8": "INT8 dynamic (Olive)"}.get(args.encoder_precision, "") + enc_label = { + "int4": "INT4 k-quant (Olive)", + "int8": "INT8 dynamic (Olive)", + }[args.encoder_precision] for f in files: - tag = f" ← {enc_label}" if f.name.startswith("encoder") and enc_label else "" + tag = f" ← {enc_label}" if f.name.startswith("encoder") else "" print(f" {f.name} ({f.stat().st_size / (1024 * 1024):.1f} MB){tag}") diff --git a/nvidia-nemotron-asr-streaming-multilingual-0.6b/cpu/requirements.txt b/nvidia-nemotron-asr-streaming-multilingual-0.6b/cpu/requirements.txt new file mode 100644 index 000000000..bba1001c4 --- /dev/null +++ b/nvidia-nemotron-asr-streaming-multilingual-0.6b/cpu/requirements.txt @@ -0,0 +1,11 @@ +datasets>=3.0.0 +huggingface_hub>=0.23.0 +nemo_toolkit[asr]>=2.7.1 +numpy>=2.3.4 +olive-ai[speech] @ git+https://github.com/microsoft/Olive.git@main +onnx>=1.20.1 +onnxruntime>=1.24.4 +onnxruntime-genai>=0.13.0 +sentencepiece>=0.2.1 +torch>=2.9.1 +whisper-normalizer>=0.0.5 diff --git a/nvidia-nemotron-asr-streaming-multilingual-0.6b/scripts/README.md b/nvidia-nemotron-asr-streaming-multilingual-0.6b/scripts/README.md index 912beb9fa..095491a44 100644 --- a/nvidia-nemotron-asr-streaming-multilingual-0.6b/scripts/README.md +++ b/nvidia-nemotron-asr-streaming-multilingual-0.6b/scripts/README.md @@ -1,47 +1,23 @@ # Nemotron Scripts -Utility scripts for the Nemotron 3.5 ASR Streaming Multilingual 0.6B recipe. +Shared utilities and tests for the Nemotron 3.5 ASR Streaming Multilingual +0.6B recipes. -All ONNX export is now handled through Olive configs — see `src/README.md` -for the full pipeline. +ONNX export is handled through the provider-specific Olive recipes: -## Prerequisites +- [`../cpu/`](../cpu) for CPU +- [`../NvTensorRtRtx/`](../NvTensorRtRtx) for TRT-RTX -```bash -conda create -n nemotron-export python=3.10 -y -conda activate nemotron-export -pip install Cython packaging torch torchaudio onnxruntime -pip install "nemo_toolkit[asr]>=2.7.1" -``` - -## Export (via Olive) - -From the `nvidia-nemotron-asr-streaming-multilingual-0.6b` directory: +From the recipe directory, run one of: ```bash -python src/optimize.py +python cpu/optimize.py +python NvTensorRtRtx/optimize.py ``` -This exports all components (encoder, decoder, joint, tokenizer, configs) -through Olive's declarative pass system. See `src/README.md` for details. - -## Output Files - -| File | Description | -|------|-------------| -| `silero_vad.onnx` | Silero VAD model (downloaded from onnx-community/silero-vad) | -| `encoder.onnx` (+`.data`) | Multilingual streaming Conformer encoder (INT4 k-quant by default) | -| `decoder.onnx` (+`.data`) | RNNT prediction network (stateful LSTM h/c I/O, FP32) | -| `joint.onnx` (+`.data`) | Joint network (encoder + decoder → logits, FP32) | -| `genai_config.json` | Model configuration for onnxruntime-genai (includes per-language prompt IDs) | -| `audio_processor_config.json` | Mel spectrogram parameters (16 kHz, 128 mels, 512 FFT) | -| `model_config.json` | Architecture metadata used by genai | -| `tokenizer.json` | HuggingFace Unigram tokenizer (multilingual vocab) | -| `tokenizer_config.json` | T5Tokenizer class routing for ORT Extensions | -| `vocab.txt` | Raw vocabulary (one token per line) | - ## Scripts | Script | Purpose | |--------|---------| -| `export_tokenizer.py` | Extract vocab from NeMo and create ORT-compatible tokenizer | +| `export_tokenizer.py` | Extract the NeMo vocabulary and create an ORT-compatible tokenizer | +| `test_optimize.py` | Validate provider-specific export selection and model-loader behavior | diff --git a/nvidia-nemotron-asr-streaming-multilingual-0.6b/scripts/export_tokenizer.py b/nvidia-nemotron-asr-streaming-multilingual-0.6b/scripts/export_tokenizer.py index cef63128d..b8bb55834 100644 --- a/nvidia-nemotron-asr-streaming-multilingual-0.6b/scripts/export_tokenizer.py +++ b/nvidia-nemotron-asr-streaming-multilingual-0.6b/scripts/export_tokenizer.py @@ -35,7 +35,7 @@ def _restore_from_hf(model_name: str): if len(nemo_files) > 1: raise RuntimeError(f"Multiple .nemo archives in {model_name!r}: {nemo_files}") nemo_path = hf_hub_download(repo_id=model_name, filename=nemo_files[0]) - return nemo_asr.models.ASRModel.restore_from(nemo_path) + return nemo_asr.models.ASRModel.restore_from(nemo_path, map_location="cpu") def extract_vocab(model_name: str, output_dir: Path) -> list: @@ -50,7 +50,7 @@ def extract_vocab(model_name: str, output_dir: Path) -> list: sys.exit(1) asr_model = ( - nemo_asr.models.ASRModel.restore_from(model_name) + nemo_asr.models.ASRModel.restore_from(model_name, map_location="cpu") if model_name.endswith(".nemo") else _restore_from_hf(model_name) ) diff --git a/nvidia-nemotron-asr-streaming-multilingual-0.6b/scripts/test_optimize.py b/nvidia-nemotron-asr-streaming-multilingual-0.6b/scripts/test_optimize.py new file mode 100644 index 000000000..b579d409b --- /dev/null +++ b/nvidia-nemotron-asr-streaming-multilingual-0.6b/scripts/test_optimize.py @@ -0,0 +1,320 @@ +import importlib.util +import json +import sys +import tempfile +import types +from types import SimpleNamespace +from pathlib import Path +import unittest +from unittest.mock import patch + + +SCRIPTS_DIR = Path(__file__).resolve().parent +RECIPE_ROOT = SCRIPTS_DIR.parent +CPU_DIR = RECIPE_ROOT / "cpu" +TRT_RTX_DIR = RECIPE_ROOT / "NvTensorRtRtx" + + +def _load_module(name, path): + spec = importlib.util.spec_from_file_location(name, path) + module = importlib.util.module_from_spec(spec) + original_sys_path = sys.path.copy() + try: + spec.loader.exec_module(module) + finally: + sys.path[:] = original_sys_path + return module + + +cpu_optimize = _load_module("nemotron_cpu_optimize", CPU_DIR / "optimize.py") +trt_rtx_optimize = _load_module("nemotron_trt_rtx_optimize", TRT_RTX_DIR / "optimize.py") + + +def _load_export_tokenizer_module(): + return _load_module("nemotron_export_tokenizer", SCRIPTS_DIR / "export_tokenizer.py") + + +def _load_model_load_module(provider="NvTensorRtRtx"): + provider_dir = CPU_DIR if provider == "cpu" else TRT_RTX_DIR + return _load_module(f"nemotron_{provider}_model_load", provider_dir / "nemotron_model_load.py") + + +class NemotronOptimizePlanTest(unittest.TestCase): + def test_cpu_encoder_precision_selects_provider_local_config(self): + self.assertEqual( + cpu_optimize.resolve_encoder_config("int4"), + "nemotron_encoder_int4_cpu.json", + ) + self.assertEqual( + cpu_optimize.resolve_encoder_config("int8"), + "nemotron_encoder_int8_cpu.json", + ) + + def test_cpu_configs_use_cpu_execution_provider(self): + for config_name in ( + "nemotron_encoder_int4_cpu.json", + "nemotron_encoder_int8_cpu.json", + "nemotron_decoder_fp32_cpu.json", + "nemotron_joint_fp32_cpu.json", + ): + with self.subTest(config_name=config_name): + config = json.loads((CPU_DIR / config_name).read_text()) + self.assertEqual( + config["systems"]["local_system"]["accelerators"][0]["execution_providers"], + ["CPUExecutionProvider"], + ) + self.assertEqual( + config["input_model"]["model_script"], + "cpu/nemotron_model_load.py", + ) + + def test_trtrtx_configs_use_fp16_opset23(self): + expected_loaders = { + "nemotron_encoder_fp16_trtrtx.json": ( + "encoder_fp16_model_loader", + "encoder_fp16_dummy_inputs", + ), + "nemotron_decoder_fp16_trtrtx.json": ( + "decoder_fp16_model_loader", + "decoder_fp16_dummy_inputs", + ), + "nemotron_joint_fp16_trtrtx.json": ( + "joint_fp16_model_loader", + "joint_fp16_dummy_inputs", + ), + } + for config_name, (model_loader, dummy_inputs_func) in expected_loaders.items(): + with self.subTest(config_name=config_name): + config = json.loads((TRT_RTX_DIR / config_name).read_text()) + self.assertEqual( + config["systems"]["local_system"]["accelerators"][0]["execution_providers"], + ["NvTensorRTRTXExecutionProvider"], + ) + self.assertEqual(config["passes"]["convert"]["target_opset"], 23) + self.assertEqual(config["input_model"]["model_loader"], model_loader) + self.assertEqual(config["input_model"]["dummy_inputs_func"], dummy_inputs_func) + self.assertEqual( + config["input_model"]["model_script"], + "NvTensorRtRtx/nemotron_model_load.py", + ) + + def test_cpu_pipeline_uses_only_cpu_configs(self): + with patch.object(cpu_optimize, "_run_olive_pipeline") as run_pipeline: + cpu_optimize.run_olive_pipelines("output", "model", "int8") + + self.assertEqual( + [call.args[0] for call in run_pipeline.call_args_list], + [ + "nemotron_encoder_int8_cpu.json", + "nemotron_decoder_fp32_cpu.json", + "nemotron_joint_fp32_cpu.json", + ], + ) + + def test_trtrtx_pipeline_uses_only_trtrtx_configs(self): + with patch.object(trt_rtx_optimize, "_run_olive_pipeline") as run_pipeline: + trt_rtx_optimize.run_olive_pipelines("output", "model") + + self.assertEqual( + [call.args[0] for call in run_pipeline.call_args_list], + [ + "nemotron_encoder_fp16_trtrtx.json", + "nemotron_decoder_fp16_trtrtx.json", + "nemotron_joint_fp16_trtrtx.json", + ], + ) + + def test_cpu_entry_point_runs_only_cpu_pipeline(self): + model_load = types.ModuleType("cpu.nemotron_model_load") + model_load.MODEL_NAME = "nvidia/test-model" + model_load.CHUNK_SIZE = 0.16 + + with ( + patch.dict(sys.modules, {"cpu.nemotron_model_load": model_load}), + patch.object(sys, "argv", ["optimize.py"]), + patch.object(cpu_optimize, "run_olive_pipelines") as run_olive_pipelines, + patch.object(cpu_optimize, "run_tokenizer_export"), + patch.object(cpu_optimize, "generate_configs") as generate_configs, + patch.object(cpu_optimize, "download_silero_vad"), + ): + cpu_optimize.main() + + self.assertEqual( + run_olive_pipelines.call_args.kwargs["encoder_precision"], + "int4", + ) + self.assertTrue(generate_configs.call_args.kwargs["include_vad"]) + + def test_trtrtx_entry_point_runs_only_trtrtx_pipeline(self): + model_load = types.ModuleType("NvTensorRtRtx.nemotron_model_load") + model_load.MODEL_NAME = "nvidia/test-model" + model_load.CHUNK_SIZE = 0.16 + + with ( + patch.dict(sys.modules, {"NvTensorRtRtx.nemotron_model_load": model_load}), + patch.object(sys, "argv", ["optimize.py"]), + patch.object(trt_rtx_optimize, "run_olive_pipelines") as run_olive_pipelines, + patch.object(trt_rtx_optimize, "run_tokenizer_export"), + patch.object(trt_rtx_optimize, "generate_configs") as generate_configs, + ): + trt_rtx_optimize.main() + + self.assertEqual( + run_olive_pipelines.call_args.kwargs, + { + "output_dir": trt_rtx_optimize.DEFAULT_OUTPUT_DIR, + "model_path": "nvidia/test-model", + }, + ) + self.assertFalse(generate_configs.call_args.kwargs["include_vad"]) + + def test_provider_readmes_document_vad_behavior(self): + cpu_readme = (CPU_DIR / "README.md").read_text() + trtrtx_readme = (TRT_RTX_DIR / "README.md").read_text() + + self.assertIn("downloads Silero VAD", cpu_readme) + self.assertIn("Silero VAD is omitted", trtrtx_readme) + + def test_tokenizer_export_restores_local_checkpoint_on_cpu(self): + export_tokenizer = _load_export_tokenizer_module() + + calls = [] + + module_names = [ + "nemo", + "nemo.collections", + "nemo.collections.asr", + ] + saved_modules = {name: sys.modules.get(name) for name in module_names} + for name in module_names: + sys.modules.pop(name, None) + + try: + nemo_module = types.ModuleType("nemo") + nemo_module.__path__ = [] + collections_module = types.ModuleType("nemo.collections") + collections_module.__path__ = [] + asr_module = types.ModuleType("nemo.collections.asr") + + class DummyTokenizer: + def ids_to_tokens(self, ids): + return [f"tok_{ids[0]}"] + + class DummyModel: + tokenizer = DummyTokenizer() + cfg = SimpleNamespace(joint=SimpleNamespace(num_classes=1)) + + class ASRModel: + @staticmethod + def restore_from(model_name, **kwargs): + calls.append(("restore", model_name, kwargs)) + return DummyModel() + + asr_module.models = SimpleNamespace(ASRModel=ASRModel) + nemo_module.collections = collections_module + collections_module.asr = asr_module + sys.modules["nemo"] = nemo_module + sys.modules["nemo.collections"] = collections_module + sys.modules["nemo.collections.asr"] = asr_module + + with tempfile.TemporaryDirectory() as tmpdir: + tokens = export_tokenizer.extract_vocab("local.nemo", Path(tmpdir)) + + self.assertEqual(calls, [("restore", "local.nemo", {"map_location": "cpu"})]) + self.assertEqual(tokens, ["tok_0", ""]) + finally: + for name in module_names: + if saved_modules[name] is None: + sys.modules.pop(name, None) + else: + sys.modules[name] = saved_modules[name] + + def test_fp16_encoder_dummy_inputs_expose_fp16_floating_tensors(self): + import torch + + model_load = _load_model_load_module() + inputs = model_load.encoder_fp16_dummy_inputs(None) + + self.assertEqual(inputs[0].dtype, torch.float16) + self.assertEqual(inputs[2].dtype, torch.float16) + self.assertEqual(inputs[3].dtype, torch.float16) + self.assertEqual(inputs[1].dtype, torch.int64) + self.assertEqual(inputs[4].dtype, torch.int64) + self.assertEqual(inputs[5].dtype, torch.int64) + + def test_fp16_decoder_and_joint_dummy_inputs_expose_fp16_floating_tensors(self): + import torch + + model_load = _load_model_load_module() + decoder_inputs = model_load.decoder_fp16_dummy_inputs(None) + joint_inputs = model_load.joint_fp16_dummy_inputs(None) + + self.assertEqual(decoder_inputs[0].dtype, torch.int64) + self.assertEqual(decoder_inputs[1].dtype, torch.float16) + self.assertEqual(decoder_inputs[2].dtype, torch.float16) + self.assertTrue(all(value.dtype == torch.float16 for value in joint_inputs)) + + def test_fp16_component_loaders_convert_decoder_and_joint(self): + import torch + import torch.nn as nn + + model_load = _load_model_load_module() + decoder_wrapper = nn.Linear(2, 2) + joint_wrapper = nn.Linear(2, 2) + with ( + patch.object(model_load, "decoder_model_loader", return_value=decoder_wrapper), + patch.object(model_load, "joint_model_loader", return_value=joint_wrapper), + ): + decoder = model_load.decoder_fp16_model_loader("unused") + joint = model_load.joint_fp16_model_loader("unused") + + self.assertEqual(next(decoder.parameters()).dtype, torch.float16) + self.assertEqual(next(joint.parameters()).dtype, torch.float16) + + def test_configure_encoder_for_sdpa_enables_each_attention_module(self): + model_load = _load_model_load_module() + attentions = [SimpleNamespace(use_pytorch_sdpa=False) for _ in range(2)] + encoder = SimpleNamespace(layers=[SimpleNamespace(self_attn=attention) for attention in attentions]) + + model_load.configure_encoder_for_sdpa(encoder) + + self.assertTrue(all(attention.use_pytorch_sdpa for attention in attentions)) + + def test_fp16_encoder_wrapper_preserves_fp16_public_outputs(self): + import torch + import torch.nn as nn + + model_load = _load_model_load_module() + + class FakeEncoder(nn.Module): + def __init__(self): + super().__init__() + self.weight = nn.Parameter(torch.ones((), dtype=torch.float16)) + + def forward_for_export(self, audio_signal, length, cache_last_channel, cache_last_time, cache_last_channel_len): + return audio_signal, length, cache_last_channel, cache_last_time, cache_last_channel_len + + class FakePromptKernel(nn.Module): + def __init__(self): + super().__init__() + self.weight = nn.Parameter(torch.ones((), dtype=torch.float16)) + + def forward(self, value): + return value[..., :model_load.D_MODEL] + + wrapper = model_load.StreamingEncoderWrapper(FakeEncoder(), FakePromptKernel()) + outputs = wrapper( + torch.zeros(1, 1, model_load.D_MODEL, dtype=torch.float16), + torch.ones(1, dtype=torch.int64), + torch.zeros(1, 1, 1, model_load.D_MODEL, dtype=torch.float16), + torch.zeros(1, 1, model_load.D_MODEL, 1, dtype=torch.float16), + torch.zeros(1, dtype=torch.int64), + torch.zeros(1, dtype=torch.int64), + ) + + self.assertEqual(outputs[0].dtype, torch.float16) + self.assertEqual(outputs[2].dtype, torch.float16) + self.assertEqual(outputs[3].dtype, torch.float16) + +if __name__ == "__main__": + unittest.main() diff --git a/nvidia-nemotron-asr-streaming-multilingual-0.6b/src/README.md b/nvidia-nemotron-asr-streaming-multilingual-0.6b/src/README.md deleted file mode 100644 index 522984908..000000000 --- a/nvidia-nemotron-asr-streaming-multilingual-0.6b/src/README.md +++ /dev/null @@ -1,93 +0,0 @@ -# Nemotron 3.5 ASR Streaming Multilingual 0.6B (INT4, CPU/CUDA) - -This recipe exports **nvidia/NVIDIA-Nemotron-3.5-ASR-Streaming-Multilingual-0.6b** -(100+ languages) to ONNX, optimizes the encoder, and produces deployment-ready -artifacts for `onnxruntime-genai`. - -All model components are handled through Olive's declarative pass system: -- **Encoder**: OnnxConversion → OnnxKQuantQuantization (INT4 default, or INT8 dynamic) -- **Decoder**: OnnxConversion (FP32) -- **Joint**: OnnxConversion (FP32) - -## Files -- `src/nemotron_encoder_int4_cpu.json` – Olive encoder config (convert → INT4 k-quant) -- `src/nemotron_encoder_int8_cpu.json` – Olive encoder config (convert → INT8 dynamic, optional) -- `src/nemotron_decoder_fp32_cpu.json` – Olive decoder config (convert only) -- `src/nemotron_joint_fp32_cpu.json` – Olive joint config (convert only) -- `src/nemotron_model_load.py` – model loaders + dummy inputs for all components -- `src/optimize.py` – full pipeline script (Olive × 3 + tokenizer + configs + VAD) -- `scripts/export_tokenizer.py` – tokenizer export - -## Setup -From repo root: - -```bash -python -m venv .venv -source .venv/bin/activate -pip install -r nvidia-nemotron-asr-streaming-multilingual-0.6b/src/requirements.txt -``` - -## Run - -From the `nvidia-nemotron-asr-streaming-multilingual-0.6b` directory: - -```bash -cd nvidia-nemotron-asr-streaming-multilingual-0.6b - -# Full pipeline, INT4 encoder (default) -python src/optimize.py - -# Or INT8 encoder -python src/optimize.py --encoder-precision int8 - -# Custom output directory -python src/optimize.py --output-dir build/multilingual_onnx_int4 -``` - -This runs the full pipeline: -1. **Encoder** — Olive: OnnxConversion → INT4/INT8 quantization -2. **Decoder** — Olive: OnnxConversion (FP32) -3. **Joint** — Olive: OnnxConversion (FP32) -4. **Tokenizer** — exports vocab + tokenizer.json -5. **Configs** — generates genai_config.json + audio_processor_config.json -6. **VAD** — downloads Silero VAD ONNX model - -Or run individual components directly with Olive CLI: - -```bash -python -m olive run --config src/nemotron_encoder_int4_cpu.json -python -m olive run --config src/nemotron_encoder_int8_cpu.json -python -m olive run --config src/nemotron_decoder_fp32_cpu.json -python -m olive run --config src/nemotron_joint_fp32_cpu.json -``` - -## Output -Expected optimized artifacts in `src/build/onnx_models_int4/` (default output directory): -- `encoder.onnx` (INT4 k-quant, ~660 MB) -- `decoder.onnx` (FP32, ~57 MB) -- `joint.onnx` (FP32, ~36 MB) -- `silero_vad.onnx` (~2 MB) -- `genai_config.json` -- `audio_processor_config.json` -- `model_config.json` -- `tokenizer.json` -- `tokenizer_config.json` -- `vocab.txt` - -Total size: ~760 MB (INT4 encoder). - -## Inference - -Use the multilingual-aware example from `onnxruntime-genai`, passing a `--language` code: - -```bash -python onnxruntime-genai/examples/python/nemotron_speech.py \ - --model_path src/build/onnx_models_int4 \ - --audio_file path/to/audio.wav \ - --language de \ - -e cpu -``` - -Supported language codes match the NeMo multilingual prompt schema -(e.g. `en`, `de`, `fr`, `es`, `pt`, `it`, `nl`, `pl`, `zh-CN`, `ja-JP`, `auto`, …). -The full mapping is printed via `--help`. diff --git a/nvidia-nemotron-asr-streaming-multilingual-0.6b/src/info.yaml b/nvidia-nemotron-asr-streaming-multilingual-0.6b/src/info.yaml deleted file mode 100644 index 8ec225c76..000000000 --- a/nvidia-nemotron-asr-streaming-multilingual-0.6b/src/info.yaml +++ /dev/null @@ -1,22 +0,0 @@ -name: nvidia-nemotron-asr-streaming-multilingual-0.6b -provider: NVIDIA -model_id: nvidia/NVIDIA-Nemotron-3.5-ASR-Streaming-Multilingual-0.6b -task: automatic-speech-recognition -framework: ONNX Runtime -execution_provider: - - CPUExecutionProvider - - CUDAExecutionProvider -summary: > - Recipe for exporting the multilingual NVIDIA Nemotron 3.5 ASR Streaming RNNT model - (0.6B params, 100+ languages) to ONNX and optimizing encoder weights to INT4 - (k_quant) for low-memory CPU/CUDA inference. - -artifacts: - - src/nemotron_encoder_int4_cpu.json - - src/nemotron_encoder_int8_cpu.json - - src/nemotron_decoder_fp32_cpu.json - - src/nemotron_joint_fp32_cpu.json - - src/optimize.py - -scripts: - - scripts/export_tokenizer.py From 851b284ad801e7a84dbc242e38307f8f4fab0351 Mon Sep 17 00:00:00 2001 From: David Fan <30608893+jiafatom@users.noreply.github.com> Date: Wed, 22 Jul 2026 13:42:16 -0700 Subject: [PATCH 05/10] Add int8 embedding + mixed int4/int8 decoder quantization to gemma-4-E2B mixed recipe (#558) Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> --- google-gemma-4-E2B-it/README.md | 46 +- .../cpu/mixed/embedding.json | 19 + google-gemma-4-E2B-it/cpu/mixed/text.json | 425 +++++++++++++++++ .../cuda/mixed/embedding.json | 22 + google-gemma-4-E2B-it/cuda/mixed/text.json | 434 +++++++++++++++++- 5 files changed, 937 insertions(+), 9 deletions(-) create mode 100644 google-gemma-4-E2B-it/cpu/mixed/embedding.json create mode 100644 google-gemma-4-E2B-it/cuda/mixed/embedding.json diff --git a/google-gemma-4-E2B-it/README.md b/google-gemma-4-E2B-it/README.md index 7fa82a4a2..be7e3c30f 100644 --- a/google-gemma-4-E2B-it/README.md +++ b/google-gemma-4-E2B-it/README.md @@ -36,25 +36,53 @@ Install ONNX Runtime GenAI: | `cuda/fp16/config.json` | `MobiusBuilder(fp16)` | `cuda/fp16/models` | | `cuda/int4/config.json` | `MobiusBuilder(fp16)` → `OnnxKQuantQuantization(bits=4, block=32)` | `cuda/int4/models` | -### Mixed quantization (separate text / vision / audio) +### Mixed quantization (separate text / vision / audio / embedding) These recipes split the model into components with per-component -quantization — int4 for the text decoder, int8 for vision and audio -encoders — for better accuracy vs. latency trade-offs. +quantization — a **mixed int4/int8 text decoder**, int8 for the vision and +audio encoders, and int8 for the token embedding — for better accuracy vs. +latency/size trade-offs. + +**Mixed-bit decoder**: the decoder is int4 K-Quant by default, but the most +quantization-sensitive weights are upcast to int8 via the +`customized_weight_config` in `text.json`. This targets, in every one of the +35 transformer layers, the `down_proj`, `gate_proj`, `up_proj`, and +`o_proj` MatMuls plus the global `lm_head` (141 weights total → int8; the +remaining `q/k/v_proj` and per-layer gates stay int4). Empirically these +nodes carry most of the int4 accuracy loss, so upcasting only them recovers +most of the fp16 quality for a small size cost (CUDA decoder 1.41 GB pure-int4 +→ 2.50 GB mixed). + +Validation (CUDA, full eval sets), mixed int4/int8 decoder vs. the pure-int4 +decoder baseline: + +| Metric | int4 decoder | mixed int4/int8 decoder | +|---|---|---| +| AI2D exact_match (3,088) | 57.7% | 62.86% (+5.2) | +| FLEURS en_us strict WER (647) | 9.48% | 8.94% (−0.54) | +| MMLU 5-shot (14,042) | — | 60.09% (PT bf16 ref 60.80%) | +| decoder size | 1.41 GB | 2.50 GB | + +> Note: `customized_weight_config` keys are exact exported node names +> (e.g. `.../down_proj/MatMul_node_124`). These are deterministic for a given +> MobiusBuilder export but can shift if the export graph changes; regenerate +> the config against the current export if node names move. | Recipe | Pipeline | Output dir | |---|---|---| | `cpu/mixed/export.json` | `MobiusBuilder(fp32)` — export all components | `cpu/mixed/models` | -| `cpu/mixed/text.json` | `OnnxKQuantQuantization(int4, block=32)` — quantize decoder | `cpu/mixed/models/decoder` | +| `cpu/mixed/text.json` | `OnnxKQuantQuantization(int4, block=32)` + int8 upcast of sensitive weights — quantize decoder | `cpu/mixed/models/decoder` | | `cpu/mixed/vision.json` | `OnnxBlockWiseRtnQuantization(int8, block=128)` — quantize vision encoder | `cpu/mixed/models/vision_encoder` | | `cpu/mixed/audio.json` | `OnnxBlockWiseRtnQuantization(int8, block=128)` — quantize audio encoder | `cpu/mixed/models/audio_encoder` | +| `cpu/mixed/embedding.json` | `OnnxBlockWiseRtnQuantization(int8, block=128)` — quantize token embedding | `cpu/mixed/models/embedding` | | `cuda/mixed/export.json` | `MobiusBuilder(fp16)` — export all components | `cuda/mixed/models` | -| `cuda/mixed/text.json` | `OnnxKQuantQuantization(int4, block=32)` — quantize decoder | `cuda/mixed/models/decoder` | +| `cuda/mixed/text.json` | `OnnxKQuantQuantization(int4, block=32)` + int8 upcast of sensitive weights — quantize decoder | `cuda/mixed/models/decoder` | | `cuda/mixed/vision.json` | `OnnxBlockWiseRtnQuantization(int8, block=32)` — quantize vision encoder | `cuda/mixed/models/vision_encoder` | | `cuda/mixed/audio.json` | `OnnxBlockWiseRtnQuantization(int8, block=32)` — quantize audio encoder | `cuda/mixed/models/audio_encoder` | +| `cuda/mixed/embedding.json` | `OnnxBlockWiseRtnQuantization(int8, block=32)` — quantize token embedding | `cuda/mixed/models/embedding` | -**Run order**: export first, then text, vision, and audio (the latter three -can run in parallel): +**Run order**: export first, then text, vision, audio, and embedding (the +latter four can run in parallel): ```bash # CPU mixed @@ -62,12 +90,14 @@ olive run --config cpu/mixed/export.json olive run --config cpu/mixed/text.json olive run --config cpu/mixed/vision.json olive run --config cpu/mixed/audio.json +olive run --config cpu/mixed/embedding.json # CUDA mixed olive run --config cuda/mixed/export.json olive run --config cuda/mixed/text.json olive run --config cuda/mixed/vision.json olive run --config cuda/mixed/audio.json +olive run --config cuda/mixed/embedding.json ``` K-Quant (Q4_K_M) is significantly faster with GPU acceleration — @@ -117,7 +147,7 @@ python inference.py --variant int4 --prompt "Hello" # CUDA INT4 python inference.py --device gpu --variant int4 --prompt "Explain quantum computing" -# CUDA mixed (int4 decoder + int8 vision/audio) +# CUDA mixed (mixed int4/int8 decoder + int8 vision/audio/embedding) python inference.py --device gpu --variant mixed --prompt "Explain quantum computing" # Interactive mode diff --git a/google-gemma-4-E2B-it/cpu/mixed/embedding.json b/google-gemma-4-E2B-it/cpu/mixed/embedding.json new file mode 100644 index 000000000..6418793cd --- /dev/null +++ b/google-gemma-4-E2B-it/cpu/mixed/embedding.json @@ -0,0 +1,19 @@ +{ + "input_model": { + "type": "ONNXModel", + "model_path": "cpu/mixed/models/embedding/model.onnx" + }, + "passes": { + "int8": { + "type": "OnnxBlockWiseRtnQuantization", + "bits": 8, + "block_size": 128, + "is_symmetric": true, + "accuracy_level": 4, + "save_as_external_data": true, + "external_data_name": "model.onnx.data" + } + }, + "no_artifacts": true, + "output_dir": "cpu/mixed/models/embedding" +} diff --git a/google-gemma-4-E2B-it/cpu/mixed/text.json b/google-gemma-4-E2B-it/cpu/mixed/text.json index ebf96a466..e36e5f179 100644 --- a/google-gemma-4-E2B-it/cpu/mixed/text.json +++ b/google-gemma-4-E2B-it/cpu/mixed/text.json @@ -8,6 +8,431 @@ "type": "OnnxKQuantQuantization", "bits": 4, "block_size": 32, + "customized_weight_config": { + "decoder/model/layers.0/mlp/down_proj/MatMul_node_120": { + "bits": 8 + }, + "decoder/model/layers.1/mlp/down_proj/MatMul_node_169": { + "bits": 8 + }, + "decoder/model/layers.2/mlp/down_proj/MatMul_node_218": { + "bits": 8 + }, + "decoder/model/layers.3/mlp/down_proj/MatMul_node_267": { + "bits": 8 + }, + "decoder/model/layers.4/mlp/down_proj/MatMul_node_316": { + "bits": 8 + }, + "decoder/model/layers.5/mlp/down_proj/MatMul_node_365": { + "bits": 8 + }, + "decoder/model/layers.6/mlp/down_proj/MatMul_node_414": { + "bits": 8 + }, + "decoder/model/layers.7/mlp/down_proj/MatMul_node_463": { + "bits": 8 + }, + "decoder/model/layers.8/mlp/down_proj/MatMul_node_512": { + "bits": 8 + }, + "decoder/model/layers.9/mlp/down_proj/MatMul_node_561": { + "bits": 8 + }, + "decoder/model/layers.10/mlp/down_proj/MatMul_node_610": { + "bits": 8 + }, + "decoder/model/layers.11/mlp/down_proj/MatMul_node_659": { + "bits": 8 + }, + "decoder/model/layers.12/mlp/down_proj/MatMul_node_708": { + "bits": 8 + }, + "decoder/model/layers.13/mlp/down_proj/MatMul_node_757": { + "bits": 8 + }, + "decoder/model/layers.14/mlp/down_proj/MatMul_node_806": { + "bits": 8 + }, + "decoder/model/layers.15/mlp/down_proj/MatMul_node_842": { + "bits": 8 + }, + "decoder/model/layers.16/mlp/down_proj/MatMul_node_878": { + "bits": 8 + }, + "decoder/model/layers.17/mlp/down_proj/MatMul_node_914": { + "bits": 8 + }, + "decoder/model/layers.18/mlp/down_proj/MatMul_node_950": { + "bits": 8 + }, + "decoder/model/layers.19/mlp/down_proj/MatMul_node_986": { + "bits": 8 + }, + "decoder/model/layers.20/mlp/down_proj/MatMul_node_1022": { + "bits": 8 + }, + "decoder/model/layers.21/mlp/down_proj/MatMul_node_1058": { + "bits": 8 + }, + "decoder/model/layers.22/mlp/down_proj/MatMul_node_1094": { + "bits": 8 + }, + "decoder/model/layers.23/mlp/down_proj/MatMul_node_1130": { + "bits": 8 + }, + "decoder/model/layers.24/mlp/down_proj/MatMul_node_1166": { + "bits": 8 + }, + "decoder/model/layers.25/mlp/down_proj/MatMul_node_1202": { + "bits": 8 + }, + "decoder/model/layers.26/mlp/down_proj/MatMul_node_1238": { + "bits": 8 + }, + "decoder/model/layers.27/mlp/down_proj/MatMul_node_1274": { + "bits": 8 + }, + "decoder/model/layers.28/mlp/down_proj/MatMul_node_1310": { + "bits": 8 + }, + "decoder/model/layers.29/mlp/down_proj/MatMul_node_1346": { + "bits": 8 + }, + "decoder/model/layers.30/mlp/down_proj/MatMul_node_1382": { + "bits": 8 + }, + "decoder/model/layers.31/mlp/down_proj/MatMul_node_1418": { + "bits": 8 + }, + "decoder/model/layers.32/mlp/down_proj/MatMul_node_1454": { + "bits": 8 + }, + "decoder/model/layers.33/mlp/down_proj/MatMul_node_1490": { + "bits": 8 + }, + "decoder/model/layers.34/mlp/down_proj/MatMul_node_1526": { + "bits": 8 + }, + "decoder/lm_head/MatMul_node_1540": { + "bits": 8 + }, + "decoder/model/layers.0/self_attn/o_proj/MatMul_node_109": { + "bits": 8 + }, + "decoder/model/layers.1/self_attn/o_proj/MatMul_node_158": { + "bits": 8 + }, + "decoder/model/layers.2/self_attn/o_proj/MatMul_node_207": { + "bits": 8 + }, + "decoder/model/layers.3/self_attn/o_proj/MatMul_node_256": { + "bits": 8 + }, + "decoder/model/layers.4/self_attn/o_proj/MatMul_node_305": { + "bits": 8 + }, + "decoder/model/layers.5/self_attn/o_proj/MatMul_node_354": { + "bits": 8 + }, + "decoder/model/layers.6/self_attn/o_proj/MatMul_node_403": { + "bits": 8 + }, + "decoder/model/layers.7/self_attn/o_proj/MatMul_node_452": { + "bits": 8 + }, + "decoder/model/layers.8/self_attn/o_proj/MatMul_node_501": { + "bits": 8 + }, + "decoder/model/layers.9/self_attn/o_proj/MatMul_node_550": { + "bits": 8 + }, + "decoder/model/layers.10/self_attn/o_proj/MatMul_node_599": { + "bits": 8 + }, + "decoder/model/layers.11/self_attn/o_proj/MatMul_node_648": { + "bits": 8 + }, + "decoder/model/layers.12/self_attn/o_proj/MatMul_node_697": { + "bits": 8 + }, + "decoder/model/layers.13/self_attn/o_proj/MatMul_node_746": { + "bits": 8 + }, + "decoder/model/layers.14/self_attn/o_proj/MatMul_node_795": { + "bits": 8 + }, + "decoder/model/layers.15/self_attn/o_proj/MatMul_node_831": { + "bits": 8 + }, + "decoder/model/layers.16/self_attn/o_proj/MatMul_node_867": { + "bits": 8 + }, + "decoder/model/layers.17/self_attn/o_proj/MatMul_node_903": { + "bits": 8 + }, + "decoder/model/layers.18/self_attn/o_proj/MatMul_node_939": { + "bits": 8 + }, + "decoder/model/layers.19/self_attn/o_proj/MatMul_node_975": { + "bits": 8 + }, + "decoder/model/layers.20/self_attn/o_proj/MatMul_node_1011": { + "bits": 8 + }, + "decoder/model/layers.21/self_attn/o_proj/MatMul_node_1047": { + "bits": 8 + }, + "decoder/model/layers.22/self_attn/o_proj/MatMul_node_1083": { + "bits": 8 + }, + "decoder/model/layers.23/self_attn/o_proj/MatMul_node_1119": { + "bits": 8 + }, + "decoder/model/layers.24/self_attn/o_proj/MatMul_node_1155": { + "bits": 8 + }, + "decoder/model/layers.25/self_attn/o_proj/MatMul_node_1191": { + "bits": 8 + }, + "decoder/model/layers.26/self_attn/o_proj/MatMul_node_1227": { + "bits": 8 + }, + "decoder/model/layers.27/self_attn/o_proj/MatMul_node_1263": { + "bits": 8 + }, + "decoder/model/layers.28/self_attn/o_proj/MatMul_node_1299": { + "bits": 8 + }, + "decoder/model/layers.29/self_attn/o_proj/MatMul_node_1335": { + "bits": 8 + }, + "decoder/model/layers.30/self_attn/o_proj/MatMul_node_1371": { + "bits": 8 + }, + "decoder/model/layers.31/self_attn/o_proj/MatMul_node_1407": { + "bits": 8 + }, + "decoder/model/layers.32/self_attn/o_proj/MatMul_node_1443": { + "bits": 8 + }, + "decoder/model/layers.33/self_attn/o_proj/MatMul_node_1479": { + "bits": 8 + }, + "decoder/model/layers.34/self_attn/o_proj/MatMul_node_1515": { + "bits": 8 + }, + "decoder/model/layers.0/mlp/gate_proj/MatMul_node_114": { + "bits": 8 + }, + "decoder/model/layers.1/mlp/gate_proj/MatMul_node_163": { + "bits": 8 + }, + "decoder/model/layers.2/mlp/gate_proj/MatMul_node_212": { + "bits": 8 + }, + "decoder/model/layers.3/mlp/gate_proj/MatMul_node_261": { + "bits": 8 + }, + "decoder/model/layers.4/mlp/gate_proj/MatMul_node_310": { + "bits": 8 + }, + "decoder/model/layers.5/mlp/gate_proj/MatMul_node_359": { + "bits": 8 + }, + "decoder/model/layers.6/mlp/gate_proj/MatMul_node_408": { + "bits": 8 + }, + "decoder/model/layers.7/mlp/gate_proj/MatMul_node_457": { + "bits": 8 + }, + "decoder/model/layers.8/mlp/gate_proj/MatMul_node_506": { + "bits": 8 + }, + "decoder/model/layers.9/mlp/gate_proj/MatMul_node_555": { + "bits": 8 + }, + "decoder/model/layers.10/mlp/gate_proj/MatMul_node_604": { + "bits": 8 + }, + "decoder/model/layers.11/mlp/gate_proj/MatMul_node_653": { + "bits": 8 + }, + "decoder/model/layers.12/mlp/gate_proj/MatMul_node_702": { + "bits": 8 + }, + "decoder/model/layers.13/mlp/gate_proj/MatMul_node_751": { + "bits": 8 + }, + "decoder/model/layers.14/mlp/gate_proj/MatMul_node_800": { + "bits": 8 + }, + "decoder/model/layers.15/mlp/gate_proj/MatMul_node_836": { + "bits": 8 + }, + "decoder/model/layers.16/mlp/gate_proj/MatMul_node_872": { + "bits": 8 + }, + "decoder/model/layers.17/mlp/gate_proj/MatMul_node_908": { + "bits": 8 + }, + "decoder/model/layers.18/mlp/gate_proj/MatMul_node_944": { + "bits": 8 + }, + "decoder/model/layers.19/mlp/gate_proj/MatMul_node_980": { + "bits": 8 + }, + "decoder/model/layers.20/mlp/gate_proj/MatMul_node_1016": { + "bits": 8 + }, + "decoder/model/layers.21/mlp/gate_proj/MatMul_node_1052": { + "bits": 8 + }, + "decoder/model/layers.22/mlp/gate_proj/MatMul_node_1088": { + "bits": 8 + }, + "decoder/model/layers.23/mlp/gate_proj/MatMul_node_1124": { + "bits": 8 + }, + "decoder/model/layers.24/mlp/gate_proj/MatMul_node_1160": { + "bits": 8 + }, + "decoder/model/layers.25/mlp/gate_proj/MatMul_node_1196": { + "bits": 8 + }, + "decoder/model/layers.26/mlp/gate_proj/MatMul_node_1232": { + "bits": 8 + }, + "decoder/model/layers.27/mlp/gate_proj/MatMul_node_1268": { + "bits": 8 + }, + "decoder/model/layers.28/mlp/gate_proj/MatMul_node_1304": { + "bits": 8 + }, + "decoder/model/layers.29/mlp/gate_proj/MatMul_node_1340": { + "bits": 8 + }, + "decoder/model/layers.30/mlp/gate_proj/MatMul_node_1376": { + "bits": 8 + }, + "decoder/model/layers.31/mlp/gate_proj/MatMul_node_1412": { + "bits": 8 + }, + "decoder/model/layers.32/mlp/gate_proj/MatMul_node_1448": { + "bits": 8 + }, + "decoder/model/layers.33/mlp/gate_proj/MatMul_node_1484": { + "bits": 8 + }, + "decoder/model/layers.34/mlp/gate_proj/MatMul_node_1520": { + "bits": 8 + }, + "decoder/model/layers.0/mlp/up_proj/MatMul_node_117": { + "bits": 8 + }, + "decoder/model/layers.1/mlp/up_proj/MatMul_node_166": { + "bits": 8 + }, + "decoder/model/layers.2/mlp/up_proj/MatMul_node_215": { + "bits": 8 + }, + "decoder/model/layers.3/mlp/up_proj/MatMul_node_264": { + "bits": 8 + }, + "decoder/model/layers.4/mlp/up_proj/MatMul_node_313": { + "bits": 8 + }, + "decoder/model/layers.5/mlp/up_proj/MatMul_node_362": { + "bits": 8 + }, + "decoder/model/layers.6/mlp/up_proj/MatMul_node_411": { + "bits": 8 + }, + "decoder/model/layers.7/mlp/up_proj/MatMul_node_460": { + "bits": 8 + }, + "decoder/model/layers.8/mlp/up_proj/MatMul_node_509": { + "bits": 8 + }, + "decoder/model/layers.9/mlp/up_proj/MatMul_node_558": { + "bits": 8 + }, + "decoder/model/layers.10/mlp/up_proj/MatMul_node_607": { + "bits": 8 + }, + "decoder/model/layers.11/mlp/up_proj/MatMul_node_656": { + "bits": 8 + }, + "decoder/model/layers.12/mlp/up_proj/MatMul_node_705": { + "bits": 8 + }, + "decoder/model/layers.13/mlp/up_proj/MatMul_node_754": { + "bits": 8 + }, + "decoder/model/layers.14/mlp/up_proj/MatMul_node_803": { + "bits": 8 + }, + "decoder/model/layers.15/mlp/up_proj/MatMul_node_839": { + "bits": 8 + }, + "decoder/model/layers.16/mlp/up_proj/MatMul_node_875": { + "bits": 8 + }, + "decoder/model/layers.17/mlp/up_proj/MatMul_node_911": { + "bits": 8 + }, + "decoder/model/layers.18/mlp/up_proj/MatMul_node_947": { + "bits": 8 + }, + "decoder/model/layers.19/mlp/up_proj/MatMul_node_983": { + "bits": 8 + }, + "decoder/model/layers.20/mlp/up_proj/MatMul_node_1019": { + "bits": 8 + }, + "decoder/model/layers.21/mlp/up_proj/MatMul_node_1055": { + "bits": 8 + }, + "decoder/model/layers.22/mlp/up_proj/MatMul_node_1091": { + "bits": 8 + }, + "decoder/model/layers.23/mlp/up_proj/MatMul_node_1127": { + "bits": 8 + }, + "decoder/model/layers.24/mlp/up_proj/MatMul_node_1163": { + "bits": 8 + }, + "decoder/model/layers.25/mlp/up_proj/MatMul_node_1199": { + "bits": 8 + }, + "decoder/model/layers.26/mlp/up_proj/MatMul_node_1235": { + "bits": 8 + }, + "decoder/model/layers.27/mlp/up_proj/MatMul_node_1271": { + "bits": 8 + }, + "decoder/model/layers.28/mlp/up_proj/MatMul_node_1307": { + "bits": 8 + }, + "decoder/model/layers.29/mlp/up_proj/MatMul_node_1343": { + "bits": 8 + }, + "decoder/model/layers.30/mlp/up_proj/MatMul_node_1379": { + "bits": 8 + }, + "decoder/model/layers.31/mlp/up_proj/MatMul_node_1415": { + "bits": 8 + }, + "decoder/model/layers.32/mlp/up_proj/MatMul_node_1451": { + "bits": 8 + }, + "decoder/model/layers.33/mlp/up_proj/MatMul_node_1487": { + "bits": 8 + }, + "decoder/model/layers.34/mlp/up_proj/MatMul_node_1523": { + "bits": 8 + } + }, "save_as_external_data": true } }, diff --git a/google-gemma-4-E2B-it/cuda/mixed/embedding.json b/google-gemma-4-E2B-it/cuda/mixed/embedding.json new file mode 100644 index 000000000..0eb68cfdd --- /dev/null +++ b/google-gemma-4-E2B-it/cuda/mixed/embedding.json @@ -0,0 +1,22 @@ +{ + "input_model": { + "type": "ONNXModel", + "model_path": "cuda/mixed/models/embedding/model.onnx" + }, + "passes": { + "int8": { + "type": "OnnxBlockWiseRtnQuantization", + "bits": 8, + "block_size": 32, + "is_symmetric": false, + "save_as_external_data": true, + "external_data_name": "model.onnx.data" + } + }, + "target": { + "type": "LocalSystem", + "accelerators": [{ "device": "gpu", "execution_providers": ["CUDAExecutionProvider"] }] + }, + "no_artifacts": true, + "output_dir": "cuda/mixed/models/embedding" +} diff --git a/google-gemma-4-E2B-it/cuda/mixed/text.json b/google-gemma-4-E2B-it/cuda/mixed/text.json index 33eb4500b..f408970ca 100644 --- a/google-gemma-4-E2B-it/cuda/mixed/text.json +++ b/google-gemma-4-E2B-it/cuda/mixed/text.json @@ -8,12 +8,444 @@ "type": "OnnxKQuantQuantization", "bits": 4, "block_size": 32, + "customized_weight_config": { + "decoder/model/layers.0/mlp/down_proj/MatMul_node_124": { + "bits": 8 + }, + "decoder/model/layers.1/mlp/down_proj/MatMul_node_173": { + "bits": 8 + }, + "decoder/model/layers.2/mlp/down_proj/MatMul_node_222": { + "bits": 8 + }, + "decoder/model/layers.3/mlp/down_proj/MatMul_node_271": { + "bits": 8 + }, + "decoder/model/layers.4/mlp/down_proj/MatMul_node_320": { + "bits": 8 + }, + "decoder/model/layers.5/mlp/down_proj/MatMul_node_369": { + "bits": 8 + }, + "decoder/model/layers.6/mlp/down_proj/MatMul_node_418": { + "bits": 8 + }, + "decoder/model/layers.7/mlp/down_proj/MatMul_node_467": { + "bits": 8 + }, + "decoder/model/layers.8/mlp/down_proj/MatMul_node_516": { + "bits": 8 + }, + "decoder/model/layers.9/mlp/down_proj/MatMul_node_565": { + "bits": 8 + }, + "decoder/model/layers.10/mlp/down_proj/MatMul_node_614": { + "bits": 8 + }, + "decoder/model/layers.11/mlp/down_proj/MatMul_node_663": { + "bits": 8 + }, + "decoder/model/layers.12/mlp/down_proj/MatMul_node_712": { + "bits": 8 + }, + "decoder/model/layers.13/mlp/down_proj/MatMul_node_761": { + "bits": 8 + }, + "decoder/model/layers.14/mlp/down_proj/MatMul_node_810": { + "bits": 8 + }, + "decoder/model/layers.15/mlp/down_proj/MatMul_node_846": { + "bits": 8 + }, + "decoder/model/layers.16/mlp/down_proj/MatMul_node_882": { + "bits": 8 + }, + "decoder/model/layers.17/mlp/down_proj/MatMul_node_918": { + "bits": 8 + }, + "decoder/model/layers.18/mlp/down_proj/MatMul_node_954": { + "bits": 8 + }, + "decoder/model/layers.19/mlp/down_proj/MatMul_node_990": { + "bits": 8 + }, + "decoder/model/layers.20/mlp/down_proj/MatMul_node_1026": { + "bits": 8 + }, + "decoder/model/layers.21/mlp/down_proj/MatMul_node_1062": { + "bits": 8 + }, + "decoder/model/layers.22/mlp/down_proj/MatMul_node_1098": { + "bits": 8 + }, + "decoder/model/layers.23/mlp/down_proj/MatMul_node_1134": { + "bits": 8 + }, + "decoder/model/layers.24/mlp/down_proj/MatMul_node_1170": { + "bits": 8 + }, + "decoder/model/layers.25/mlp/down_proj/MatMul_node_1206": { + "bits": 8 + }, + "decoder/model/layers.26/mlp/down_proj/MatMul_node_1242": { + "bits": 8 + }, + "decoder/model/layers.27/mlp/down_proj/MatMul_node_1278": { + "bits": 8 + }, + "decoder/model/layers.28/mlp/down_proj/MatMul_node_1314": { + "bits": 8 + }, + "decoder/model/layers.29/mlp/down_proj/MatMul_node_1350": { + "bits": 8 + }, + "decoder/model/layers.30/mlp/down_proj/MatMul_node_1386": { + "bits": 8 + }, + "decoder/model/layers.31/mlp/down_proj/MatMul_node_1422": { + "bits": 8 + }, + "decoder/model/layers.32/mlp/down_proj/MatMul_node_1458": { + "bits": 8 + }, + "decoder/model/layers.33/mlp/down_proj/MatMul_node_1494": { + "bits": 8 + }, + "decoder/model/layers.34/mlp/down_proj/MatMul_node_1530": { + "bits": 8 + }, + "decoder/lm_head/MatMul_node_1544": { + "bits": 8 + }, + "decoder/model/layers.0/self_attn/o_proj/MatMul_node_113": { + "bits": 8 + }, + "decoder/model/layers.1/self_attn/o_proj/MatMul_node_162": { + "bits": 8 + }, + "decoder/model/layers.2/self_attn/o_proj/MatMul_node_211": { + "bits": 8 + }, + "decoder/model/layers.3/self_attn/o_proj/MatMul_node_260": { + "bits": 8 + }, + "decoder/model/layers.4/self_attn/o_proj/MatMul_node_309": { + "bits": 8 + }, + "decoder/model/layers.5/self_attn/o_proj/MatMul_node_358": { + "bits": 8 + }, + "decoder/model/layers.6/self_attn/o_proj/MatMul_node_407": { + "bits": 8 + }, + "decoder/model/layers.7/self_attn/o_proj/MatMul_node_456": { + "bits": 8 + }, + "decoder/model/layers.8/self_attn/o_proj/MatMul_node_505": { + "bits": 8 + }, + "decoder/model/layers.9/self_attn/o_proj/MatMul_node_554": { + "bits": 8 + }, + "decoder/model/layers.10/self_attn/o_proj/MatMul_node_603": { + "bits": 8 + }, + "decoder/model/layers.11/self_attn/o_proj/MatMul_node_652": { + "bits": 8 + }, + "decoder/model/layers.12/self_attn/o_proj/MatMul_node_701": { + "bits": 8 + }, + "decoder/model/layers.13/self_attn/o_proj/MatMul_node_750": { + "bits": 8 + }, + "decoder/model/layers.14/self_attn/o_proj/MatMul_node_799": { + "bits": 8 + }, + "decoder/model/layers.15/self_attn/o_proj/MatMul_node_835": { + "bits": 8 + }, + "decoder/model/layers.16/self_attn/o_proj/MatMul_node_871": { + "bits": 8 + }, + "decoder/model/layers.17/self_attn/o_proj/MatMul_node_907": { + "bits": 8 + }, + "decoder/model/layers.18/self_attn/o_proj/MatMul_node_943": { + "bits": 8 + }, + "decoder/model/layers.19/self_attn/o_proj/MatMul_node_979": { + "bits": 8 + }, + "decoder/model/layers.20/self_attn/o_proj/MatMul_node_1015": { + "bits": 8 + }, + "decoder/model/layers.21/self_attn/o_proj/MatMul_node_1051": { + "bits": 8 + }, + "decoder/model/layers.22/self_attn/o_proj/MatMul_node_1087": { + "bits": 8 + }, + "decoder/model/layers.23/self_attn/o_proj/MatMul_node_1123": { + "bits": 8 + }, + "decoder/model/layers.24/self_attn/o_proj/MatMul_node_1159": { + "bits": 8 + }, + "decoder/model/layers.25/self_attn/o_proj/MatMul_node_1195": { + "bits": 8 + }, + "decoder/model/layers.26/self_attn/o_proj/MatMul_node_1231": { + "bits": 8 + }, + "decoder/model/layers.27/self_attn/o_proj/MatMul_node_1267": { + "bits": 8 + }, + "decoder/model/layers.28/self_attn/o_proj/MatMul_node_1303": { + "bits": 8 + }, + "decoder/model/layers.29/self_attn/o_proj/MatMul_node_1339": { + "bits": 8 + }, + "decoder/model/layers.30/self_attn/o_proj/MatMul_node_1375": { + "bits": 8 + }, + "decoder/model/layers.31/self_attn/o_proj/MatMul_node_1411": { + "bits": 8 + }, + "decoder/model/layers.32/self_attn/o_proj/MatMul_node_1447": { + "bits": 8 + }, + "decoder/model/layers.33/self_attn/o_proj/MatMul_node_1483": { + "bits": 8 + }, + "decoder/model/layers.34/self_attn/o_proj/MatMul_node_1519": { + "bits": 8 + }, + "decoder/model/layers.0/mlp/gate_proj/MatMul_node_118": { + "bits": 8 + }, + "decoder/model/layers.1/mlp/gate_proj/MatMul_node_167": { + "bits": 8 + }, + "decoder/model/layers.2/mlp/gate_proj/MatMul_node_216": { + "bits": 8 + }, + "decoder/model/layers.3/mlp/gate_proj/MatMul_node_265": { + "bits": 8 + }, + "decoder/model/layers.4/mlp/gate_proj/MatMul_node_314": { + "bits": 8 + }, + "decoder/model/layers.5/mlp/gate_proj/MatMul_node_363": { + "bits": 8 + }, + "decoder/model/layers.6/mlp/gate_proj/MatMul_node_412": { + "bits": 8 + }, + "decoder/model/layers.7/mlp/gate_proj/MatMul_node_461": { + "bits": 8 + }, + "decoder/model/layers.8/mlp/gate_proj/MatMul_node_510": { + "bits": 8 + }, + "decoder/model/layers.9/mlp/gate_proj/MatMul_node_559": { + "bits": 8 + }, + "decoder/model/layers.10/mlp/gate_proj/MatMul_node_608": { + "bits": 8 + }, + "decoder/model/layers.11/mlp/gate_proj/MatMul_node_657": { + "bits": 8 + }, + "decoder/model/layers.12/mlp/gate_proj/MatMul_node_706": { + "bits": 8 + }, + "decoder/model/layers.13/mlp/gate_proj/MatMul_node_755": { + "bits": 8 + }, + "decoder/model/layers.14/mlp/gate_proj/MatMul_node_804": { + "bits": 8 + }, + "decoder/model/layers.15/mlp/gate_proj/MatMul_node_840": { + "bits": 8 + }, + "decoder/model/layers.16/mlp/gate_proj/MatMul_node_876": { + "bits": 8 + }, + "decoder/model/layers.17/mlp/gate_proj/MatMul_node_912": { + "bits": 8 + }, + "decoder/model/layers.18/mlp/gate_proj/MatMul_node_948": { + "bits": 8 + }, + "decoder/model/layers.19/mlp/gate_proj/MatMul_node_984": { + "bits": 8 + }, + "decoder/model/layers.20/mlp/gate_proj/MatMul_node_1020": { + "bits": 8 + }, + "decoder/model/layers.21/mlp/gate_proj/MatMul_node_1056": { + "bits": 8 + }, + "decoder/model/layers.22/mlp/gate_proj/MatMul_node_1092": { + "bits": 8 + }, + "decoder/model/layers.23/mlp/gate_proj/MatMul_node_1128": { + "bits": 8 + }, + "decoder/model/layers.24/mlp/gate_proj/MatMul_node_1164": { + "bits": 8 + }, + "decoder/model/layers.25/mlp/gate_proj/MatMul_node_1200": { + "bits": 8 + }, + "decoder/model/layers.26/mlp/gate_proj/MatMul_node_1236": { + "bits": 8 + }, + "decoder/model/layers.27/mlp/gate_proj/MatMul_node_1272": { + "bits": 8 + }, + "decoder/model/layers.28/mlp/gate_proj/MatMul_node_1308": { + "bits": 8 + }, + "decoder/model/layers.29/mlp/gate_proj/MatMul_node_1344": { + "bits": 8 + }, + "decoder/model/layers.30/mlp/gate_proj/MatMul_node_1380": { + "bits": 8 + }, + "decoder/model/layers.31/mlp/gate_proj/MatMul_node_1416": { + "bits": 8 + }, + "decoder/model/layers.32/mlp/gate_proj/MatMul_node_1452": { + "bits": 8 + }, + "decoder/model/layers.33/mlp/gate_proj/MatMul_node_1488": { + "bits": 8 + }, + "decoder/model/layers.34/mlp/gate_proj/MatMul_node_1524": { + "bits": 8 + }, + "decoder/model/layers.0/mlp/up_proj/MatMul_node_121": { + "bits": 8 + }, + "decoder/model/layers.1/mlp/up_proj/MatMul_node_170": { + "bits": 8 + }, + "decoder/model/layers.2/mlp/up_proj/MatMul_node_219": { + "bits": 8 + }, + "decoder/model/layers.3/mlp/up_proj/MatMul_node_268": { + "bits": 8 + }, + "decoder/model/layers.4/mlp/up_proj/MatMul_node_317": { + "bits": 8 + }, + "decoder/model/layers.5/mlp/up_proj/MatMul_node_366": { + "bits": 8 + }, + "decoder/model/layers.6/mlp/up_proj/MatMul_node_415": { + "bits": 8 + }, + "decoder/model/layers.7/mlp/up_proj/MatMul_node_464": { + "bits": 8 + }, + "decoder/model/layers.8/mlp/up_proj/MatMul_node_513": { + "bits": 8 + }, + "decoder/model/layers.9/mlp/up_proj/MatMul_node_562": { + "bits": 8 + }, + "decoder/model/layers.10/mlp/up_proj/MatMul_node_611": { + "bits": 8 + }, + "decoder/model/layers.11/mlp/up_proj/MatMul_node_660": { + "bits": 8 + }, + "decoder/model/layers.12/mlp/up_proj/MatMul_node_709": { + "bits": 8 + }, + "decoder/model/layers.13/mlp/up_proj/MatMul_node_758": { + "bits": 8 + }, + "decoder/model/layers.14/mlp/up_proj/MatMul_node_807": { + "bits": 8 + }, + "decoder/model/layers.15/mlp/up_proj/MatMul_node_843": { + "bits": 8 + }, + "decoder/model/layers.16/mlp/up_proj/MatMul_node_879": { + "bits": 8 + }, + "decoder/model/layers.17/mlp/up_proj/MatMul_node_915": { + "bits": 8 + }, + "decoder/model/layers.18/mlp/up_proj/MatMul_node_951": { + "bits": 8 + }, + "decoder/model/layers.19/mlp/up_proj/MatMul_node_987": { + "bits": 8 + }, + "decoder/model/layers.20/mlp/up_proj/MatMul_node_1023": { + "bits": 8 + }, + "decoder/model/layers.21/mlp/up_proj/MatMul_node_1059": { + "bits": 8 + }, + "decoder/model/layers.22/mlp/up_proj/MatMul_node_1095": { + "bits": 8 + }, + "decoder/model/layers.23/mlp/up_proj/MatMul_node_1131": { + "bits": 8 + }, + "decoder/model/layers.24/mlp/up_proj/MatMul_node_1167": { + "bits": 8 + }, + "decoder/model/layers.25/mlp/up_proj/MatMul_node_1203": { + "bits": 8 + }, + "decoder/model/layers.26/mlp/up_proj/MatMul_node_1239": { + "bits": 8 + }, + "decoder/model/layers.27/mlp/up_proj/MatMul_node_1275": { + "bits": 8 + }, + "decoder/model/layers.28/mlp/up_proj/MatMul_node_1311": { + "bits": 8 + }, + "decoder/model/layers.29/mlp/up_proj/MatMul_node_1347": { + "bits": 8 + }, + "decoder/model/layers.30/mlp/up_proj/MatMul_node_1383": { + "bits": 8 + }, + "decoder/model/layers.31/mlp/up_proj/MatMul_node_1419": { + "bits": 8 + }, + "decoder/model/layers.32/mlp/up_proj/MatMul_node_1455": { + "bits": 8 + }, + "decoder/model/layers.33/mlp/up_proj/MatMul_node_1491": { + "bits": 8 + }, + "decoder/model/layers.34/mlp/up_proj/MatMul_node_1527": { + "bits": 8 + } + }, "save_as_external_data": true } }, "target": { "type": "LocalSystem", - "accelerators": [{ "device": "gpu", "execution_providers": ["CUDAExecutionProvider"] }] + "accelerators": [ + { + "device": "gpu", + "execution_providers": [ + "CUDAExecutionProvider" + ] + } + ] }, "no_artifacts": true, "output_dir": "cuda/mixed/models/decoder" From 0ed9f84dcb08150c2f1470adf5c04f19ef9dd6bb Mon Sep 17 00:00:00 2001 From: Yen-Shi Wang <6960565+yen-shi@users.noreply.github.com> Date: Sat, 1 Aug 2026 03:31:27 +0800 Subject: [PATCH 06/10] Add Qwen3.5 and Qwen3.6 NvTensorRtRtx recipes (#565) Co-authored-by: Codex --- .../Qwen3.5-0.8B_model_builder_int4.json | 36 +++++++++++++++++++ Qwen-Qwen3.5-0.8B/NvTensorRtRtx/README.md | 27 ++++++++++++++ Qwen-Qwen3.5-0.8B/NvTensorRtRtx/info.yml | 6 ++++ .../Qwen3.5-27B_model_builder_int4.json | 36 +++++++++++++++++++ Qwen-Qwen3.5-27B/NvTensorRtRtx/README.md | 27 ++++++++++++++ Qwen-Qwen3.5-27B/NvTensorRtRtx/info.yml | 6 ++++ .../Qwen3.5-2B_model_builder_int4.json | 36 +++++++++++++++++++ Qwen-Qwen3.5-2B/NvTensorRtRtx/README.md | 27 ++++++++++++++ Qwen-Qwen3.5-2B/NvTensorRtRtx/info.yml | 6 ++++ .../Qwen3.5-35B-A3B_model_builder_int4.json | 36 +++++++++++++++++++ Qwen-Qwen3.5-35B-A3B/NvTensorRtRtx/README.md | 27 ++++++++++++++ Qwen-Qwen3.5-35B-A3B/NvTensorRtRtx/info.yml | 6 ++++ .../Qwen3.5-4B_model_builder_int4.json | 36 +++++++++++++++++++ Qwen-Qwen3.5-4B/NvTensorRtRtx/README.md | 27 ++++++++++++++ Qwen-Qwen3.5-4B/NvTensorRtRtx/info.yml | 6 ++++ .../Qwen3.5-9B_model_builder_int4.json | 36 +++++++++++++++++++ Qwen-Qwen3.5-9B/NvTensorRtRtx/README.md | 27 ++++++++++++++ Qwen-Qwen3.5-9B/NvTensorRtRtx/info.yml | 6 ++++ .../Qwen3.6-27B_model_builder_int4.json | 36 +++++++++++++++++++ Qwen-Qwen3.6-27B/NvTensorRtRtx/README.md | 28 +++++++++++++++ Qwen-Qwen3.6-27B/NvTensorRtRtx/info.yml | 6 ++++ .../Qwen3.6-35B-A3B_model_builder_int4.json | 36 +++++++++++++++++++ Qwen-Qwen3.6-35B-A3B/NvTensorRtRtx/README.md | 28 +++++++++++++++ Qwen-Qwen3.6-35B-A3B/NvTensorRtRtx/info.yml | 6 ++++ 24 files changed, 554 insertions(+) create mode 100644 Qwen-Qwen3.5-0.8B/NvTensorRtRtx/Qwen3.5-0.8B_model_builder_int4.json create mode 100644 Qwen-Qwen3.5-0.8B/NvTensorRtRtx/README.md create mode 100644 Qwen-Qwen3.5-0.8B/NvTensorRtRtx/info.yml create mode 100644 Qwen-Qwen3.5-27B/NvTensorRtRtx/Qwen3.5-27B_model_builder_int4.json create mode 100644 Qwen-Qwen3.5-27B/NvTensorRtRtx/README.md create mode 100644 Qwen-Qwen3.5-27B/NvTensorRtRtx/info.yml create mode 100644 Qwen-Qwen3.5-2B/NvTensorRtRtx/Qwen3.5-2B_model_builder_int4.json create mode 100644 Qwen-Qwen3.5-2B/NvTensorRtRtx/README.md create mode 100644 Qwen-Qwen3.5-2B/NvTensorRtRtx/info.yml create mode 100644 Qwen-Qwen3.5-35B-A3B/NvTensorRtRtx/Qwen3.5-35B-A3B_model_builder_int4.json create mode 100644 Qwen-Qwen3.5-35B-A3B/NvTensorRtRtx/README.md create mode 100644 Qwen-Qwen3.5-35B-A3B/NvTensorRtRtx/info.yml create mode 100644 Qwen-Qwen3.5-4B/NvTensorRtRtx/Qwen3.5-4B_model_builder_int4.json create mode 100644 Qwen-Qwen3.5-4B/NvTensorRtRtx/README.md create mode 100644 Qwen-Qwen3.5-4B/NvTensorRtRtx/info.yml create mode 100644 Qwen-Qwen3.5-9B/NvTensorRtRtx/Qwen3.5-9B_model_builder_int4.json create mode 100644 Qwen-Qwen3.5-9B/NvTensorRtRtx/README.md create mode 100644 Qwen-Qwen3.5-9B/NvTensorRtRtx/info.yml create mode 100644 Qwen-Qwen3.6-27B/NvTensorRtRtx/Qwen3.6-27B_model_builder_int4.json create mode 100644 Qwen-Qwen3.6-27B/NvTensorRtRtx/README.md create mode 100644 Qwen-Qwen3.6-27B/NvTensorRtRtx/info.yml create mode 100644 Qwen-Qwen3.6-35B-A3B/NvTensorRtRtx/Qwen3.6-35B-A3B_model_builder_int4.json create mode 100644 Qwen-Qwen3.6-35B-A3B/NvTensorRtRtx/README.md create mode 100644 Qwen-Qwen3.6-35B-A3B/NvTensorRtRtx/info.yml diff --git a/Qwen-Qwen3.5-0.8B/NvTensorRtRtx/Qwen3.5-0.8B_model_builder_int4.json b/Qwen-Qwen3.5-0.8B/NvTensorRtRtx/Qwen3.5-0.8B_model_builder_int4.json new file mode 100644 index 000000000..889999dd4 --- /dev/null +++ b/Qwen-Qwen3.5-0.8B/NvTensorRtRtx/Qwen3.5-0.8B_model_builder_int4.json @@ -0,0 +1,36 @@ +{ + "input_model": { + "type": "HfModel", + "model_path": "Qwen/Qwen3.5-0.8B", + "task": "text-classification" + }, + "systems": { + "local_system": { + "type": "LocalSystem", + "accelerators": [ + { + "device": "gpu", + "execution_providers": ["NvTensorRTRTXExecutionProvider"] + } + ] + } + }, + "engine": { + "target": "local_system" + }, + "passes": { + "mb": { + "type": "ModelBuilder", + "precision": "int4", + "int4_block_size": 32, + "int4_algo_config": "rtn", + "int4_is_symmetric": true, + "use_qdq": true, + "enable_cuda_graph": true, + "extra_options": { + "exclude_embeds": false + } + } + }, + "log_severity_level": 0 +} diff --git a/Qwen-Qwen3.5-0.8B/NvTensorRtRtx/README.md b/Qwen-Qwen3.5-0.8B/NvTensorRtRtx/README.md new file mode 100644 index 000000000..a2714eb48 --- /dev/null +++ b/Qwen-Qwen3.5-0.8B/NvTensorRtRtx/README.md @@ -0,0 +1,27 @@ +# Qwen3.5-0.8B optimization + +This folder contains an Olive recipe for exporting the text-only component of `Qwen/Qwen3.5-0.8B` for the +`NvTensorRTRTXExecutionProvider` (also known as the `NvTensorRtRtx` EP). + +## INT4 weight-only quantization + +The `Qwen3.5-0.8B_model_builder_int4.json` recipe uses the ONNX Runtime GenAI `ModelBuilder` to: + +1. Export a standalone text model by including the token embedding layer (`exclude_embeds=false`). +2. Apply symmetric INT4 RTN weight-only quantization with a block size of 32. +3. Export quantized matrix multiplications directly in INT4 QDQ format (`use_qdq=true`). + +The vision encoder is not exported. + +## Setup + +1. Install Olive. +2. Install a Transformers 5.x release that recognizes the `qwen3_5` architecture. +3. Install an ONNX Runtime GenAI package or build that supports Qwen3.5 hybrid models and the + `NvTensorRTRTXExecutionProvider`. + +## Run + +```bash +olive run --config Qwen3.5-0.8B_model_builder_int4.json +``` diff --git a/Qwen-Qwen3.5-0.8B/NvTensorRtRtx/info.yml b/Qwen-Qwen3.5-0.8B/NvTensorRtRtx/info.yml new file mode 100644 index 000000000..b49eebfca --- /dev/null +++ b/Qwen-Qwen3.5-0.8B/NvTensorRtRtx/info.yml @@ -0,0 +1,6 @@ +arch: qwen3_5_text +recipes: + - name: Qwen3.5-0.8B_Model_Builder_INT4 + file: Qwen3.5-0.8B_model_builder_int4.json + devices: gpu + eps: NvTensorRTRTXExecutionProvider diff --git a/Qwen-Qwen3.5-27B/NvTensorRtRtx/Qwen3.5-27B_model_builder_int4.json b/Qwen-Qwen3.5-27B/NvTensorRtRtx/Qwen3.5-27B_model_builder_int4.json new file mode 100644 index 000000000..3e472357a --- /dev/null +++ b/Qwen-Qwen3.5-27B/NvTensorRtRtx/Qwen3.5-27B_model_builder_int4.json @@ -0,0 +1,36 @@ +{ + "input_model": { + "type": "HfModel", + "model_path": "Qwen/Qwen3.5-27B", + "task": "text-classification" + }, + "systems": { + "local_system": { + "type": "LocalSystem", + "accelerators": [ + { + "device": "gpu", + "execution_providers": ["NvTensorRTRTXExecutionProvider"] + } + ] + } + }, + "engine": { + "target": "local_system" + }, + "passes": { + "mb": { + "type": "ModelBuilder", + "precision": "int4", + "int4_block_size": 32, + "int4_algo_config": "rtn", + "int4_is_symmetric": true, + "use_qdq": true, + "enable_cuda_graph": true, + "extra_options": { + "exclude_embeds": false + } + } + }, + "log_severity_level": 0 +} diff --git a/Qwen-Qwen3.5-27B/NvTensorRtRtx/README.md b/Qwen-Qwen3.5-27B/NvTensorRtRtx/README.md new file mode 100644 index 000000000..57f57bf9a --- /dev/null +++ b/Qwen-Qwen3.5-27B/NvTensorRtRtx/README.md @@ -0,0 +1,27 @@ +# Qwen3.5-27B optimization + +This folder contains an Olive recipe for exporting the text-only component of `Qwen/Qwen3.5-27B` for the +`NvTensorRTRTXExecutionProvider` (also known as the `NvTensorRtRtx` EP). + +## INT4 weight-only quantization + +The `Qwen3.5-27B_model_builder_int4.json` recipe uses the ONNX Runtime GenAI `ModelBuilder` to: + +1. Export a standalone text model by including the token embedding layer (`exclude_embeds=false`). +2. Apply symmetric INT4 RTN weight-only quantization with a block size of 32. +3. Export quantized matrix multiplications directly in INT4 QDQ format (`use_qdq=true`). + +The vision encoder is not exported. + +## Setup + +1. Install Olive. +2. Install a Transformers 5.x release that recognizes the `qwen3_5` architecture. +3. Install an ONNX Runtime GenAI package or build that supports Qwen3.5 hybrid models and the + `NvTensorRTRTXExecutionProvider`. + +## Run + +```bash +olive run --config Qwen3.5-27B_model_builder_int4.json +``` diff --git a/Qwen-Qwen3.5-27B/NvTensorRtRtx/info.yml b/Qwen-Qwen3.5-27B/NvTensorRtRtx/info.yml new file mode 100644 index 000000000..7d862358f --- /dev/null +++ b/Qwen-Qwen3.5-27B/NvTensorRtRtx/info.yml @@ -0,0 +1,6 @@ +arch: qwen3_5_text +recipes: + - name: Qwen3.5-27B_Model_Builder_INT4 + file: Qwen3.5-27B_model_builder_int4.json + devices: gpu + eps: NvTensorRTRTXExecutionProvider diff --git a/Qwen-Qwen3.5-2B/NvTensorRtRtx/Qwen3.5-2B_model_builder_int4.json b/Qwen-Qwen3.5-2B/NvTensorRtRtx/Qwen3.5-2B_model_builder_int4.json new file mode 100644 index 000000000..dce63e35e --- /dev/null +++ b/Qwen-Qwen3.5-2B/NvTensorRtRtx/Qwen3.5-2B_model_builder_int4.json @@ -0,0 +1,36 @@ +{ + "input_model": { + "type": "HfModel", + "model_path": "Qwen/Qwen3.5-2B", + "task": "text-classification" + }, + "systems": { + "local_system": { + "type": "LocalSystem", + "accelerators": [ + { + "device": "gpu", + "execution_providers": ["NvTensorRTRTXExecutionProvider"] + } + ] + } + }, + "engine": { + "target": "local_system" + }, + "passes": { + "mb": { + "type": "ModelBuilder", + "precision": "int4", + "int4_block_size": 32, + "int4_algo_config": "rtn", + "int4_is_symmetric": true, + "use_qdq": true, + "enable_cuda_graph": true, + "extra_options": { + "exclude_embeds": false + } + } + }, + "log_severity_level": 0 +} diff --git a/Qwen-Qwen3.5-2B/NvTensorRtRtx/README.md b/Qwen-Qwen3.5-2B/NvTensorRtRtx/README.md new file mode 100644 index 000000000..61df346d7 --- /dev/null +++ b/Qwen-Qwen3.5-2B/NvTensorRtRtx/README.md @@ -0,0 +1,27 @@ +# Qwen3.5-2B optimization + +This folder contains an Olive recipe for exporting the text-only component of `Qwen/Qwen3.5-2B` for the +`NvTensorRTRTXExecutionProvider` (also known as the `NvTensorRtRtx` EP). + +## INT4 weight-only quantization + +The `Qwen3.5-2B_model_builder_int4.json` recipe uses the ONNX Runtime GenAI `ModelBuilder` to: + +1. Export a standalone text model by including the token embedding layer (`exclude_embeds=false`). +2. Apply symmetric INT4 RTN weight-only quantization with a block size of 32. +3. Export quantized matrix multiplications directly in INT4 QDQ format (`use_qdq=true`). + +The vision encoder is not exported. + +## Setup + +1. Install Olive. +2. Install a Transformers 5.x release that recognizes the `qwen3_5` architecture. +3. Install an ONNX Runtime GenAI package or build that supports Qwen3.5 hybrid models and the + `NvTensorRTRTXExecutionProvider`. + +## Run + +```bash +olive run --config Qwen3.5-2B_model_builder_int4.json +``` diff --git a/Qwen-Qwen3.5-2B/NvTensorRtRtx/info.yml b/Qwen-Qwen3.5-2B/NvTensorRtRtx/info.yml new file mode 100644 index 000000000..b8c1df27c --- /dev/null +++ b/Qwen-Qwen3.5-2B/NvTensorRtRtx/info.yml @@ -0,0 +1,6 @@ +arch: qwen3_5_text +recipes: + - name: Qwen3.5-2B_Model_Builder_INT4 + file: Qwen3.5-2B_model_builder_int4.json + devices: gpu + eps: NvTensorRTRTXExecutionProvider diff --git a/Qwen-Qwen3.5-35B-A3B/NvTensorRtRtx/Qwen3.5-35B-A3B_model_builder_int4.json b/Qwen-Qwen3.5-35B-A3B/NvTensorRtRtx/Qwen3.5-35B-A3B_model_builder_int4.json new file mode 100644 index 000000000..15c911073 --- /dev/null +++ b/Qwen-Qwen3.5-35B-A3B/NvTensorRtRtx/Qwen3.5-35B-A3B_model_builder_int4.json @@ -0,0 +1,36 @@ +{ + "input_model": { + "type": "HfModel", + "model_path": "Qwen/Qwen3.5-35B-A3B", + "task": "text-classification" + }, + "systems": { + "local_system": { + "type": "LocalSystem", + "accelerators": [ + { + "device": "gpu", + "execution_providers": ["NvTensorRTRTXExecutionProvider"] + } + ] + } + }, + "engine": { + "target": "local_system" + }, + "passes": { + "mb": { + "type": "ModelBuilder", + "precision": "int4", + "int4_block_size": 32, + "int4_algo_config": "rtn", + "int4_is_symmetric": true, + "use_qdq": true, + "enable_cuda_graph": true, + "extra_options": { + "exclude_embeds": false + } + } + }, + "log_severity_level": 0 +} diff --git a/Qwen-Qwen3.5-35B-A3B/NvTensorRtRtx/README.md b/Qwen-Qwen3.5-35B-A3B/NvTensorRtRtx/README.md new file mode 100644 index 000000000..6def9aab9 --- /dev/null +++ b/Qwen-Qwen3.5-35B-A3B/NvTensorRtRtx/README.md @@ -0,0 +1,27 @@ +# Qwen3.5-35B-A3B optimization + +This folder contains an Olive recipe for exporting the text-only component of `Qwen/Qwen3.5-35B-A3B` for the +`NvTensorRTRTXExecutionProvider` (also known as the `NvTensorRtRtx` EP). + +## INT4 weight-only quantization + +The `Qwen3.5-35B-A3B_model_builder_int4.json` recipe uses the ONNX Runtime GenAI `ModelBuilder` to: + +1. Export a standalone text model by including the token embedding layer (`exclude_embeds=false`). +2. Apply symmetric INT4 RTN weight-only quantization with a block size of 32. +3. Export quantized matrix multiplications directly in INT4 QDQ format (`use_qdq=true`). + +The vision encoder is not exported. + +## Setup + +1. Install Olive. +2. Install a Transformers 5.x release that recognizes the `qwen3_5_moe` architecture. +3. Install an ONNX Runtime GenAI package or build that supports Qwen3.5 MoE hybrid models and the + `NvTensorRTRTXExecutionProvider`. + +## Run + +```bash +olive run --config Qwen3.5-35B-A3B_model_builder_int4.json +``` diff --git a/Qwen-Qwen3.5-35B-A3B/NvTensorRtRtx/info.yml b/Qwen-Qwen3.5-35B-A3B/NvTensorRtRtx/info.yml new file mode 100644 index 000000000..278ca8e87 --- /dev/null +++ b/Qwen-Qwen3.5-35B-A3B/NvTensorRtRtx/info.yml @@ -0,0 +1,6 @@ +arch: qwen3_5_moe_text +recipes: + - name: Qwen3.5-35B-A3B_Model_Builder_INT4 + file: Qwen3.5-35B-A3B_model_builder_int4.json + devices: gpu + eps: NvTensorRTRTXExecutionProvider diff --git a/Qwen-Qwen3.5-4B/NvTensorRtRtx/Qwen3.5-4B_model_builder_int4.json b/Qwen-Qwen3.5-4B/NvTensorRtRtx/Qwen3.5-4B_model_builder_int4.json new file mode 100644 index 000000000..111f4a1d1 --- /dev/null +++ b/Qwen-Qwen3.5-4B/NvTensorRtRtx/Qwen3.5-4B_model_builder_int4.json @@ -0,0 +1,36 @@ +{ + "input_model": { + "type": "HfModel", + "model_path": "Qwen/Qwen3.5-4B", + "task": "text-classification" + }, + "systems": { + "local_system": { + "type": "LocalSystem", + "accelerators": [ + { + "device": "gpu", + "execution_providers": ["NvTensorRTRTXExecutionProvider"] + } + ] + } + }, + "engine": { + "target": "local_system" + }, + "passes": { + "mb": { + "type": "ModelBuilder", + "precision": "int4", + "int4_block_size": 32, + "int4_algo_config": "rtn", + "int4_is_symmetric": true, + "use_qdq": true, + "enable_cuda_graph": true, + "extra_options": { + "exclude_embeds": false + } + } + }, + "log_severity_level": 0 +} diff --git a/Qwen-Qwen3.5-4B/NvTensorRtRtx/README.md b/Qwen-Qwen3.5-4B/NvTensorRtRtx/README.md new file mode 100644 index 000000000..9fe01137d --- /dev/null +++ b/Qwen-Qwen3.5-4B/NvTensorRtRtx/README.md @@ -0,0 +1,27 @@ +# Qwen3.5-4B optimization + +This folder contains an Olive recipe for exporting the text-only component of `Qwen/Qwen3.5-4B` for the +`NvTensorRTRTXExecutionProvider` (also known as the `NvTensorRtRtx` EP). + +## INT4 weight-only quantization + +The `Qwen3.5-4B_model_builder_int4.json` recipe uses the ONNX Runtime GenAI `ModelBuilder` to: + +1. Export a standalone text model by including the token embedding layer (`exclude_embeds=false`). +2. Apply symmetric INT4 RTN weight-only quantization with a block size of 32. +3. Export quantized matrix multiplications directly in INT4 QDQ format (`use_qdq=true`). + +The vision encoder is not exported. + +## Setup + +1. Install Olive. +2. Install a Transformers 5.x release that recognizes the `qwen3_5` architecture. +3. Install an ONNX Runtime GenAI package or build that supports Qwen3.5 hybrid models and the + `NvTensorRTRTXExecutionProvider`. + +## Run + +```bash +olive run --config Qwen3.5-4B_model_builder_int4.json +``` diff --git a/Qwen-Qwen3.5-4B/NvTensorRtRtx/info.yml b/Qwen-Qwen3.5-4B/NvTensorRtRtx/info.yml new file mode 100644 index 000000000..38ce2b3cd --- /dev/null +++ b/Qwen-Qwen3.5-4B/NvTensorRtRtx/info.yml @@ -0,0 +1,6 @@ +arch: qwen3_5_text +recipes: + - name: Qwen3.5-4B_Model_Builder_INT4 + file: Qwen3.5-4B_model_builder_int4.json + devices: gpu + eps: NvTensorRTRTXExecutionProvider diff --git a/Qwen-Qwen3.5-9B/NvTensorRtRtx/Qwen3.5-9B_model_builder_int4.json b/Qwen-Qwen3.5-9B/NvTensorRtRtx/Qwen3.5-9B_model_builder_int4.json new file mode 100644 index 000000000..5765122d9 --- /dev/null +++ b/Qwen-Qwen3.5-9B/NvTensorRtRtx/Qwen3.5-9B_model_builder_int4.json @@ -0,0 +1,36 @@ +{ + "input_model": { + "type": "HfModel", + "model_path": "Qwen/Qwen3.5-9B", + "task": "text-classification" + }, + "systems": { + "local_system": { + "type": "LocalSystem", + "accelerators": [ + { + "device": "gpu", + "execution_providers": ["NvTensorRTRTXExecutionProvider"] + } + ] + } + }, + "engine": { + "target": "local_system" + }, + "passes": { + "mb": { + "type": "ModelBuilder", + "precision": "int4", + "int4_block_size": 32, + "int4_algo_config": "rtn", + "int4_is_symmetric": true, + "use_qdq": true, + "enable_cuda_graph": true, + "extra_options": { + "exclude_embeds": false + } + } + }, + "log_severity_level": 0 +} diff --git a/Qwen-Qwen3.5-9B/NvTensorRtRtx/README.md b/Qwen-Qwen3.5-9B/NvTensorRtRtx/README.md new file mode 100644 index 000000000..3f735f331 --- /dev/null +++ b/Qwen-Qwen3.5-9B/NvTensorRtRtx/README.md @@ -0,0 +1,27 @@ +# Qwen3.5-9B optimization + +This folder contains an Olive recipe for exporting the text-only component of `Qwen/Qwen3.5-9B` for the +`NvTensorRTRTXExecutionProvider` (also known as the `NvTensorRtRtx` EP). + +## INT4 weight-only quantization + +The `Qwen3.5-9B_model_builder_int4.json` recipe uses the ONNX Runtime GenAI `ModelBuilder` to: + +1. Export a standalone text model by including the token embedding layer (`exclude_embeds=false`). +2. Apply symmetric INT4 RTN weight-only quantization with a block size of 32. +3. Export quantized matrix multiplications directly in INT4 QDQ format (`use_qdq=true`). + +The vision encoder is not exported. + +## Setup + +1. Install Olive. +2. Install a Transformers 5.x release that recognizes the `qwen3_5` architecture. +3. Install an ONNX Runtime GenAI package or build that supports Qwen3.5 hybrid models and the + `NvTensorRTRTXExecutionProvider`. + +## Run + +```bash +olive run --config Qwen3.5-9B_model_builder_int4.json +``` diff --git a/Qwen-Qwen3.5-9B/NvTensorRtRtx/info.yml b/Qwen-Qwen3.5-9B/NvTensorRtRtx/info.yml new file mode 100644 index 000000000..e4aefe026 --- /dev/null +++ b/Qwen-Qwen3.5-9B/NvTensorRtRtx/info.yml @@ -0,0 +1,6 @@ +arch: qwen3_5_text +recipes: + - name: Qwen3.5-9B_Model_Builder_INT4 + file: Qwen3.5-9B_model_builder_int4.json + devices: gpu + eps: NvTensorRTRTXExecutionProvider diff --git a/Qwen-Qwen3.6-27B/NvTensorRtRtx/Qwen3.6-27B_model_builder_int4.json b/Qwen-Qwen3.6-27B/NvTensorRtRtx/Qwen3.6-27B_model_builder_int4.json new file mode 100644 index 000000000..38fc46175 --- /dev/null +++ b/Qwen-Qwen3.6-27B/NvTensorRtRtx/Qwen3.6-27B_model_builder_int4.json @@ -0,0 +1,36 @@ +{ + "input_model": { + "type": "HfModel", + "model_path": "Qwen/Qwen3.6-27B", + "task": "text-classification" + }, + "systems": { + "local_system": { + "type": "LocalSystem", + "accelerators": [ + { + "device": "gpu", + "execution_providers": ["NvTensorRTRTXExecutionProvider"] + } + ] + } + }, + "engine": { + "target": "local_system" + }, + "passes": { + "mb": { + "type": "ModelBuilder", + "precision": "int4", + "int4_block_size": 32, + "int4_algo_config": "rtn", + "int4_is_symmetric": true, + "use_qdq": true, + "enable_cuda_graph": true, + "extra_options": { + "exclude_embeds": false + } + } + }, + "log_severity_level": 0 +} diff --git a/Qwen-Qwen3.6-27B/NvTensorRtRtx/README.md b/Qwen-Qwen3.6-27B/NvTensorRtRtx/README.md new file mode 100644 index 000000000..ade8f38cf --- /dev/null +++ b/Qwen-Qwen3.6-27B/NvTensorRtRtx/README.md @@ -0,0 +1,28 @@ +# Qwen3.6-27B optimization + +This folder contains an Olive recipe for exporting the text-only component of `Qwen/Qwen3.6-27B` for the +`NvTensorRTRTXExecutionProvider` (also known as the `NvTensorRtRtx` EP). + +## INT4 weight-only quantization + +The `Qwen3.6-27B_model_builder_int4.json` recipe uses the ONNX Runtime GenAI Qwen3.5 hybrid model builder, which +matches the architecture declared by the Qwen3.6 checkpoint, to: + +1. Export a standalone text model by including the token embedding layer (`exclude_embeds=false`). +2. Apply symmetric INT4 RTN weight-only quantization with a block size of 32. +3. Export quantized matrix multiplications directly in INT4 QDQ format (`use_qdq=true`). + +The vision encoder is not exported. + +## Setup + +1. Install Olive. +2. Install a Transformers 5.x release that recognizes the `qwen3_5` architecture. +3. Install an ONNX Runtime GenAI package or build that supports Qwen3.5 hybrid models and the + `NvTensorRTRTXExecutionProvider`. + +## Run + +```bash +olive run --config Qwen3.6-27B_model_builder_int4.json +``` diff --git a/Qwen-Qwen3.6-27B/NvTensorRtRtx/info.yml b/Qwen-Qwen3.6-27B/NvTensorRtRtx/info.yml new file mode 100644 index 000000000..4ccc6289d --- /dev/null +++ b/Qwen-Qwen3.6-27B/NvTensorRtRtx/info.yml @@ -0,0 +1,6 @@ +arch: qwen3_5_text +recipes: + - name: Qwen3.6-27B_Model_Builder_INT4 + file: Qwen3.6-27B_model_builder_int4.json + devices: gpu + eps: NvTensorRTRTXExecutionProvider diff --git a/Qwen-Qwen3.6-35B-A3B/NvTensorRtRtx/Qwen3.6-35B-A3B_model_builder_int4.json b/Qwen-Qwen3.6-35B-A3B/NvTensorRtRtx/Qwen3.6-35B-A3B_model_builder_int4.json new file mode 100644 index 000000000..48fe42eab --- /dev/null +++ b/Qwen-Qwen3.6-35B-A3B/NvTensorRtRtx/Qwen3.6-35B-A3B_model_builder_int4.json @@ -0,0 +1,36 @@ +{ + "input_model": { + "type": "HfModel", + "model_path": "Qwen/Qwen3.6-35B-A3B", + "task": "text-classification" + }, + "systems": { + "local_system": { + "type": "LocalSystem", + "accelerators": [ + { + "device": "gpu", + "execution_providers": ["NvTensorRTRTXExecutionProvider"] + } + ] + } + }, + "engine": { + "target": "local_system" + }, + "passes": { + "mb": { + "type": "ModelBuilder", + "precision": "int4", + "int4_block_size": 32, + "int4_algo_config": "rtn", + "int4_is_symmetric": true, + "use_qdq": true, + "enable_cuda_graph": true, + "extra_options": { + "exclude_embeds": false + } + } + }, + "log_severity_level": 0 +} diff --git a/Qwen-Qwen3.6-35B-A3B/NvTensorRtRtx/README.md b/Qwen-Qwen3.6-35B-A3B/NvTensorRtRtx/README.md new file mode 100644 index 000000000..89408e7ed --- /dev/null +++ b/Qwen-Qwen3.6-35B-A3B/NvTensorRtRtx/README.md @@ -0,0 +1,28 @@ +# Qwen3.6-35B-A3B optimization + +This folder contains an Olive recipe for exporting the text-only component of `Qwen/Qwen3.6-35B-A3B` for the +`NvTensorRTRTXExecutionProvider` (also known as the `NvTensorRtRtx` EP). + +## INT4 weight-only quantization + +The `Qwen3.6-35B-A3B_model_builder_int4.json` recipe uses the ONNX Runtime GenAI Qwen3.5 MoE hybrid model builder, +which matches the architecture declared by the Qwen3.6 checkpoint, to: + +1. Export a standalone text model by including the token embedding layer (`exclude_embeds=false`). +2. Apply symmetric INT4 RTN weight-only quantization with a block size of 32. +3. Export quantized matrix multiplications directly in INT4 QDQ format (`use_qdq=true`). + +The vision encoder is not exported. + +## Setup + +1. Install Olive. +2. Install a Transformers 5.x release that recognizes the `qwen3_5_moe` architecture. +3. Install an ONNX Runtime GenAI package or build that supports Qwen3.5 MoE hybrid models and the + `NvTensorRTRTXExecutionProvider`. + +## Run + +```bash +olive run --config Qwen3.6-35B-A3B_model_builder_int4.json +``` diff --git a/Qwen-Qwen3.6-35B-A3B/NvTensorRtRtx/info.yml b/Qwen-Qwen3.6-35B-A3B/NvTensorRtRtx/info.yml new file mode 100644 index 000000000..9d4745941 --- /dev/null +++ b/Qwen-Qwen3.6-35B-A3B/NvTensorRtRtx/info.yml @@ -0,0 +1,6 @@ +arch: qwen3_5_moe_text +recipes: + - name: Qwen3.6-35B-A3B_Model_Builder_INT4 + file: Qwen3.6-35B-A3B_model_builder_int4.json + devices: gpu + eps: NvTensorRTRTXExecutionProvider From 6839f291aae630cb2bda8a6ceadee2e18c3bce8c Mon Sep 17 00:00:00 2001 From: xManedge Date: Mon, 3 Aug 2026 11:13:38 -0700 Subject: [PATCH 07/10] added qwen3 8B recipe to QNN --- Qwen-Qwen3-8B/QNN/README.md | 65 +++++++++++++++ Qwen-Qwen3-8B/QNN/requirements.txt | 8 ++ Qwen-Qwen3-8B/QNN/x2_elite_config.json | 110 +++++++++++++++++++++++++ Qwen-Qwen3-8B/QNN/x_elite_config.json | 109 ++++++++++++++++++++++++ 4 files changed, 292 insertions(+) create mode 100644 Qwen-Qwen3-8B/QNN/README.md create mode 100644 Qwen-Qwen3-8B/QNN/requirements.txt create mode 100644 Qwen-Qwen3-8B/QNN/x2_elite_config.json create mode 100644 Qwen-Qwen3-8B/QNN/x_elite_config.json diff --git a/Qwen-Qwen3-8B/QNN/README.md b/Qwen-Qwen3-8B/QNN/README.md new file mode 100644 index 000000000..ef3dba7da --- /dev/null +++ b/Qwen-Qwen3-8B/QNN/README.md @@ -0,0 +1,65 @@ +# Qwen3-8B Model Optimization + +This repository demonstrates the optimization of the [Qwen3-8B](https://huggingface.co/Qwen/Qwen3-8B) model using **post-training quantization (PTQ)** techniques. This code was originally run on Ubuntu 22.04, use the same OS for maximum compatibility. + +### Quantization Python Environment Setup +Quantization is resource-intensive and requires GPU acceleration. In an x64 Python environment, install the required packages: + +```bash +pip install -r requirements.txt + +# Disable CUDA extension build (not required) +# Linux +export BUILD_CUDA_EXT=0 +# Windows +# set BUILD_CUDA_EXT=0 + +# Install GptqModel from source +pip install --no-build-isolation git+https://github.com/CodeLinaro/GPTQModel.git@rel_4.2.5 +``` + +### AOT Compilation Python Environment Setup +Model compilation using QNN Execution Provider requires a Python environment with onnxruntime-qnn installed. In a separate Python environment, install the required packages: + +```bash +# Install Olive +pip install olive-ai==0.13.0 + +# Install ONNX Runtime QNN +pip install -r https://raw.githubusercontent.com/microsoft/onnxruntime/refs/heads/main/requirements.txt +pip install onnxruntime-qnn==2.3.0 --no-deps +``` + +Replace `/path/to/qnn/env/bin` in the config file with the path to the directory containing your QNN environment's Python executable. This path can be found by running the following command in the environment: + +```bash +# Linux +command -v python +# Windows +# where python +``` + +This command will return the path to the Python executable. Set the parent directory of the executable as the `/path/to/qnn/env/bin` in the config file. + +### Run the Quantization + Compilation Config +Activate the **Quantization Python Environment** and run the workflow. + +For Snapdragon X Elite: + +```bash +olive run --config x_elite_config.json +``` + +For Snapdragon X2 Elite: + +```bash +olive run --config x2_elite_config.json +``` + +Olive will run the AOT compilation step in the **AOT Compilation Python Environment** specified in the config file using a subprocess. All other steps will run in the **Quantization Python Environment** natively. + +Optimized model saved in: `models/qwen3_8B/` + +> If optimization fails during context binary generation, rerun the command. The process will resume from the last completed step. + +> If the Static Quantization (SQ) pass fails with `Failed to allocate memory buffer of size...`, rerun the command without clearing the cache. Olive will resume from the last completed step and the pass will succeed. \ No newline at end of file diff --git a/Qwen-Qwen3-8B/QNN/requirements.txt b/Qwen-Qwen3-8B/QNN/requirements.txt new file mode 100644 index 000000000..0c847b014 --- /dev/null +++ b/Qwen-Qwen3-8B/QNN/requirements.txt @@ -0,0 +1,8 @@ +datasets +olive-ai==0.13.0 +# these are the versions the recipes were last validated with +onnxruntime-genai-cuda==0.11.2 +onnxruntime-gpu==1.27.0 +optimum +# newer transformers might have incompatibility with gptq passes +transformers==4.57.3 diff --git a/Qwen-Qwen3-8B/QNN/x2_elite_config.json b/Qwen-Qwen3-8B/QNN/x2_elite_config.json new file mode 100644 index 000000000..8ba3c3945 --- /dev/null +++ b/Qwen-Qwen3-8B/QNN/x2_elite_config.json @@ -0,0 +1,110 @@ +{ + "input_model": { "type": "HfModel", "model_path": "Qwen/Qwen3-8B" }, + "systems": { + "qnn_system": { + "type": "PythonEnvironment", + "python_environment_path": "/path/to/qnn/env/bin", + "accelerators": [ { "execution_providers": [ "QNNExecutionProvider" ] } ] + } + }, + "data_configs": [ + { + "name": "wikitext2_train_joined", + "type": "HuggingfaceContainer", + "load_dataset_config": { "data_name": "wikitext", "subset": "wikitext-2-raw-v1", "split": "train" }, + "pre_process_data_config": { + "strategy": "join", + "add_special_tokens": false, + "max_seq_len": 4096, + "max_samples": 128 + } + }, + { + "name": "wikitext2_train_act", + "type": "HuggingfaceContainer", + "load_dataset_config": { "data_name": "wikitext", "subset": "wikitext-2-raw-v1", "split": "train" }, + "pre_process_data_config": { + "strategy": "line-by-line", + "add_special_tokens": true, + "max_samples": 256, + "max_seq_len": 4096 + } + } + ], + "passes": { + "q": { "type": "QuaRot" }, + "cs": { "type": "CaptureSplitInfo", "num_splits": 4, "unique_embeds_lm_head_splits": true }, + "g": { + "type": "GptqModel", + "bits": 4, + "sym": true, + "group_size": -1, + "lm_head": true, + "device": "cuda", + "data_config": "wikitext2_train_joined", + "dynamic": { + "+:.*lm_head*": { "bits": 4, "sym": true, "group_size": 32, "desc_act": false }, + "+:.*layers\\.[0-7]\\.mlp\\.down_proj.*": { "bits": 8, "sym": true, "group_size": -1, "desc_act": true } + } + }, + "mb": { + "type": "ModelBuilder", + "precision": "int4", + "int4_block_size": 32, + "int4_accuracy_level": 4, + "int4_op_types_to_quantize": [ "Gather" ] + }, + "mq": { + "type": "MatMulNBitsToQDQ", + "use_int4": true, + "add_zero_point": true, + "nodes_to_exclude": [ "/lm_head/MatMulNBits" ], + "save_as_external_data": true + }, + "f16": { + "type": "OnnxFloatToFloat16", + "op_include_list": [ "GroupQueryAttention" ], + "keep_io_types": [ "logits" ], + "save_as_external_data": true + }, + "gs": { + "type": "GraphSurgeries", + "surgeries": [ + { "surgeon": "RemoveRopeMultiCache" }, + { "surgeon": "AttentionMaskToSequenceLengths" }, + { "surgeon": "RemoveGidxFromMatMulNBits" }, + { "surgeon": "SimplifiedLayerNormToL2Norm" } + ], + "save_as_external_data": true + }, + "sq": { + "type": "OnnxStaticQuantization", + "data_config": "wikitext2_train_act", + "activation_type": "uint16", + "precision": "uint8", + "calibration_providers": [ "CUDAExecutionProvider" ], + "quant_preprocess": true, + "op_types_to_exclude": [ "GatherBlockQuantized", "GroupQueryAttention", "MatMulNBits" ], + "save_as_external_data": true, + "extra_options": { "CalibStridedMinMax": 1 } + }, + "sp": { "type": "SplitModel" }, + "st": { "type": "StaticLLM", "batch_size": 1, "context_length": 64 }, + "cb": { + "type": "EPContextBinaryGenerator", + "provider_options": { + "htp_performance_mode": "burst", + "extended_udma": "1", + "htp_graph_finalization_optimization_mode": "3", + "soc_model": "87" + }, + "weight_sharing": true + }, + "cp": { "type": "ComposeOnnxModels" } + }, + "target": "qnn_system", + "log_severity_level": 1, + "output_dir": "models/qwen3_8B", + "cache_dir": "cache", + "no_artifacts": true +} diff --git a/Qwen-Qwen3-8B/QNN/x_elite_config.json b/Qwen-Qwen3-8B/QNN/x_elite_config.json new file mode 100644 index 000000000..40f047283 --- /dev/null +++ b/Qwen-Qwen3-8B/QNN/x_elite_config.json @@ -0,0 +1,109 @@ +{ + "input_model": { "type": "HfModel", "model_path": "Qwen/Qwen3-8B" }, + "systems": { + "qnn_system": { + "type": "PythonEnvironment", + "python_environment_path": "/path/to/qnn/env/bin", + "accelerators": [ { "execution_providers": [ "QNNExecutionProvider" ] } ] + } + }, + "data_configs": [ + { + "name": "wikitext2_train_joined", + "type": "HuggingfaceContainer", + "load_dataset_config": { "data_name": "wikitext", "subset": "wikitext-2-raw-v1", "split": "train" }, + "pre_process_data_config": { + "strategy": "join", + "add_special_tokens": false, + "max_seq_len": 4096, + "max_samples": 128 + } + }, + { + "name": "wikitext2_train_act", + "type": "HuggingfaceContainer", + "load_dataset_config": { "data_name": "wikitext", "subset": "wikitext-2-raw-v1", "split": "train" }, + "pre_process_data_config": { + "strategy": "line-by-line", + "add_special_tokens": true, + "max_samples": 256, + "max_seq_len": 4096 + } + } + ], + "passes": { + "q": { "type": "QuaRot" }, + "cs": { "type": "CaptureSplitInfo", "num_splits": 4, "unique_embeds_lm_head_splits": true }, + "g": { + "type": "GptqModel", + "bits": 4, + "sym": true, + "group_size": -1, + "lm_head": true, + "device": "cuda", + "data_config": "wikitext2_train_joined", + "dynamic": { + "+:.*lm_head*": { "bits": 4, "sym": true, "group_size": 32, "desc_act": false }, + "+:.*layers\\.[0-7]\\.mlp\\.down_proj.*": { "bits": 8, "sym": true, "group_size": -1, "desc_act": true } + } + }, + "mb": { + "type": "ModelBuilder", + "precision": "int4", + "int4_block_size": 32, + "int4_accuracy_level": 4, + "int4_op_types_to_quantize": [ "Gather" ] + }, + "mq": { + "type": "MatMulNBitsToQDQ", + "use_int4": true, + "add_zero_point": true, + "nodes_to_exclude": [ "/lm_head/MatMulNBits" ], + "save_as_external_data": true + }, + "f16": { + "type": "OnnxFloatToFloat16", + "op_include_list": [ "GroupQueryAttention" ], + "keep_io_types": [ "logits" ], + "save_as_external_data": true + }, + "gs": { + "type": "GraphSurgeries", + "surgeries": [ + { "surgeon": "RemoveRopeMultiCache" }, + { "surgeon": "AttentionMaskToSequenceLengths" }, + { "surgeon": "RemoveGidxFromMatMulNBits" }, + { "surgeon": "SimplifiedLayerNormToL2Norm" } + ], + "save_as_external_data": true + }, + "sq": { + "type": "OnnxStaticQuantization", + "data_config": "wikitext2_train_act", + "activation_type": "uint16", + "precision": "uint8", + "calibration_providers": [ "CUDAExecutionProvider" ], + "quant_preprocess": true, + "op_types_to_exclude": [ "GatherBlockQuantized", "GroupQueryAttention", "MatMulNBits" ], + "save_as_external_data": true, + "extra_options": { "CalibStridedMinMax": 1 } + }, + "sp": { "type": "SplitModel" }, + "st": { "type": "StaticLLM", "batch_size": 1, "context_length": 64 }, + "cb": { + "type": "EPContextBinaryGenerator", + "provider_options": { + "htp_performance_mode": "burst", + "htp_graph_finalization_optimization_mode": "3", + "soc_model": "60" + }, + "weight_sharing": true + }, + "cp": { "type": "ComposeOnnxModels" } + }, + "target": "qnn_system", + "log_severity_level": 1, + "output_dir": "models/qwen3_8B", + "cache_dir": "cache", + "no_artifacts": true +} From 84c20ad7520c49f13655d5af63d82096bc5661e5 Mon Sep 17 00:00:00 2001 From: xManedge Date: Mon, 3 Aug 2026 11:17:29 -0700 Subject: [PATCH 08/10] added qwen3 8B to qnn recipies --- Qwen-Qwen3-8B/QNN/README.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/Qwen-Qwen3-8B/QNN/README.md b/Qwen-Qwen3-8B/QNN/README.md index ef3dba7da..d5cb16627 100644 --- a/Qwen-Qwen3-8B/QNN/README.md +++ b/Qwen-Qwen3-8B/QNN/README.md @@ -62,4 +62,4 @@ Optimized model saved in: `models/qwen3_8B/` > If optimization fails during context binary generation, rerun the command. The process will resume from the last completed step. -> If the Static Quantization (SQ) pass fails with `Failed to allocate memory buffer of size...`, rerun the command without clearing the cache. Olive will resume from the last completed step and the pass will succeed. \ No newline at end of file +> If the Static Quantization (SQ) pass fails with `Failed to allocate memory buffer of size...`, rerun the command without clearing the cache. Olive will resume from the last completed step and the pass will succeed. From 1daf92189c0416744a20df436dcd84d811c2d17e Mon Sep 17 00:00:00 2001 From: xManedge Date: Mon, 3 Aug 2026 11:26:31 -0700 Subject: [PATCH 09/10] Potential fix for pull request finding Co-authored-by: Copilot Autofix powered by AI <175728472+Copilot@users.noreply.github.com> --- Qwen-Qwen3-8B/QNN/requirements.txt | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/Qwen-Qwen3-8B/QNN/requirements.txt b/Qwen-Qwen3-8B/QNN/requirements.txt index 0c847b014..67f3f5de8 100644 --- a/Qwen-Qwen3-8B/QNN/requirements.txt +++ b/Qwen-Qwen3-8B/QNN/requirements.txt @@ -1,6 +1,6 @@ datasets olive-ai==0.13.0 -# these are the versions the recipes were last validated with +# pinned packages below are the versions the recipes were last validated with (datasets/optimum are intentionally left unpinned) onnxruntime-genai-cuda==0.11.2 onnxruntime-gpu==1.27.0 optimum From 7271d6cfdb5026e320987897c1bbc55bdde446d6 Mon Sep 17 00:00:00 2001 From: xManedge Date: Mon, 3 Aug 2026 11:33:22 -0700 Subject: [PATCH 10/10] added qwen3 8B to QNN --- Qwen-Qwen3-8B/QNN/requirements.txt | 1 - 1 file changed, 1 deletion(-) diff --git a/Qwen-Qwen3-8B/QNN/requirements.txt b/Qwen-Qwen3-8B/QNN/requirements.txt index 67f3f5de8..8d9863816 100644 --- a/Qwen-Qwen3-8B/QNN/requirements.txt +++ b/Qwen-Qwen3-8B/QNN/requirements.txt @@ -1,6 +1,5 @@ datasets olive-ai==0.13.0 -# pinned packages below are the versions the recipes were last validated with (datasets/optimum are intentionally left unpinned) onnxruntime-genai-cuda==0.11.2 onnxruntime-gpu==1.27.0 optimum