From 13271fa897030ef1d7f7a44f82b9c12736f981d6 Mon Sep 17 00:00:00 2001 From: Pooja Ganesh Date: Mon, 20 Jul 2026 18:29:57 -0700 Subject: [PATCH] Add RyzenAI 8D + MLPerf recipes (Qwen3-8B, Phi-4, Llama-3.1-8B) --- .../Qwen1.5-7B-Chat_quark_ryzenai_llm.json | 5 +- .../RyzenAI/requirements_ryzenai_llm.txt | 10 +- .../RyzenAI/Qwen2-1.5B_quark_ryzenai_llm.json | 5 +- .../RyzenAI/requirements_ryzenai_llm.txt | 10 +- .../RyzenAI/Qwen2-7B_quark_ryzenai_llm.json | 5 +- .../RyzenAI/requirements_ryzenai_llm.txt | 10 +- ...en2.5-0.5B-Instruct_quark_ryzenai_llm.json | 5 +- .../RyzenAI/requirements_ryzenai_llm.txt | 10 +- ...en2.5-1.5B-Instruct_quark_ryzenai_llm.json | 5 +- .../RyzenAI/requirements_ryzenai_llm.txt | 10 +- ...Qwen2.5-3B-Instruct_quark_ryzenai_llm.json | 5 +- .../RyzenAI/requirements_ryzenai_llm.txt | 10 +- .../RyzenAI/Qwen2.5-3B_quark_ryzenai_llm.json | 5 +- .../RyzenAI/requirements_ryzenai_llm.txt | 10 +- ...Qwen2.5-7B-Instruct_quark_ryzenai_llm.json | 5 +- .../RyzenAI/requirements_ryzenai_llm.txt | 10 +- ...Coder-0.5B-Instruct_quark_ryzenai_llm.json | 5 +- .../RyzenAI/requirements_ryzenai_llm.txt | 10 +- ...Coder-1.5B-Instruct_quark_ryzenai_llm.json | 5 +- .../RyzenAI/requirements_ryzenai_llm.txt | 10 +- ...5-Coder-7B-Instruct_quark_ryzenai_llm.json | 5 +- .../RyzenAI/requirements_ryzenai_llm.txt | 10 +- .../RyzenAI/Qwen3-8B_quark_ryzenai_llm.json | 33 +++++++ ...wen3-8B_quark_ryzenai_llm_full_fusion.json | 47 +++++++++ .../Qwen3-8B_quark_ryzenai_llm_hybrid.json | 47 +++++++++ Qwen-Qwen3-8B/RyzenAI/README.md | 96 +++++++++++++++++++ Qwen-Qwen3-8B/RyzenAI/info.yaml | 16 ++++ .../RyzenAI/requirements_ryzenai_llm.txt | 27 ++++++ ...lama-7b-Instruct-hf_quark_ryzenai_llm.json | 5 +- .../RyzenAI/requirements_ryzenai_llm.txt | 10 +- ...R1-Distill-Llama-8B_quark_ryzenai_llm.json | 5 +- .../RyzenAI/requirements_ryzenai_llm.txt | 10 +- ...1-Distill-Qwen-1.5B_quark_ryzenai_llm.json | 5 +- .../RyzenAI/requirements_ryzenai_llm.txt | 10 +- ...-R1-Distill-Qwen-7B_quark_ryzenai_llm.json | 5 +- .../RyzenAI/requirements_ryzenai_llm.txt | 10 +- gpt-oss-20b/RyzenAI/README.md | 1 - .../RyzenAI/requirements_ryzenai_llm.txt | 10 +- .../RyzenAI/requirements_ryzenai_llm.txt | 10 +- .../Llama-2-7b-hf_quark_ryzenai_llm.json | 5 +- .../RyzenAI/requirements_ryzenai_llm.txt | 10 +- ...ama-3.1-8B-Instruct_quark_ryzenai_llm.json | 5 +- ...nstruct_quark_ryzenai_llm_full_fusion.json | 35 +++++++ ...-8B-Instruct_quark_ryzenai_llm_hybrid.json | 35 +++++++ .../RyzenAI/README.md | 30 +++--- .../RyzenAI/info.yaml | 10 ++ .../RyzenAI/requirements_ryzenai_llm.txt | 10 +- .../Meta-Llama-3.1-8B_quark_ryzenai_llm.json | 5 +- .../RyzenAI/requirements_ryzenai_llm.txt | 10 +- ...ama-3.2-1B-Instruct_quark_ryzenai_llm.json | 5 +- .../RyzenAI/requirements_ryzenai_llm.txt | 10 +- .../Llama-3.2-1B_quark_ryzenai_llm.json | 5 +- .../RyzenAI/requirements_ryzenai_llm.txt | 10 +- ...ama-3.2-3B-Instruct_quark_ryzenai_llm.json | 5 +- .../RyzenAI/requirements_ryzenai_llm.txt | 10 +- .../Llama-3.2-3B_quark_ryzenai_llm.json | 5 +- .../RyzenAI/requirements_ryzenai_llm.txt | 10 +- .../Meta-Llama-3-8B_quark_ryzenai_llm.json | 5 +- .../RyzenAI/requirements_ryzenai_llm.txt | 10 +- ...-mini-128k-instruct_quark_ryzenai_llm.json | 5 +- .../RyzenAI/requirements_ryzenai_llm.txt | 10 +- ...-3-mini-4k-instruct_quark_ryzenai_llm.json | 5 +- .../RyzenAI/requirements_ryzenai_llm.txt | 10 +- ...i-3.5-mini-instruct_quark_ryzenai_llm.json | 5 +- .../RyzenAI/requirements_ryzenai_llm.txt | 10 +- ...Phi-4-mini-instruct_quark_ryzenai_llm.json | 5 +- ...nstruct_quark_ryzenai_llm_full_fusion.json | 36 +++++++ ...ini-instruct_quark_ryzenai_llm_hybrid.json | 36 +++++++ .../RyzenAI/README.md | 21 ++-- .../RyzenAI/info.yaml | 10 ++ .../RyzenAI/requirements_ryzenai_llm.txt | 10 +- ...hi-4-mini-reasoning_quark_ryzenai_llm.json | 5 +- .../RyzenAI/requirements_ryzenai_llm.txt | 10 +- ...al-7B-Instruct-v0.1_quark_ryzenai_llm.json | 5 +- .../RyzenAI/requirements_ryzenai_llm.txt | 10 +- ...al-7B-Instruct-v0.2_quark_ryzenai_llm.json | 5 +- .../RyzenAI/requirements_ryzenai_llm.txt | 10 +- ...al-7B-Instruct-v0.3_quark_ryzenai_llm.json | 5 +- .../RyzenAI/requirements_ryzenai_llm.txt | 10 +- .../Mistral-7B-v0.3_quark_ryzenai_llm.json | 5 +- .../RyzenAI/requirements_ryzenai_llm.txt | 10 +- 81 files changed, 756 insertions(+), 224 deletions(-) create mode 100644 Qwen-Qwen3-8B/RyzenAI/Qwen3-8B_quark_ryzenai_llm.json create mode 100644 Qwen-Qwen3-8B/RyzenAI/Qwen3-8B_quark_ryzenai_llm_full_fusion.json create mode 100644 Qwen-Qwen3-8B/RyzenAI/Qwen3-8B_quark_ryzenai_llm_hybrid.json create mode 100644 Qwen-Qwen3-8B/RyzenAI/README.md create mode 100644 Qwen-Qwen3-8B/RyzenAI/info.yaml create mode 100644 Qwen-Qwen3-8B/RyzenAI/requirements_ryzenai_llm.txt create mode 100644 meta-llama-Llama-3.1-8B-Instruct/RyzenAI/Llama-3.1-8B-Instruct_quark_ryzenai_llm_full_fusion.json create mode 100644 meta-llama-Llama-3.1-8B-Instruct/RyzenAI/Llama-3.1-8B-Instruct_quark_ryzenai_llm_hybrid.json create mode 100644 microsoft-Phi-4-mini-instruct/RyzenAI/Phi-4-mini-instruct_quark_ryzenai_llm_full_fusion.json create mode 100644 microsoft-Phi-4-mini-instruct/RyzenAI/Phi-4-mini-instruct_quark_ryzenai_llm_hybrid.json diff --git a/Qwen-Qwen1.5-7B-Chat/RyzenAI/Qwen1.5-7B-Chat_quark_ryzenai_llm.json b/Qwen-Qwen1.5-7B-Chat/RyzenAI/Qwen1.5-7B-Chat_quark_ryzenai_llm.json index 0a527fc0f..401371749 100644 --- a/Qwen-Qwen1.5-7B-Chat/RyzenAI/Qwen1.5-7B-Chat_quark_ryzenai_llm.json +++ b/Qwen-Qwen1.5-7B-Chat/RyzenAI/Qwen1.5-7B-Chat_quark_ryzenai_llm.json @@ -13,7 +13,10 @@ }, "mg": { "type": "RyzenGenerateModelLLM", - "recipe": "token_fusion" + "recipe": "token_fusion", + "extra_options": { + "max_seq_len": "16384" + } } }, "log_severity_level": 1, diff --git a/Qwen-Qwen1.5-7B-Chat/RyzenAI/requirements_ryzenai_llm.txt b/Qwen-Qwen1.5-7B-Chat/RyzenAI/requirements_ryzenai_llm.txt index 53ba97fc9..22bd7b945 100644 --- a/Qwen-Qwen1.5-7B-Chat/RyzenAI/requirements_ryzenai_llm.txt +++ b/Qwen-Qwen1.5-7B-Chat/RyzenAI/requirements_ryzenai_llm.txt @@ -1,22 +1,22 @@ ---extra-index-url=https://pypi.amd.com/olive/1.7.1-6D/simple/ +--extra-index-url=https://pypi.amd.com/olive/1.8.0-8D/simple/ # AMD model generation --extra-index-url=https://pypi.amd.com/simple accelerate # Quark -amd-quark==0.11 +amd-quark==0.12.post1 datasets evaluate model-generate nltk -numpy==1.26.4 +numpy==2.5.1 # Pin onnx version -onnx==1.18.0 +onnx==1.22.0 onnxruntime -onnxruntime-genai +onnxruntime-genai==0.14.0 onnxsim optimum diff --git a/Qwen-Qwen2-1.5B/RyzenAI/Qwen2-1.5B_quark_ryzenai_llm.json b/Qwen-Qwen2-1.5B/RyzenAI/Qwen2-1.5B_quark_ryzenai_llm.json index a59066870..764ee1dc1 100644 --- a/Qwen-Qwen2-1.5B/RyzenAI/Qwen2-1.5B_quark_ryzenai_llm.json +++ b/Qwen-Qwen2-1.5B/RyzenAI/Qwen2-1.5B_quark_ryzenai_llm.json @@ -13,7 +13,10 @@ }, "mg": { "type": "RyzenGenerateModelLLM", - "recipe": "token_fusion" + "recipe": "token_fusion", + "extra_options": { + "max_seq_len": "16384" + } } }, "log_severity_level": 1, diff --git a/Qwen-Qwen2-1.5B/RyzenAI/requirements_ryzenai_llm.txt b/Qwen-Qwen2-1.5B/RyzenAI/requirements_ryzenai_llm.txt index 53ba97fc9..22bd7b945 100644 --- a/Qwen-Qwen2-1.5B/RyzenAI/requirements_ryzenai_llm.txt +++ b/Qwen-Qwen2-1.5B/RyzenAI/requirements_ryzenai_llm.txt @@ -1,22 +1,22 @@ ---extra-index-url=https://pypi.amd.com/olive/1.7.1-6D/simple/ +--extra-index-url=https://pypi.amd.com/olive/1.8.0-8D/simple/ # AMD model generation --extra-index-url=https://pypi.amd.com/simple accelerate # Quark -amd-quark==0.11 +amd-quark==0.12.post1 datasets evaluate model-generate nltk -numpy==1.26.4 +numpy==2.5.1 # Pin onnx version -onnx==1.18.0 +onnx==1.22.0 onnxruntime -onnxruntime-genai +onnxruntime-genai==0.14.0 onnxsim optimum diff --git a/Qwen-Qwen2-7B/RyzenAI/Qwen2-7B_quark_ryzenai_llm.json b/Qwen-Qwen2-7B/RyzenAI/Qwen2-7B_quark_ryzenai_llm.json index 3a91a1839..b7367b243 100644 --- a/Qwen-Qwen2-7B/RyzenAI/Qwen2-7B_quark_ryzenai_llm.json +++ b/Qwen-Qwen2-7B/RyzenAI/Qwen2-7B_quark_ryzenai_llm.json @@ -13,7 +13,10 @@ }, "mg": { "type": "RyzenGenerateModelLLM", - "recipe": "token_fusion" + "recipe": "token_fusion", + "extra_options": { + "max_seq_len": "16384" + } } }, "log_severity_level": 1, diff --git a/Qwen-Qwen2-7B/RyzenAI/requirements_ryzenai_llm.txt b/Qwen-Qwen2-7B/RyzenAI/requirements_ryzenai_llm.txt index 53ba97fc9..22bd7b945 100644 --- a/Qwen-Qwen2-7B/RyzenAI/requirements_ryzenai_llm.txt +++ b/Qwen-Qwen2-7B/RyzenAI/requirements_ryzenai_llm.txt @@ -1,22 +1,22 @@ ---extra-index-url=https://pypi.amd.com/olive/1.7.1-6D/simple/ +--extra-index-url=https://pypi.amd.com/olive/1.8.0-8D/simple/ # AMD model generation --extra-index-url=https://pypi.amd.com/simple accelerate # Quark -amd-quark==0.11 +amd-quark==0.12.post1 datasets evaluate model-generate nltk -numpy==1.26.4 +numpy==2.5.1 # Pin onnx version -onnx==1.18.0 +onnx==1.22.0 onnxruntime -onnxruntime-genai +onnxruntime-genai==0.14.0 onnxsim optimum diff --git a/Qwen-Qwen2.5-0.5B-Instruct/RyzenAI/Qwen2.5-0.5B-Instruct_quark_ryzenai_llm.json b/Qwen-Qwen2.5-0.5B-Instruct/RyzenAI/Qwen2.5-0.5B-Instruct_quark_ryzenai_llm.json index 3e0214537..a08c9392f 100644 --- a/Qwen-Qwen2.5-0.5B-Instruct/RyzenAI/Qwen2.5-0.5B-Instruct_quark_ryzenai_llm.json +++ b/Qwen-Qwen2.5-0.5B-Instruct/RyzenAI/Qwen2.5-0.5B-Instruct_quark_ryzenai_llm.json @@ -13,7 +13,10 @@ }, "mg": { "type": "RyzenGenerateModelLLM", - "recipe": "token_fusion" + "recipe": "token_fusion", + "extra_options": { + "max_seq_len": "16384" + } } }, "log_severity_level": 1, diff --git a/Qwen-Qwen2.5-0.5B-Instruct/RyzenAI/requirements_ryzenai_llm.txt b/Qwen-Qwen2.5-0.5B-Instruct/RyzenAI/requirements_ryzenai_llm.txt index 53ba97fc9..22bd7b945 100644 --- a/Qwen-Qwen2.5-0.5B-Instruct/RyzenAI/requirements_ryzenai_llm.txt +++ b/Qwen-Qwen2.5-0.5B-Instruct/RyzenAI/requirements_ryzenai_llm.txt @@ -1,22 +1,22 @@ ---extra-index-url=https://pypi.amd.com/olive/1.7.1-6D/simple/ +--extra-index-url=https://pypi.amd.com/olive/1.8.0-8D/simple/ # AMD model generation --extra-index-url=https://pypi.amd.com/simple accelerate # Quark -amd-quark==0.11 +amd-quark==0.12.post1 datasets evaluate model-generate nltk -numpy==1.26.4 +numpy==2.5.1 # Pin onnx version -onnx==1.18.0 +onnx==1.22.0 onnxruntime -onnxruntime-genai +onnxruntime-genai==0.14.0 onnxsim optimum diff --git a/Qwen-Qwen2.5-1.5B-Instruct/RyzenAI/Qwen2.5-1.5B-Instruct_quark_ryzenai_llm.json b/Qwen-Qwen2.5-1.5B-Instruct/RyzenAI/Qwen2.5-1.5B-Instruct_quark_ryzenai_llm.json index 74320e6c2..f130e8dd2 100644 --- a/Qwen-Qwen2.5-1.5B-Instruct/RyzenAI/Qwen2.5-1.5B-Instruct_quark_ryzenai_llm.json +++ b/Qwen-Qwen2.5-1.5B-Instruct/RyzenAI/Qwen2.5-1.5B-Instruct_quark_ryzenai_llm.json @@ -13,7 +13,10 @@ }, "mg": { "type": "RyzenGenerateModelLLM", - "recipe": "token_fusion" + "recipe": "token_fusion", + "extra_options": { + "max_seq_len": "16384" + } } }, "log_severity_level": 1, diff --git a/Qwen-Qwen2.5-1.5B-Instruct/RyzenAI/requirements_ryzenai_llm.txt b/Qwen-Qwen2.5-1.5B-Instruct/RyzenAI/requirements_ryzenai_llm.txt index 53ba97fc9..22bd7b945 100644 --- a/Qwen-Qwen2.5-1.5B-Instruct/RyzenAI/requirements_ryzenai_llm.txt +++ b/Qwen-Qwen2.5-1.5B-Instruct/RyzenAI/requirements_ryzenai_llm.txt @@ -1,22 +1,22 @@ ---extra-index-url=https://pypi.amd.com/olive/1.7.1-6D/simple/ +--extra-index-url=https://pypi.amd.com/olive/1.8.0-8D/simple/ # AMD model generation --extra-index-url=https://pypi.amd.com/simple accelerate # Quark -amd-quark==0.11 +amd-quark==0.12.post1 datasets evaluate model-generate nltk -numpy==1.26.4 +numpy==2.5.1 # Pin onnx version -onnx==1.18.0 +onnx==1.22.0 onnxruntime -onnxruntime-genai +onnxruntime-genai==0.14.0 onnxsim optimum diff --git a/Qwen-Qwen2.5-3B-Instruct/RyzenAI/Qwen2.5-3B-Instruct_quark_ryzenai_llm.json b/Qwen-Qwen2.5-3B-Instruct/RyzenAI/Qwen2.5-3B-Instruct_quark_ryzenai_llm.json index 13a2f7335..55de34f10 100644 --- a/Qwen-Qwen2.5-3B-Instruct/RyzenAI/Qwen2.5-3B-Instruct_quark_ryzenai_llm.json +++ b/Qwen-Qwen2.5-3B-Instruct/RyzenAI/Qwen2.5-3B-Instruct_quark_ryzenai_llm.json @@ -13,7 +13,10 @@ }, "mg": { "type": "RyzenGenerateModelLLM", - "recipe": "token_fusion" + "recipe": "token_fusion", + "extra_options": { + "max_seq_len": "16384" + } } }, "log_severity_level": 1, diff --git a/Qwen-Qwen2.5-3B-Instruct/RyzenAI/requirements_ryzenai_llm.txt b/Qwen-Qwen2.5-3B-Instruct/RyzenAI/requirements_ryzenai_llm.txt index 53ba97fc9..22bd7b945 100644 --- a/Qwen-Qwen2.5-3B-Instruct/RyzenAI/requirements_ryzenai_llm.txt +++ b/Qwen-Qwen2.5-3B-Instruct/RyzenAI/requirements_ryzenai_llm.txt @@ -1,22 +1,22 @@ ---extra-index-url=https://pypi.amd.com/olive/1.7.1-6D/simple/ +--extra-index-url=https://pypi.amd.com/olive/1.8.0-8D/simple/ # AMD model generation --extra-index-url=https://pypi.amd.com/simple accelerate # Quark -amd-quark==0.11 +amd-quark==0.12.post1 datasets evaluate model-generate nltk -numpy==1.26.4 +numpy==2.5.1 # Pin onnx version -onnx==1.18.0 +onnx==1.22.0 onnxruntime -onnxruntime-genai +onnxruntime-genai==0.14.0 onnxsim optimum diff --git a/Qwen-Qwen2.5-3B/RyzenAI/Qwen2.5-3B_quark_ryzenai_llm.json b/Qwen-Qwen2.5-3B/RyzenAI/Qwen2.5-3B_quark_ryzenai_llm.json index 7bafb5d9c..c8c5b8dd0 100644 --- a/Qwen-Qwen2.5-3B/RyzenAI/Qwen2.5-3B_quark_ryzenai_llm.json +++ b/Qwen-Qwen2.5-3B/RyzenAI/Qwen2.5-3B_quark_ryzenai_llm.json @@ -13,7 +13,10 @@ }, "mg": { "type": "RyzenGenerateModelLLM", - "recipe": "token_fusion" + "recipe": "token_fusion", + "extra_options": { + "max_seq_len": "16384" + } } }, "log_severity_level": 1, diff --git a/Qwen-Qwen2.5-3B/RyzenAI/requirements_ryzenai_llm.txt b/Qwen-Qwen2.5-3B/RyzenAI/requirements_ryzenai_llm.txt index 53ba97fc9..22bd7b945 100644 --- a/Qwen-Qwen2.5-3B/RyzenAI/requirements_ryzenai_llm.txt +++ b/Qwen-Qwen2.5-3B/RyzenAI/requirements_ryzenai_llm.txt @@ -1,22 +1,22 @@ ---extra-index-url=https://pypi.amd.com/olive/1.7.1-6D/simple/ +--extra-index-url=https://pypi.amd.com/olive/1.8.0-8D/simple/ # AMD model generation --extra-index-url=https://pypi.amd.com/simple accelerate # Quark -amd-quark==0.11 +amd-quark==0.12.post1 datasets evaluate model-generate nltk -numpy==1.26.4 +numpy==2.5.1 # Pin onnx version -onnx==1.18.0 +onnx==1.22.0 onnxruntime -onnxruntime-genai +onnxruntime-genai==0.14.0 onnxsim optimum diff --git a/Qwen-Qwen2.5-7B-Instruct/RyzenAI/Qwen2.5-7B-Instruct_quark_ryzenai_llm.json b/Qwen-Qwen2.5-7B-Instruct/RyzenAI/Qwen2.5-7B-Instruct_quark_ryzenai_llm.json index bbc8cfb1e..b0feb3312 100644 --- a/Qwen-Qwen2.5-7B-Instruct/RyzenAI/Qwen2.5-7B-Instruct_quark_ryzenai_llm.json +++ b/Qwen-Qwen2.5-7B-Instruct/RyzenAI/Qwen2.5-7B-Instruct_quark_ryzenai_llm.json @@ -13,7 +13,10 @@ }, "mg": { "type": "RyzenGenerateModelLLM", - "recipe": "token_fusion" + "recipe": "token_fusion", + "extra_options": { + "max_seq_len": "16384" + } } }, "log_severity_level": 1, diff --git a/Qwen-Qwen2.5-7B-Instruct/RyzenAI/requirements_ryzenai_llm.txt b/Qwen-Qwen2.5-7B-Instruct/RyzenAI/requirements_ryzenai_llm.txt index 53ba97fc9..22bd7b945 100644 --- a/Qwen-Qwen2.5-7B-Instruct/RyzenAI/requirements_ryzenai_llm.txt +++ b/Qwen-Qwen2.5-7B-Instruct/RyzenAI/requirements_ryzenai_llm.txt @@ -1,22 +1,22 @@ ---extra-index-url=https://pypi.amd.com/olive/1.7.1-6D/simple/ +--extra-index-url=https://pypi.amd.com/olive/1.8.0-8D/simple/ # AMD model generation --extra-index-url=https://pypi.amd.com/simple accelerate # Quark -amd-quark==0.11 +amd-quark==0.12.post1 datasets evaluate model-generate nltk -numpy==1.26.4 +numpy==2.5.1 # Pin onnx version -onnx==1.18.0 +onnx==1.22.0 onnxruntime -onnxruntime-genai +onnxruntime-genai==0.14.0 onnxsim optimum diff --git a/Qwen-Qwen2.5-Coder-0.5B-Instruct/RyzenAI/Qwen2.5-Coder-0.5B-Instruct_quark_ryzenai_llm.json b/Qwen-Qwen2.5-Coder-0.5B-Instruct/RyzenAI/Qwen2.5-Coder-0.5B-Instruct_quark_ryzenai_llm.json index 38d738400..dd3e17637 100644 --- a/Qwen-Qwen2.5-Coder-0.5B-Instruct/RyzenAI/Qwen2.5-Coder-0.5B-Instruct_quark_ryzenai_llm.json +++ b/Qwen-Qwen2.5-Coder-0.5B-Instruct/RyzenAI/Qwen2.5-Coder-0.5B-Instruct_quark_ryzenai_llm.json @@ -13,7 +13,10 @@ }, "mg": { "type": "RyzenGenerateModelLLM", - "recipe": "token_fusion" + "recipe": "token_fusion", + "extra_options": { + "max_seq_len": "16384" + } } }, "log_severity_level": 1, diff --git a/Qwen-Qwen2.5-Coder-0.5B-Instruct/RyzenAI/requirements_ryzenai_llm.txt b/Qwen-Qwen2.5-Coder-0.5B-Instruct/RyzenAI/requirements_ryzenai_llm.txt index 53ba97fc9..22bd7b945 100644 --- a/Qwen-Qwen2.5-Coder-0.5B-Instruct/RyzenAI/requirements_ryzenai_llm.txt +++ b/Qwen-Qwen2.5-Coder-0.5B-Instruct/RyzenAI/requirements_ryzenai_llm.txt @@ -1,22 +1,22 @@ ---extra-index-url=https://pypi.amd.com/olive/1.7.1-6D/simple/ +--extra-index-url=https://pypi.amd.com/olive/1.8.0-8D/simple/ # AMD model generation --extra-index-url=https://pypi.amd.com/simple accelerate # Quark -amd-quark==0.11 +amd-quark==0.12.post1 datasets evaluate model-generate nltk -numpy==1.26.4 +numpy==2.5.1 # Pin onnx version -onnx==1.18.0 +onnx==1.22.0 onnxruntime -onnxruntime-genai +onnxruntime-genai==0.14.0 onnxsim optimum diff --git a/Qwen-Qwen2.5-Coder-1.5B-Instruct/RyzenAI/Qwen2.5-Coder-1.5B-Instruct_quark_ryzenai_llm.json b/Qwen-Qwen2.5-Coder-1.5B-Instruct/RyzenAI/Qwen2.5-Coder-1.5B-Instruct_quark_ryzenai_llm.json index 77072b860..4b874ad76 100644 --- a/Qwen-Qwen2.5-Coder-1.5B-Instruct/RyzenAI/Qwen2.5-Coder-1.5B-Instruct_quark_ryzenai_llm.json +++ b/Qwen-Qwen2.5-Coder-1.5B-Instruct/RyzenAI/Qwen2.5-Coder-1.5B-Instruct_quark_ryzenai_llm.json @@ -13,7 +13,10 @@ }, "mg": { "type": "RyzenGenerateModelLLM", - "recipe": "token_fusion" + "recipe": "token_fusion", + "extra_options": { + "max_seq_len": "16384" + } } }, "log_severity_level": 1, diff --git a/Qwen-Qwen2.5-Coder-1.5B-Instruct/RyzenAI/requirements_ryzenai_llm.txt b/Qwen-Qwen2.5-Coder-1.5B-Instruct/RyzenAI/requirements_ryzenai_llm.txt index 53ba97fc9..22bd7b945 100644 --- a/Qwen-Qwen2.5-Coder-1.5B-Instruct/RyzenAI/requirements_ryzenai_llm.txt +++ b/Qwen-Qwen2.5-Coder-1.5B-Instruct/RyzenAI/requirements_ryzenai_llm.txt @@ -1,22 +1,22 @@ ---extra-index-url=https://pypi.amd.com/olive/1.7.1-6D/simple/ +--extra-index-url=https://pypi.amd.com/olive/1.8.0-8D/simple/ # AMD model generation --extra-index-url=https://pypi.amd.com/simple accelerate # Quark -amd-quark==0.11 +amd-quark==0.12.post1 datasets evaluate model-generate nltk -numpy==1.26.4 +numpy==2.5.1 # Pin onnx version -onnx==1.18.0 +onnx==1.22.0 onnxruntime -onnxruntime-genai +onnxruntime-genai==0.14.0 onnxsim optimum diff --git a/Qwen-Qwen2.5-Coder-7B-Instruct/RyzenAI/Qwen2.5-Coder-7B-Instruct_quark_ryzenai_llm.json b/Qwen-Qwen2.5-Coder-7B-Instruct/RyzenAI/Qwen2.5-Coder-7B-Instruct_quark_ryzenai_llm.json index a80ae0884..685c00312 100644 --- a/Qwen-Qwen2.5-Coder-7B-Instruct/RyzenAI/Qwen2.5-Coder-7B-Instruct_quark_ryzenai_llm.json +++ b/Qwen-Qwen2.5-Coder-7B-Instruct/RyzenAI/Qwen2.5-Coder-7B-Instruct_quark_ryzenai_llm.json @@ -13,7 +13,10 @@ }, "mg": { "type": "RyzenGenerateModelLLM", - "recipe": "token_fusion" + "recipe": "token_fusion", + "extra_options": { + "max_seq_len": "16384" + } } }, "log_severity_level": 1, diff --git a/Qwen-Qwen2.5-Coder-7B-Instruct/RyzenAI/requirements_ryzenai_llm.txt b/Qwen-Qwen2.5-Coder-7B-Instruct/RyzenAI/requirements_ryzenai_llm.txt index 53ba97fc9..22bd7b945 100644 --- a/Qwen-Qwen2.5-Coder-7B-Instruct/RyzenAI/requirements_ryzenai_llm.txt +++ b/Qwen-Qwen2.5-Coder-7B-Instruct/RyzenAI/requirements_ryzenai_llm.txt @@ -1,22 +1,22 @@ ---extra-index-url=https://pypi.amd.com/olive/1.7.1-6D/simple/ +--extra-index-url=https://pypi.amd.com/olive/1.8.0-8D/simple/ # AMD model generation --extra-index-url=https://pypi.amd.com/simple accelerate # Quark -amd-quark==0.11 +amd-quark==0.12.post1 datasets evaluate model-generate nltk -numpy==1.26.4 +numpy==2.5.1 # Pin onnx version -onnx==1.18.0 +onnx==1.22.0 onnxruntime -onnxruntime-genai +onnxruntime-genai==0.14.0 onnxsim optimum diff --git a/Qwen-Qwen3-8B/RyzenAI/Qwen3-8B_quark_ryzenai_llm.json b/Qwen-Qwen3-8B/RyzenAI/Qwen3-8B_quark_ryzenai_llm.json new file mode 100644 index 000000000..482c83f38 --- /dev/null +++ b/Qwen-Qwen3-8B/RyzenAI/Qwen3-8B_quark_ryzenai_llm.json @@ -0,0 +1,33 @@ +{ + "input_model": { "type": "HFModel", "model_path": "Qwen/Qwen3-8B" }, + "passes": { + "qq": { + "type": "QuarkQuantization", + "quant_scheme": "uint4_wo_128", + "quant_algo": "awq", + "dataset": "pileval_for_awq_benchmark", + "data_type": "bfloat16", + "num_calib_data": 128, + "model_export": ["hf_format"], + "exclude_layers": [], + "layer_quant_scheme": [["lm_head", "uint4_wo_32"]], + "algo_configs": { + "awq": { + "scaling_layers": [], + "model_decoder_layers": "model.layers" + } + } + }, + "mg": { + "type": "RyzenGenerateModelLLM", + "recipe": "token_fusion", + "extra_options": { + "max_seq_len": "16384" + } + } + }, + "log_severity_level": 1, + "output_dir": "models/Qwen3-8B-rai", + "cache_dir": "olive_cache", + "no_artifacts": true +} diff --git a/Qwen-Qwen3-8B/RyzenAI/Qwen3-8B_quark_ryzenai_llm_full_fusion.json b/Qwen-Qwen3-8B/RyzenAI/Qwen3-8B_quark_ryzenai_llm_full_fusion.json new file mode 100644 index 000000000..569e9225f --- /dev/null +++ b/Qwen-Qwen3-8B/RyzenAI/Qwen3-8B_quark_ryzenai_llm_full_fusion.json @@ -0,0 +1,47 @@ +{ + "input_model": { + "type": "HFModel", + "model_path": "Qwen/Qwen3-8B" + }, + "passes": { + "qq": { + "type": "QuarkQuantization", + "quant_scheme": "uint4_wo_128", + "quant_algo": "awq", + "dataset": "pileval_for_awq_benchmark", + "data_type": "bfloat16", + "num_calib_data": 128, + "model_export": [ + "hf_format" + ], + "exclude_layers": [], + "layer_quant_scheme": [ + [ + "lm_head", + "uint4_wo_32" + ] + ], + "algo_configs": { + "awq": { + "scaling_layers": [], + "model_decoder_layers": "model.layers" + } + } + }, + "mg": { + "type": "RyzenGenerateModelLLM", + "mode": "npu", + "recipe": "full_fusion", + "extra_options": { + "model_type": "qwen3-8b", + "max_seq_len": "16384", + "flash_mha": "true", + "attributes": "enable_flat_kv=true enable_flatmlp=true" + } + } + }, + "log_severity_level": 1, + "output_dir": "models/Qwen3-8B-full_fusion-rai", + "cache_dir": "olive_cache", + "no_artifacts": true +} diff --git a/Qwen-Qwen3-8B/RyzenAI/Qwen3-8B_quark_ryzenai_llm_hybrid.json b/Qwen-Qwen3-8B/RyzenAI/Qwen3-8B_quark_ryzenai_llm_hybrid.json new file mode 100644 index 000000000..8d274805c --- /dev/null +++ b/Qwen-Qwen3-8B/RyzenAI/Qwen3-8B_quark_ryzenai_llm_hybrid.json @@ -0,0 +1,47 @@ +{ + "input_model": { + "type": "HFModel", + "model_path": "Qwen/Qwen3-8B" + }, + "passes": { + "qq": { + "type": "QuarkQuantization", + "quant_scheme": "uint4_wo_128", + "quant_algo": "awq", + "dataset": "pileval_for_awq_benchmark", + "data_type": "bfloat16", + "num_calib_data": 128, + "model_export": [ + "hf_format" + ], + "exclude_layers": [], + "layer_quant_scheme": [ + [ + "lm_head", + "uint4_wo_32" + ] + ], + "algo_configs": { + "awq": { + "scaling_layers": [], + "model_decoder_layers": "model.layers" + } + } + }, + "mg": { + "type": "RyzenGenerateModelLLM", + "mode": "hybrid", + "recipe": "prefill_fusion", + "extra_options": { + "model_type": "qwen3-8b", + "max_seq_len": "16384", + "flash_mha": "true", + "attributes": "enable_flat_kv=true enable_flatmlp=true" + } + } + }, + "log_severity_level": 1, + "output_dir": "models/Qwen3-8B-hybrid-rai", + "cache_dir": "olive_cache", + "no_artifacts": true +} diff --git a/Qwen-Qwen3-8B/RyzenAI/README.md b/Qwen-Qwen3-8B/RyzenAI/README.md new file mode 100644 index 000000000..1eae1dd08 --- /dev/null +++ b/Qwen-Qwen3-8B/RyzenAI/README.md @@ -0,0 +1,96 @@ +# Model Optimization and Quantization for AMD NPU +This folder contains sample Olive configuration to optimize Qwen models for AMD NPU. + +## ✅ Supported Models and Configs + +| Model Name (Hugging Face) | Config File Name | Device | +| :------------------------ | :------------------------------------------------- | :-------------------- | +| `Qwen/Qwen3-8B` | `Qwen3-8B_quark_ryzenai_llm.json` | npu | +| `Qwen/Qwen3-8B` | `Qwen3-8B_quark_ryzenai_llm_full_fusion.json` * | npu (higher throughput)| +| `Qwen/Qwen3-8B` | `Qwen3-8B_quark_ryzenai_llm_hybrid.json` * | npu + gpu hybrid | + +> \* Same as the MLPerf submission recipe. + +## **Run the Quantization Config** + +### **Quark quantization** + +For LLMs - follow the below commands to generate the optimized model for RyzenAI Execution Provider. + +**Platform Support:** +- ✅ **Windows with CUDA** - Supported +- ✅ **Windows with CPU** - Supported +- ⏳ **Planned for future release:** Linux with ROCm, Linux with CUDA, Windows with ROCm + +For more details about quark, see the [Quark Documentation](https://quark.docs.amd.com/latest/) + +#### **Create a Python 3.12 conda environment and run the below commands** +```bash +conda create -n olive python=3.12 +conda activate olive +``` + +#### **Install Olive** + +**Option 1: Install from PyPI** +```bash +pip install olive-ai[auto-opt] +pip install transformers onnxruntime-genai +``` + +**Option 2: Install from source** +```bash +git clone https://github.com/microsoft/Olive.git +cd Olive +pip install -e . +pip install -r requirements.txt +``` + +#### **Install RyzenAI LLM dependencies** + +```bash +cd olive-recipes/Qwen-Qwen3-8B/RyzenAI +pip install --force-reinstall -r requirements_ryzenai_llm.txt +``` + +#### **Install PyTorch** + +Make sure to install the correct version of PyTorch before running quantization: + +**For AMD GPUs (ROCm):** +```bash +pip3 install torch torchvision torchaudio --index-url https://download.pytorch.org/whl/rocm6.1 + +python -c "import torch; print(torch.cuda.is_available())" # Must return `True` +``` + +**For NVIDIA GPUs (CUDA):** +```bash +pip install torch==2.7.1 torchvision==0.22.1 torchaudio==2.7.1 --index-url https://download.pytorch.org/whl/cu128 + +python -c "import torch; print(torch.cuda.is_available())" # Must return `True` +``` + +**For CPU-only (Windows):** +```bash +pip install torch==2.7.0 torchvision torchaudio --index-url https://download.pytorch.org/whl/cpu +python -c "import torch; print(torch.__version__)" # Should print 2.7.0+cpu +``` + +#### **Generate optimized LLM model for RyzenAI NPU** +Follow the above setup instructions, then run the below command to generate the optimized LLM model for RyzenAI EP + +```bash +# Qwen3-8B (token_fusion) +olive run --config Qwen3-8B_quark_ryzenai_llm.json + +# Qwen3-8B (full_fusion) *MLPerf +olive run --config Qwen3-8B_quark_ryzenai_llm_full_fusion.json + +# Qwen3-8B (hybrid) *MLPerf +olive run --config Qwen3-8B_quark_ryzenai_llm_hybrid.json +``` + +✅ Optimized model saved in: `models/Qwen3-8B-rai/` (and `-full_fusion-rai` / `-hybrid-rai` for the MLPerf recipes) + +> **Note:** Output model is saved in `output_dir` mentioned in the json files. diff --git a/Qwen-Qwen3-8B/RyzenAI/info.yaml b/Qwen-Qwen3-8B/RyzenAI/info.yaml new file mode 100644 index 000000000..3059c388d --- /dev/null +++ b/Qwen-Qwen3-8B/RyzenAI/info.yaml @@ -0,0 +1,16 @@ +arch: qwen3 +recipes: + - name: Qwen3-8B_RyzenAI + file: Qwen3-8B_quark_ryzenai_llm.json + devices: npu + eps: RyzenAIExecutionProvider + + - name: Qwen3-8B_RyzenAI (full_fusion) + file: Qwen3-8B_quark_ryzenai_llm_full_fusion.json + devices: npu + eps: RyzenAIExecutionProvider + + - name: Qwen3-8B_RyzenAI (hybrid) + file: Qwen3-8B_quark_ryzenai_llm_hybrid.json + devices: [npu, gpu] + eps: RyzenAIExecutionProvider diff --git a/Qwen-Qwen3-8B/RyzenAI/requirements_ryzenai_llm.txt b/Qwen-Qwen3-8B/RyzenAI/requirements_ryzenai_llm.txt new file mode 100644 index 000000000..22bd7b945 --- /dev/null +++ b/Qwen-Qwen3-8B/RyzenAI/requirements_ryzenai_llm.txt @@ -0,0 +1,27 @@ +--extra-index-url=https://pypi.amd.com/olive/1.8.0-8D/simple/ +# AMD model generation +--extra-index-url=https://pypi.amd.com/simple +accelerate + +# Quark +amd-quark==0.12.post1 +datasets +evaluate + +model-generate + +nltk +numpy==2.5.1 + +# Pin onnx version +onnx==1.22.0 +onnxruntime +onnxruntime-genai==0.14.0 +onnxsim +optimum + +ryzenai-dynamic-dispatch +ryzenai-onnx-utils +sentencepiece +tabulate +transformers==4.57.6 diff --git a/codellama-CodeLlama-7b-Instruct-hf/RyzenAI/CodeLlama-7b-Instruct-hf_quark_ryzenai_llm.json b/codellama-CodeLlama-7b-Instruct-hf/RyzenAI/CodeLlama-7b-Instruct-hf_quark_ryzenai_llm.json index 5ee508bcb..18ad616db 100644 --- a/codellama-CodeLlama-7b-Instruct-hf/RyzenAI/CodeLlama-7b-Instruct-hf_quark_ryzenai_llm.json +++ b/codellama-CodeLlama-7b-Instruct-hf/RyzenAI/CodeLlama-7b-Instruct-hf_quark_ryzenai_llm.json @@ -13,7 +13,10 @@ }, "mg": { "type": "RyzenGenerateModelLLM", - "recipe": "token_fusion" + "recipe": "token_fusion", + "extra_options": { + "max_seq_len": "16384" + } } }, "log_severity_level": 1, diff --git a/codellama-CodeLlama-7b-Instruct-hf/RyzenAI/requirements_ryzenai_llm.txt b/codellama-CodeLlama-7b-Instruct-hf/RyzenAI/requirements_ryzenai_llm.txt index 53ba97fc9..22bd7b945 100644 --- a/codellama-CodeLlama-7b-Instruct-hf/RyzenAI/requirements_ryzenai_llm.txt +++ b/codellama-CodeLlama-7b-Instruct-hf/RyzenAI/requirements_ryzenai_llm.txt @@ -1,22 +1,22 @@ ---extra-index-url=https://pypi.amd.com/olive/1.7.1-6D/simple/ +--extra-index-url=https://pypi.amd.com/olive/1.8.0-8D/simple/ # AMD model generation --extra-index-url=https://pypi.amd.com/simple accelerate # Quark -amd-quark==0.11 +amd-quark==0.12.post1 datasets evaluate model-generate nltk -numpy==1.26.4 +numpy==2.5.1 # Pin onnx version -onnx==1.18.0 +onnx==1.22.0 onnxruntime -onnxruntime-genai +onnxruntime-genai==0.14.0 onnxsim optimum diff --git a/deepseek-ai-DeepSeek-R1-Distill-Llama-8B/RyzenAI/DeepSeek-R1-Distill-Llama-8B_quark_ryzenai_llm.json b/deepseek-ai-DeepSeek-R1-Distill-Llama-8B/RyzenAI/DeepSeek-R1-Distill-Llama-8B_quark_ryzenai_llm.json index 4726057ab..8f752da62 100644 --- a/deepseek-ai-DeepSeek-R1-Distill-Llama-8B/RyzenAI/DeepSeek-R1-Distill-Llama-8B_quark_ryzenai_llm.json +++ b/deepseek-ai-DeepSeek-R1-Distill-Llama-8B/RyzenAI/DeepSeek-R1-Distill-Llama-8B_quark_ryzenai_llm.json @@ -16,7 +16,10 @@ }, "mg": { "type": "RyzenGenerateModelLLM", - "recipe": "token_fusion" + "recipe": "token_fusion", + "extra_options": { + "max_seq_len": "16384" + } } }, "log_severity_level": 1, diff --git a/deepseek-ai-DeepSeek-R1-Distill-Llama-8B/RyzenAI/requirements_ryzenai_llm.txt b/deepseek-ai-DeepSeek-R1-Distill-Llama-8B/RyzenAI/requirements_ryzenai_llm.txt index 53ba97fc9..22bd7b945 100644 --- a/deepseek-ai-DeepSeek-R1-Distill-Llama-8B/RyzenAI/requirements_ryzenai_llm.txt +++ b/deepseek-ai-DeepSeek-R1-Distill-Llama-8B/RyzenAI/requirements_ryzenai_llm.txt @@ -1,22 +1,22 @@ ---extra-index-url=https://pypi.amd.com/olive/1.7.1-6D/simple/ +--extra-index-url=https://pypi.amd.com/olive/1.8.0-8D/simple/ # AMD model generation --extra-index-url=https://pypi.amd.com/simple accelerate # Quark -amd-quark==0.11 +amd-quark==0.12.post1 datasets evaluate model-generate nltk -numpy==1.26.4 +numpy==2.5.1 # Pin onnx version -onnx==1.18.0 +onnx==1.22.0 onnxruntime -onnxruntime-genai +onnxruntime-genai==0.14.0 onnxsim optimum diff --git a/deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B/RyzenAI/DeepSeek-R1-Distill-Qwen-1.5B_quark_ryzenai_llm.json b/deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B/RyzenAI/DeepSeek-R1-Distill-Qwen-1.5B_quark_ryzenai_llm.json index b85d3196b..75efa4f76 100644 --- a/deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B/RyzenAI/DeepSeek-R1-Distill-Qwen-1.5B_quark_ryzenai_llm.json +++ b/deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B/RyzenAI/DeepSeek-R1-Distill-Qwen-1.5B_quark_ryzenai_llm.json @@ -13,7 +13,10 @@ }, "mg": { "type": "RyzenGenerateModelLLM", - "recipe": "token_fusion" + "recipe": "token_fusion", + "extra_options": { + "max_seq_len": "16384" + } } }, "log_severity_level": 1, diff --git a/deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B/RyzenAI/requirements_ryzenai_llm.txt b/deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B/RyzenAI/requirements_ryzenai_llm.txt index 53ba97fc9..22bd7b945 100644 --- a/deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B/RyzenAI/requirements_ryzenai_llm.txt +++ b/deepseek-ai-DeepSeek-R1-Distill-Qwen-1.5B/RyzenAI/requirements_ryzenai_llm.txt @@ -1,22 +1,22 @@ ---extra-index-url=https://pypi.amd.com/olive/1.7.1-6D/simple/ +--extra-index-url=https://pypi.amd.com/olive/1.8.0-8D/simple/ # AMD model generation --extra-index-url=https://pypi.amd.com/simple accelerate # Quark -amd-quark==0.11 +amd-quark==0.12.post1 datasets evaluate model-generate nltk -numpy==1.26.4 +numpy==2.5.1 # Pin onnx version -onnx==1.18.0 +onnx==1.22.0 onnxruntime -onnxruntime-genai +onnxruntime-genai==0.14.0 onnxsim optimum diff --git a/deepseek-ai-DeepSeek-R1-Distill-Qwen-7B/RyzenAI/DeepSeek-R1-Distill-Qwen-7B_quark_ryzenai_llm.json b/deepseek-ai-DeepSeek-R1-Distill-Qwen-7B/RyzenAI/DeepSeek-R1-Distill-Qwen-7B_quark_ryzenai_llm.json index 88cb7edc5..dacdf161b 100644 --- a/deepseek-ai-DeepSeek-R1-Distill-Qwen-7B/RyzenAI/DeepSeek-R1-Distill-Qwen-7B_quark_ryzenai_llm.json +++ b/deepseek-ai-DeepSeek-R1-Distill-Qwen-7B/RyzenAI/DeepSeek-R1-Distill-Qwen-7B_quark_ryzenai_llm.json @@ -13,7 +13,10 @@ }, "mg": { "type": "RyzenGenerateModelLLM", - "recipe": "token_fusion" + "recipe": "token_fusion", + "extra_options": { + "max_seq_len": "16384" + } } }, "log_severity_level": 1, diff --git a/deepseek-ai-DeepSeek-R1-Distill-Qwen-7B/RyzenAI/requirements_ryzenai_llm.txt b/deepseek-ai-DeepSeek-R1-Distill-Qwen-7B/RyzenAI/requirements_ryzenai_llm.txt index 53ba97fc9..22bd7b945 100644 --- a/deepseek-ai-DeepSeek-R1-Distill-Qwen-7B/RyzenAI/requirements_ryzenai_llm.txt +++ b/deepseek-ai-DeepSeek-R1-Distill-Qwen-7B/RyzenAI/requirements_ryzenai_llm.txt @@ -1,22 +1,22 @@ ---extra-index-url=https://pypi.amd.com/olive/1.7.1-6D/simple/ +--extra-index-url=https://pypi.amd.com/olive/1.8.0-8D/simple/ # AMD model generation --extra-index-url=https://pypi.amd.com/simple accelerate # Quark -amd-quark==0.11 +amd-quark==0.12.post1 datasets evaluate model-generate nltk -numpy==1.26.4 +numpy==2.5.1 # Pin onnx version -onnx==1.18.0 +onnx==1.22.0 onnxruntime -onnxruntime-genai +onnxruntime-genai==0.14.0 onnxsim optimum diff --git a/gpt-oss-20b/RyzenAI/README.md b/gpt-oss-20b/RyzenAI/README.md index bb41d29bc..b2125c562 100644 --- a/gpt-oss-20b/RyzenAI/README.md +++ b/gpt-oss-20b/RyzenAI/README.md @@ -86,7 +86,6 @@ hf download onnxruntime/gpt-oss-20b-onnx --include "cpu_and_mobile/cpu-int4-rtn- 2. Run the Olive recipe: ```bash -# Phi-4-mini-instruct olive run --config gpt-oss-20b_quark_ryzenai_llm.json ``` diff --git a/gpt-oss-20b/RyzenAI/requirements_ryzenai_llm.txt b/gpt-oss-20b/RyzenAI/requirements_ryzenai_llm.txt index 7b12ff627..7ad8a2207 100644 --- a/gpt-oss-20b/RyzenAI/requirements_ryzenai_llm.txt +++ b/gpt-oss-20b/RyzenAI/requirements_ryzenai_llm.txt @@ -1,10 +1,10 @@ ---extra-index-url=https://pypi.amd.com/olive/1.7.1-6D/simple/ +--extra-index-url=https://pypi.amd.com/olive/1.8.0-8D/simple/ # AMD model generation --extra-index-url=https://pypi.amd.com/simple accelerate # Quark -amd-quark==0.11 +amd-quark==0.12.post1 datasets evaluate huggingface_hub @@ -12,12 +12,12 @@ huggingface_hub model-generate nltk -numpy==1.26.4 +numpy==2.5.1 # Pin onnx version -onnx==1.18.0 +onnx==1.22.0 onnxruntime -onnxruntime-genai +onnxruntime-genai==0.14.0 onnxsim optimum diff --git a/meta-llama-Llama-2-7b-chat-hf/RyzenAI/requirements_ryzenai_llm.txt b/meta-llama-Llama-2-7b-chat-hf/RyzenAI/requirements_ryzenai_llm.txt index 53ba97fc9..22bd7b945 100644 --- a/meta-llama-Llama-2-7b-chat-hf/RyzenAI/requirements_ryzenai_llm.txt +++ b/meta-llama-Llama-2-7b-chat-hf/RyzenAI/requirements_ryzenai_llm.txt @@ -1,22 +1,22 @@ ---extra-index-url=https://pypi.amd.com/olive/1.7.1-6D/simple/ +--extra-index-url=https://pypi.amd.com/olive/1.8.0-8D/simple/ # AMD model generation --extra-index-url=https://pypi.amd.com/simple accelerate # Quark -amd-quark==0.11 +amd-quark==0.12.post1 datasets evaluate model-generate nltk -numpy==1.26.4 +numpy==2.5.1 # Pin onnx version -onnx==1.18.0 +onnx==1.22.0 onnxruntime -onnxruntime-genai +onnxruntime-genai==0.14.0 onnxsim optimum diff --git a/meta-llama-Llama-2-7b-hf/RyzenAI/Llama-2-7b-hf_quark_ryzenai_llm.json b/meta-llama-Llama-2-7b-hf/RyzenAI/Llama-2-7b-hf_quark_ryzenai_llm.json index 94857712d..7d172b259 100644 --- a/meta-llama-Llama-2-7b-hf/RyzenAI/Llama-2-7b-hf_quark_ryzenai_llm.json +++ b/meta-llama-Llama-2-7b-hf/RyzenAI/Llama-2-7b-hf_quark_ryzenai_llm.json @@ -13,7 +13,10 @@ }, "mg": { "type": "RyzenGenerateModelLLM", - "recipe": "token_fusion" + "recipe": "token_fusion", + "extra_options": { + "max_seq_len": "16384" + } } }, "log_severity_level": 1, diff --git a/meta-llama-Llama-2-7b-hf/RyzenAI/requirements_ryzenai_llm.txt b/meta-llama-Llama-2-7b-hf/RyzenAI/requirements_ryzenai_llm.txt index 53ba97fc9..22bd7b945 100644 --- a/meta-llama-Llama-2-7b-hf/RyzenAI/requirements_ryzenai_llm.txt +++ b/meta-llama-Llama-2-7b-hf/RyzenAI/requirements_ryzenai_llm.txt @@ -1,22 +1,22 @@ ---extra-index-url=https://pypi.amd.com/olive/1.7.1-6D/simple/ +--extra-index-url=https://pypi.amd.com/olive/1.8.0-8D/simple/ # AMD model generation --extra-index-url=https://pypi.amd.com/simple accelerate # Quark -amd-quark==0.11 +amd-quark==0.12.post1 datasets evaluate model-generate nltk -numpy==1.26.4 +numpy==2.5.1 # Pin onnx version -onnx==1.18.0 +onnx==1.22.0 onnxruntime -onnxruntime-genai +onnxruntime-genai==0.14.0 onnxsim optimum diff --git a/meta-llama-Llama-3.1-8B-Instruct/RyzenAI/Llama-3.1-8B-Instruct_quark_ryzenai_llm.json b/meta-llama-Llama-3.1-8B-Instruct/RyzenAI/Llama-3.1-8B-Instruct_quark_ryzenai_llm.json index 40ca1a48d..17c28991b 100644 --- a/meta-llama-Llama-3.1-8B-Instruct/RyzenAI/Llama-3.1-8B-Instruct_quark_ryzenai_llm.json +++ b/meta-llama-Llama-3.1-8B-Instruct/RyzenAI/Llama-3.1-8B-Instruct_quark_ryzenai_llm.json @@ -13,7 +13,10 @@ }, "mg": { "type": "RyzenGenerateModelLLM", - "recipe": "token_fusion" + "recipe": "token_fusion", + "extra_options": { + "model_type": "llama3-8b" + } } }, "log_severity_level": 1, diff --git a/meta-llama-Llama-3.1-8B-Instruct/RyzenAI/Llama-3.1-8B-Instruct_quark_ryzenai_llm_full_fusion.json b/meta-llama-Llama-3.1-8B-Instruct/RyzenAI/Llama-3.1-8B-Instruct_quark_ryzenai_llm_full_fusion.json new file mode 100644 index 000000000..c5fae2de1 --- /dev/null +++ b/meta-llama-Llama-3.1-8B-Instruct/RyzenAI/Llama-3.1-8B-Instruct_quark_ryzenai_llm_full_fusion.json @@ -0,0 +1,35 @@ +{ + "input_model": { + "type": "HFModel", + "model_path": "meta-llama/Llama-3.1-8B-Instruct" + }, + "passes": { + "qq": { + "type": "QuarkQuantization", + "quant_scheme": "uint4_wo_128", + "quant_algo": "awq", + "dataset": "pileval_for_awq_benchmark", + "data_type": "bfloat16", + "num_calib_data": 128, + "model_export": [ + "hf_format" + ], + "exclude_layers": [] + }, + "mg": { + "type": "RyzenGenerateModelLLM", + "mode": "npu", + "recipe": "full_fusion", + "extra_options": { + "model_type": "llama3-8b", + "max_seq_len": "16384", + "flash_mha": "true", + "attributes": "enable_flat_kv=true enable_flatmlp=true" + } + } + }, + "log_severity_level": 1, + "output_dir": "models/Llama-3.1-8B-Instruct-full_fusion-rai", + "cache_dir": "olive_cache", + "no_artifacts": true +} diff --git a/meta-llama-Llama-3.1-8B-Instruct/RyzenAI/Llama-3.1-8B-Instruct_quark_ryzenai_llm_hybrid.json b/meta-llama-Llama-3.1-8B-Instruct/RyzenAI/Llama-3.1-8B-Instruct_quark_ryzenai_llm_hybrid.json new file mode 100644 index 000000000..c348a50a9 --- /dev/null +++ b/meta-llama-Llama-3.1-8B-Instruct/RyzenAI/Llama-3.1-8B-Instruct_quark_ryzenai_llm_hybrid.json @@ -0,0 +1,35 @@ +{ + "input_model": { + "type": "HFModel", + "model_path": "meta-llama/Llama-3.1-8B-Instruct" + }, + "passes": { + "qq": { + "type": "QuarkQuantization", + "quant_scheme": "uint4_wo_128", + "quant_algo": "awq", + "dataset": "pileval_for_awq_benchmark", + "data_type": "bfloat16", + "num_calib_data": 128, + "model_export": [ + "hf_format" + ], + "exclude_layers": [] + }, + "mg": { + "type": "RyzenGenerateModelLLM", + "mode": "hybrid", + "recipe": "prefill_fusion", + "extra_options": { + "model_type": "llama3-8b", + "max_seq_len": "16384", + "flash_mha": "true", + "attributes": "enable_flat_kv=true enable_flatmlp=true" + } + } + }, + "log_severity_level": 1, + "output_dir": "models/Llama-3.1-8B-Instruct-hybrid-rai", + "cache_dir": "olive_cache", + "no_artifacts": true +} diff --git a/meta-llama-Llama-3.1-8B-Instruct/RyzenAI/README.md b/meta-llama-Llama-3.1-8B-Instruct/RyzenAI/README.md index 8f7243b0e..828aadd55 100644 --- a/meta-llama-Llama-3.1-8B-Instruct/RyzenAI/README.md +++ b/meta-llama-Llama-3.1-8B-Instruct/RyzenAI/README.md @@ -3,9 +3,14 @@ This folder contains sample Olive configuration to optimize LLaMA 3 models for AMD NPU. ## ✅ Supported Models and Configs -| Model Name (Hugging Face) | Config File Name | -|:--------------------------------------------------|:----------------------------------| -| `meta-llama/Llama-3.1-8B-Instruct` | `Llama-3.1-8B-Instruct_quark_ryzenai_llm.json` | + +| Model Name (Hugging Face) | Config File Name | Device | +| :---------------------------------- | :---------------------------------------------------------- | :-------------------- | +| `meta-llama/Llama-3.1-8B-Instruct` | `Llama-3.1-8B-Instruct_quark_ryzenai_llm.json` | npu | +| `meta-llama/Llama-3.1-8B-Instruct` | `Llama-3.1-8B-Instruct_quark_ryzenai_llm_full_fusion.json` *| npu (higher throughput)| +| `meta-llama/Llama-3.1-8B-Instruct` | `Llama-3.1-8B-Instruct_quark_ryzenai_llm_hybrid.json` * | npu + gpu hybrid | + +> \* Same as the MLPerf submission recipe. ## **Run the Quantization Config** @@ -49,8 +54,6 @@ cd olive-recipes/meta-llama-Llama-3.1-8B-Instruct/RyzenAI pip install --force-reinstall -r requirements_ryzenai_llm.txt ``` - - #### **Install PyTorch** Make sure to install the correct version of PyTorch before running quantization: @@ -79,19 +82,16 @@ python -c "import torch; print(torch.__version__)" # Should print 2.7.0+cpu Follow the above setup instructions, then run the below command to generate the optimized LLM model for RyzenAI EP ```bash -# Llama-3.2-1B-Instruct -olive run --config Llama-3.2-1B-Instruct_quark_ryzenai_llm.json - -# Llama-3.2-3B-Instruct -olive run --config Llama-3.2-3B-Instruct_quark_ryzenai_llm.json +# Llama-3.1-8B-Instruct (token_fusion) +olive run --config Llama-3.1-8B-Instruct_quark_ryzenai_llm.json -# Meta-Llama-3-8B -olive run --config Meta-Llama-3-8B_quark_ryzenai_llm.json +# Llama-3.1-8B-Instruct (full_fusion) *MLPerf +olive run --config Llama-3.1-8B-Instruct_quark_ryzenai_llm_full_fusion.json -# Meta-Llama-3.1-8B -olive run --config Meta-Llama-3.1-8B_quark_ryzenai_llm.json +# Llama-3.1-8B-Instruct (hybrid) *MLPerf +olive run --config Llama-3.1-8B-Instruct_quark_ryzenai_llm_hybrid.json ``` -✅ Optimized model saved in: `models/Meta-Llama-3-8B-rai/` +✅ Optimized model saved in: `models/Llama-3.1-8B-Instruct-rai/` (and `-full_fusion-rai` / `-hybrid-rai` for the MLPerf recipes) > **Note:** Output model is saved in `output_dir` mentioned in the json files. diff --git a/meta-llama-Llama-3.1-8B-Instruct/RyzenAI/info.yaml b/meta-llama-Llama-3.1-8B-Instruct/RyzenAI/info.yaml index fc4a1d6d0..41b329e70 100644 --- a/meta-llama-Llama-3.1-8B-Instruct/RyzenAI/info.yaml +++ b/meta-llama-Llama-3.1-8B-Instruct/RyzenAI/info.yaml @@ -4,3 +4,13 @@ recipes: file: Llama-3.1-8B-Instruct_quark_ryzenai_llm.json devices: npu eps: RyzenAIExecutionProvider + + - name: Llama-3.1-8B-Instruct_RyzenAI (full_fusion) + file: Llama-3.1-8B-Instruct_quark_ryzenai_llm_full_fusion.json + devices: npu + eps: RyzenAIExecutionProvider + + - name: Llama-3.1-8B-Instruct_RyzenAI (hybrid) + file: Llama-3.1-8B-Instruct_quark_ryzenai_llm_hybrid.json + devices: [npu, gpu] + eps: RyzenAIExecutionProvider diff --git a/meta-llama-Llama-3.1-8B-Instruct/RyzenAI/requirements_ryzenai_llm.txt b/meta-llama-Llama-3.1-8B-Instruct/RyzenAI/requirements_ryzenai_llm.txt index 53ba97fc9..22bd7b945 100644 --- a/meta-llama-Llama-3.1-8B-Instruct/RyzenAI/requirements_ryzenai_llm.txt +++ b/meta-llama-Llama-3.1-8B-Instruct/RyzenAI/requirements_ryzenai_llm.txt @@ -1,22 +1,22 @@ ---extra-index-url=https://pypi.amd.com/olive/1.7.1-6D/simple/ +--extra-index-url=https://pypi.amd.com/olive/1.8.0-8D/simple/ # AMD model generation --extra-index-url=https://pypi.amd.com/simple accelerate # Quark -amd-quark==0.11 +amd-quark==0.12.post1 datasets evaluate model-generate nltk -numpy==1.26.4 +numpy==2.5.1 # Pin onnx version -onnx==1.18.0 +onnx==1.22.0 onnxruntime -onnxruntime-genai +onnxruntime-genai==0.14.0 onnxsim optimum diff --git a/meta-llama-Llama-3.1-8B/RyzenAI/Meta-Llama-3.1-8B_quark_ryzenai_llm.json b/meta-llama-Llama-3.1-8B/RyzenAI/Meta-Llama-3.1-8B_quark_ryzenai_llm.json index 182eacf99..788a97902 100644 --- a/meta-llama-Llama-3.1-8B/RyzenAI/Meta-Llama-3.1-8B_quark_ryzenai_llm.json +++ b/meta-llama-Llama-3.1-8B/RyzenAI/Meta-Llama-3.1-8B_quark_ryzenai_llm.json @@ -13,7 +13,10 @@ }, "mg": { "type": "RyzenGenerateModelLLM", - "recipe": "token_fusion" + "recipe": "token_fusion", + "extra_options": { + "max_seq_len": "16384" + } } }, "log_severity_level": 1, diff --git a/meta-llama-Llama-3.1-8B/RyzenAI/requirements_ryzenai_llm.txt b/meta-llama-Llama-3.1-8B/RyzenAI/requirements_ryzenai_llm.txt index 53ba97fc9..22bd7b945 100644 --- a/meta-llama-Llama-3.1-8B/RyzenAI/requirements_ryzenai_llm.txt +++ b/meta-llama-Llama-3.1-8B/RyzenAI/requirements_ryzenai_llm.txt @@ -1,22 +1,22 @@ ---extra-index-url=https://pypi.amd.com/olive/1.7.1-6D/simple/ +--extra-index-url=https://pypi.amd.com/olive/1.8.0-8D/simple/ # AMD model generation --extra-index-url=https://pypi.amd.com/simple accelerate # Quark -amd-quark==0.11 +amd-quark==0.12.post1 datasets evaluate model-generate nltk -numpy==1.26.4 +numpy==2.5.1 # Pin onnx version -onnx==1.18.0 +onnx==1.22.0 onnxruntime -onnxruntime-genai +onnxruntime-genai==0.14.0 onnxsim optimum diff --git a/meta-llama-Llama-3.2-1B-Instruct/RyzenAI/Llama-3.2-1B-Instruct_quark_ryzenai_llm.json b/meta-llama-Llama-3.2-1B-Instruct/RyzenAI/Llama-3.2-1B-Instruct_quark_ryzenai_llm.json index 5fcb6dbc9..992f2598f 100644 --- a/meta-llama-Llama-3.2-1B-Instruct/RyzenAI/Llama-3.2-1B-Instruct_quark_ryzenai_llm.json +++ b/meta-llama-Llama-3.2-1B-Instruct/RyzenAI/Llama-3.2-1B-Instruct_quark_ryzenai_llm.json @@ -13,7 +13,10 @@ }, "mg": { "type": "RyzenGenerateModelLLM", - "recipe": "token_fusion" + "recipe": "token_fusion", + "extra_options": { + "max_seq_len": "16384" + } } }, "log_severity_level": 1, diff --git a/meta-llama-Llama-3.2-1B-Instruct/RyzenAI/requirements_ryzenai_llm.txt b/meta-llama-Llama-3.2-1B-Instruct/RyzenAI/requirements_ryzenai_llm.txt index 53ba97fc9..22bd7b945 100644 --- a/meta-llama-Llama-3.2-1B-Instruct/RyzenAI/requirements_ryzenai_llm.txt +++ b/meta-llama-Llama-3.2-1B-Instruct/RyzenAI/requirements_ryzenai_llm.txt @@ -1,22 +1,22 @@ ---extra-index-url=https://pypi.amd.com/olive/1.7.1-6D/simple/ +--extra-index-url=https://pypi.amd.com/olive/1.8.0-8D/simple/ # AMD model generation --extra-index-url=https://pypi.amd.com/simple accelerate # Quark -amd-quark==0.11 +amd-quark==0.12.post1 datasets evaluate model-generate nltk -numpy==1.26.4 +numpy==2.5.1 # Pin onnx version -onnx==1.18.0 +onnx==1.22.0 onnxruntime -onnxruntime-genai +onnxruntime-genai==0.14.0 onnxsim optimum diff --git a/meta-llama-Llama-3.2-1B/RyzenAI/Llama-3.2-1B_quark_ryzenai_llm.json b/meta-llama-Llama-3.2-1B/RyzenAI/Llama-3.2-1B_quark_ryzenai_llm.json index cb5adf37f..8119c14f2 100644 --- a/meta-llama-Llama-3.2-1B/RyzenAI/Llama-3.2-1B_quark_ryzenai_llm.json +++ b/meta-llama-Llama-3.2-1B/RyzenAI/Llama-3.2-1B_quark_ryzenai_llm.json @@ -13,7 +13,10 @@ }, "mg": { "type": "RyzenGenerateModelLLM", - "recipe": "token_fusion" + "recipe": "token_fusion", + "extra_options": { + "max_seq_len": "16384" + } } }, "log_severity_level": 1, diff --git a/meta-llama-Llama-3.2-1B/RyzenAI/requirements_ryzenai_llm.txt b/meta-llama-Llama-3.2-1B/RyzenAI/requirements_ryzenai_llm.txt index 53ba97fc9..22bd7b945 100644 --- a/meta-llama-Llama-3.2-1B/RyzenAI/requirements_ryzenai_llm.txt +++ b/meta-llama-Llama-3.2-1B/RyzenAI/requirements_ryzenai_llm.txt @@ -1,22 +1,22 @@ ---extra-index-url=https://pypi.amd.com/olive/1.7.1-6D/simple/ +--extra-index-url=https://pypi.amd.com/olive/1.8.0-8D/simple/ # AMD model generation --extra-index-url=https://pypi.amd.com/simple accelerate # Quark -amd-quark==0.11 +amd-quark==0.12.post1 datasets evaluate model-generate nltk -numpy==1.26.4 +numpy==2.5.1 # Pin onnx version -onnx==1.18.0 +onnx==1.22.0 onnxruntime -onnxruntime-genai +onnxruntime-genai==0.14.0 onnxsim optimum diff --git a/meta-llama-Llama-3.2-3B-Instruct/RyzenAI/Llama-3.2-3B-Instruct_quark_ryzenai_llm.json b/meta-llama-Llama-3.2-3B-Instruct/RyzenAI/Llama-3.2-3B-Instruct_quark_ryzenai_llm.json index 34392ee15..7db218b67 100644 --- a/meta-llama-Llama-3.2-3B-Instruct/RyzenAI/Llama-3.2-3B-Instruct_quark_ryzenai_llm.json +++ b/meta-llama-Llama-3.2-3B-Instruct/RyzenAI/Llama-3.2-3B-Instruct_quark_ryzenai_llm.json @@ -13,7 +13,10 @@ }, "mg": { "type": "RyzenGenerateModelLLM", - "recipe": "token_fusion" + "recipe": "token_fusion", + "extra_options": { + "max_seq_len": "16384" + } } }, "log_severity_level": 1, diff --git a/meta-llama-Llama-3.2-3B-Instruct/RyzenAI/requirements_ryzenai_llm.txt b/meta-llama-Llama-3.2-3B-Instruct/RyzenAI/requirements_ryzenai_llm.txt index 53ba97fc9..22bd7b945 100644 --- a/meta-llama-Llama-3.2-3B-Instruct/RyzenAI/requirements_ryzenai_llm.txt +++ b/meta-llama-Llama-3.2-3B-Instruct/RyzenAI/requirements_ryzenai_llm.txt @@ -1,22 +1,22 @@ ---extra-index-url=https://pypi.amd.com/olive/1.7.1-6D/simple/ +--extra-index-url=https://pypi.amd.com/olive/1.8.0-8D/simple/ # AMD model generation --extra-index-url=https://pypi.amd.com/simple accelerate # Quark -amd-quark==0.11 +amd-quark==0.12.post1 datasets evaluate model-generate nltk -numpy==1.26.4 +numpy==2.5.1 # Pin onnx version -onnx==1.18.0 +onnx==1.22.0 onnxruntime -onnxruntime-genai +onnxruntime-genai==0.14.0 onnxsim optimum diff --git a/meta-llama-Llama-3.2-3B/RyzenAI/Llama-3.2-3B_quark_ryzenai_llm.json b/meta-llama-Llama-3.2-3B/RyzenAI/Llama-3.2-3B_quark_ryzenai_llm.json index 9860d2a5d..d849626ed 100644 --- a/meta-llama-Llama-3.2-3B/RyzenAI/Llama-3.2-3B_quark_ryzenai_llm.json +++ b/meta-llama-Llama-3.2-3B/RyzenAI/Llama-3.2-3B_quark_ryzenai_llm.json @@ -13,7 +13,10 @@ }, "mg": { "type": "RyzenGenerateModelLLM", - "recipe": "token_fusion" + "recipe": "token_fusion", + "extra_options": { + "max_seq_len": "16384" + } } }, "log_severity_level": 1, diff --git a/meta-llama-Llama-3.2-3B/RyzenAI/requirements_ryzenai_llm.txt b/meta-llama-Llama-3.2-3B/RyzenAI/requirements_ryzenai_llm.txt index 53ba97fc9..22bd7b945 100644 --- a/meta-llama-Llama-3.2-3B/RyzenAI/requirements_ryzenai_llm.txt +++ b/meta-llama-Llama-3.2-3B/RyzenAI/requirements_ryzenai_llm.txt @@ -1,22 +1,22 @@ ---extra-index-url=https://pypi.amd.com/olive/1.7.1-6D/simple/ +--extra-index-url=https://pypi.amd.com/olive/1.8.0-8D/simple/ # AMD model generation --extra-index-url=https://pypi.amd.com/simple accelerate # Quark -amd-quark==0.11 +amd-quark==0.12.post1 datasets evaluate model-generate nltk -numpy==1.26.4 +numpy==2.5.1 # Pin onnx version -onnx==1.18.0 +onnx==1.22.0 onnxruntime -onnxruntime-genai +onnxruntime-genai==0.14.0 onnxsim optimum diff --git a/meta-llama-Meta-Llama-3-8B/RyzenAI/Meta-Llama-3-8B_quark_ryzenai_llm.json b/meta-llama-Meta-Llama-3-8B/RyzenAI/Meta-Llama-3-8B_quark_ryzenai_llm.json index 039a3e5ee..373f35219 100644 --- a/meta-llama-Meta-Llama-3-8B/RyzenAI/Meta-Llama-3-8B_quark_ryzenai_llm.json +++ b/meta-llama-Meta-Llama-3-8B/RyzenAI/Meta-Llama-3-8B_quark_ryzenai_llm.json @@ -13,7 +13,10 @@ }, "mg": { "type": "RyzenGenerateModelLLM", - "recipe": "token_fusion" + "recipe": "token_fusion", + "extra_options": { + "max_seq_len": "16384" + } } }, "log_severity_level": 1, diff --git a/meta-llama-Meta-Llama-3-8B/RyzenAI/requirements_ryzenai_llm.txt b/meta-llama-Meta-Llama-3-8B/RyzenAI/requirements_ryzenai_llm.txt index 53ba97fc9..22bd7b945 100644 --- a/meta-llama-Meta-Llama-3-8B/RyzenAI/requirements_ryzenai_llm.txt +++ b/meta-llama-Meta-Llama-3-8B/RyzenAI/requirements_ryzenai_llm.txt @@ -1,22 +1,22 @@ ---extra-index-url=https://pypi.amd.com/olive/1.7.1-6D/simple/ +--extra-index-url=https://pypi.amd.com/olive/1.8.0-8D/simple/ # AMD model generation --extra-index-url=https://pypi.amd.com/simple accelerate # Quark -amd-quark==0.11 +amd-quark==0.12.post1 datasets evaluate model-generate nltk -numpy==1.26.4 +numpy==2.5.1 # Pin onnx version -onnx==1.18.0 +onnx==1.22.0 onnxruntime -onnxruntime-genai +onnxruntime-genai==0.14.0 onnxsim optimum diff --git a/microsoft-Phi-3-mini-128k-instruct/RyzenAI/Phi-3-mini-128k-instruct_quark_ryzenai_llm.json b/microsoft-Phi-3-mini-128k-instruct/RyzenAI/Phi-3-mini-128k-instruct_quark_ryzenai_llm.json index ac5b205f1..4c7019e79 100644 --- a/microsoft-Phi-3-mini-128k-instruct/RyzenAI/Phi-3-mini-128k-instruct_quark_ryzenai_llm.json +++ b/microsoft-Phi-3-mini-128k-instruct/RyzenAI/Phi-3-mini-128k-instruct_quark_ryzenai_llm.json @@ -13,7 +13,10 @@ }, "mg": { "type": "RyzenGenerateModelLLM", - "recipe": "token_fusion" + "recipe": "token_fusion", + "extra_options": { + "max_seq_len": "16384" + } } }, "log_severity_level": 1, diff --git a/microsoft-Phi-3-mini-128k-instruct/RyzenAI/requirements_ryzenai_llm.txt b/microsoft-Phi-3-mini-128k-instruct/RyzenAI/requirements_ryzenai_llm.txt index 53ba97fc9..22bd7b945 100644 --- a/microsoft-Phi-3-mini-128k-instruct/RyzenAI/requirements_ryzenai_llm.txt +++ b/microsoft-Phi-3-mini-128k-instruct/RyzenAI/requirements_ryzenai_llm.txt @@ -1,22 +1,22 @@ ---extra-index-url=https://pypi.amd.com/olive/1.7.1-6D/simple/ +--extra-index-url=https://pypi.amd.com/olive/1.8.0-8D/simple/ # AMD model generation --extra-index-url=https://pypi.amd.com/simple accelerate # Quark -amd-quark==0.11 +amd-quark==0.12.post1 datasets evaluate model-generate nltk -numpy==1.26.4 +numpy==2.5.1 # Pin onnx version -onnx==1.18.0 +onnx==1.22.0 onnxruntime -onnxruntime-genai +onnxruntime-genai==0.14.0 onnxsim optimum diff --git a/microsoft-Phi-3-mini-4k-instruct/RyzenAI/Phi-3-mini-4k-instruct_quark_ryzenai_llm.json b/microsoft-Phi-3-mini-4k-instruct/RyzenAI/Phi-3-mini-4k-instruct_quark_ryzenai_llm.json index 80c72a1d7..72a8ef78d 100644 --- a/microsoft-Phi-3-mini-4k-instruct/RyzenAI/Phi-3-mini-4k-instruct_quark_ryzenai_llm.json +++ b/microsoft-Phi-3-mini-4k-instruct/RyzenAI/Phi-3-mini-4k-instruct_quark_ryzenai_llm.json @@ -13,7 +13,10 @@ }, "mg": { "type": "RyzenGenerateModelLLM", - "recipe": "token_fusion" + "recipe": "token_fusion", + "extra_options": { + "max_seq_len": "16384" + } } }, "log_severity_level": 1, diff --git a/microsoft-Phi-3-mini-4k-instruct/RyzenAI/requirements_ryzenai_llm.txt b/microsoft-Phi-3-mini-4k-instruct/RyzenAI/requirements_ryzenai_llm.txt index 53ba97fc9..22bd7b945 100644 --- a/microsoft-Phi-3-mini-4k-instruct/RyzenAI/requirements_ryzenai_llm.txt +++ b/microsoft-Phi-3-mini-4k-instruct/RyzenAI/requirements_ryzenai_llm.txt @@ -1,22 +1,22 @@ ---extra-index-url=https://pypi.amd.com/olive/1.7.1-6D/simple/ +--extra-index-url=https://pypi.amd.com/olive/1.8.0-8D/simple/ # AMD model generation --extra-index-url=https://pypi.amd.com/simple accelerate # Quark -amd-quark==0.11 +amd-quark==0.12.post1 datasets evaluate model-generate nltk -numpy==1.26.4 +numpy==2.5.1 # Pin onnx version -onnx==1.18.0 +onnx==1.22.0 onnxruntime -onnxruntime-genai +onnxruntime-genai==0.14.0 onnxsim optimum diff --git a/microsoft-Phi-3.5-mini-instruct/RyzenAI/Phi-3.5-mini-instruct_quark_ryzenai_llm.json b/microsoft-Phi-3.5-mini-instruct/RyzenAI/Phi-3.5-mini-instruct_quark_ryzenai_llm.json index 74d82f0aa..6a54eaafe 100644 --- a/microsoft-Phi-3.5-mini-instruct/RyzenAI/Phi-3.5-mini-instruct_quark_ryzenai_llm.json +++ b/microsoft-Phi-3.5-mini-instruct/RyzenAI/Phi-3.5-mini-instruct_quark_ryzenai_llm.json @@ -16,7 +16,10 @@ }, "mg": { "type": "RyzenGenerateModelLLM", - "recipe": "token_fusion" + "recipe": "token_fusion", + "extra_options": { + "max_seq_len": "16384" + } } }, "log_severity_level": 1, diff --git a/microsoft-Phi-3.5-mini-instruct/RyzenAI/requirements_ryzenai_llm.txt b/microsoft-Phi-3.5-mini-instruct/RyzenAI/requirements_ryzenai_llm.txt index 53ba97fc9..22bd7b945 100644 --- a/microsoft-Phi-3.5-mini-instruct/RyzenAI/requirements_ryzenai_llm.txt +++ b/microsoft-Phi-3.5-mini-instruct/RyzenAI/requirements_ryzenai_llm.txt @@ -1,22 +1,22 @@ ---extra-index-url=https://pypi.amd.com/olive/1.7.1-6D/simple/ +--extra-index-url=https://pypi.amd.com/olive/1.8.0-8D/simple/ # AMD model generation --extra-index-url=https://pypi.amd.com/simple accelerate # Quark -amd-quark==0.11 +amd-quark==0.12.post1 datasets evaluate model-generate nltk -numpy==1.26.4 +numpy==2.5.1 # Pin onnx version -onnx==1.18.0 +onnx==1.22.0 onnxruntime -onnxruntime-genai +onnxruntime-genai==0.14.0 onnxsim optimum diff --git a/microsoft-Phi-4-mini-instruct/RyzenAI/Phi-4-mini-instruct_quark_ryzenai_llm.json b/microsoft-Phi-4-mini-instruct/RyzenAI/Phi-4-mini-instruct_quark_ryzenai_llm.json index 9ce579788..34afc54e3 100644 --- a/microsoft-Phi-4-mini-instruct/RyzenAI/Phi-4-mini-instruct_quark_ryzenai_llm.json +++ b/microsoft-Phi-4-mini-instruct/RyzenAI/Phi-4-mini-instruct_quark_ryzenai_llm.json @@ -14,7 +14,10 @@ }, "mg": { "type": "RyzenGenerateModelLLM", - "recipe": "token_fusion" + "recipe": "token_fusion", + "extra_options": { + "max_seq_len": "16384" + } } }, "log_severity_level": 1, diff --git a/microsoft-Phi-4-mini-instruct/RyzenAI/Phi-4-mini-instruct_quark_ryzenai_llm_full_fusion.json b/microsoft-Phi-4-mini-instruct/RyzenAI/Phi-4-mini-instruct_quark_ryzenai_llm_full_fusion.json new file mode 100644 index 000000000..16ea294b3 --- /dev/null +++ b/microsoft-Phi-4-mini-instruct/RyzenAI/Phi-4-mini-instruct_quark_ryzenai_llm_full_fusion.json @@ -0,0 +1,36 @@ +{ + "input_model": { + "type": "HFModel", + "model_path": "microsoft/Phi-4-mini-instruct" + }, + "passes": { + "qq": { + "type": "QuarkQuantization", + "quant_scheme": "uint4_wo_128", + "quant_algo": "awq", + "dataset": "pileval_for_awq_benchmark", + "data_type": "bfloat16", + "num_calib_data": 128, + "model_export": [ + "hf_format" + ], + "exclude_layers": [], + "trust_remote_code": false + }, + "mg": { + "type": "RyzenGenerateModelLLM", + "mode": "npu", + "recipe": "full_fusion", + "extra_options": { + "model_type": "phi-4", + "max_seq_len": "16384", + "flash_mha": "true", + "attributes": "enable_flat_kv=true enable_flatmlp=true" + } + } + }, + "log_severity_level": 1, + "output_dir": "models/Phi-4-mini-instruct-full_fusion-rai", + "cache_dir": "olive_cache", + "no_artifacts": true +} diff --git a/microsoft-Phi-4-mini-instruct/RyzenAI/Phi-4-mini-instruct_quark_ryzenai_llm_hybrid.json b/microsoft-Phi-4-mini-instruct/RyzenAI/Phi-4-mini-instruct_quark_ryzenai_llm_hybrid.json new file mode 100644 index 000000000..a84370cf2 --- /dev/null +++ b/microsoft-Phi-4-mini-instruct/RyzenAI/Phi-4-mini-instruct_quark_ryzenai_llm_hybrid.json @@ -0,0 +1,36 @@ +{ + "input_model": { + "type": "HFModel", + "model_path": "microsoft/Phi-4-mini-instruct" + }, + "passes": { + "qq": { + "type": "QuarkQuantization", + "quant_scheme": "uint4_wo_128", + "quant_algo": "awq", + "dataset": "pileval_for_awq_benchmark", + "data_type": "bfloat16", + "num_calib_data": 128, + "model_export": [ + "hf_format" + ], + "exclude_layers": [], + "trust_remote_code": false + }, + "mg": { + "type": "RyzenGenerateModelLLM", + "mode": "hybrid", + "recipe": "prefill_fusion", + "extra_options": { + "model_type": "phi-4", + "max_seq_len": "16384", + "flash_mha": "true", + "attributes": "enable_flat_kv=true enable_flatmlp=true" + } + } + }, + "log_severity_level": 1, + "output_dir": "models/Phi-4-mini-instruct-hybrid-rai", + "cache_dir": "olive_cache", + "no_artifacts": true +} diff --git a/microsoft-Phi-4-mini-instruct/RyzenAI/README.md b/microsoft-Phi-4-mini-instruct/RyzenAI/README.md index 06e0adbaa..3bd13a5ad 100644 --- a/microsoft-Phi-4-mini-instruct/RyzenAI/README.md +++ b/microsoft-Phi-4-mini-instruct/RyzenAI/README.md @@ -4,9 +4,13 @@ This folder contains sample Olive configuration to optimize Phi-4 models for AMD ## ✅ Supported Models and Configs -| Model Name (Hugging Face) | Config File Name | -|:---------------------------------------------------|:----------------------------------| -| `microsoft/Phi-4-mini-reasoning` | `Phi-4-mini-reasoning_quark_ryzenai_llm.json` | +| Model Name (Hugging Face) | Config File Name | Device | +| :------------------------------- | :--------------------------------------------------------- | :-------------------- | +| `microsoft/Phi-4-mini-instruct` | `Phi-4-mini-instruct_quark_ryzenai_llm.json` | npu | +| `microsoft/Phi-4-mini-instruct` | `Phi-4-mini-instruct_quark_ryzenai_llm_full_fusion.json` * | npu (higher throughput)| +| `microsoft/Phi-4-mini-instruct` | `Phi-4-mini-instruct_quark_ryzenai_llm_hybrid.json` * | npu + gpu hybrid | + +> \* Same as the MLPerf submission recipe. ## **Run the Quantization Config** @@ -74,14 +78,19 @@ pip install torch==2.7.0 torchvision torchaudio --index-url https://download.pyt python -c "import torch; print(torch.__version__)" # Should print 2.7.0+cpu ``` - #### **Generate optimized LLM model for RyzenAI NPU** Follow the above setup instructions, then run the below command to generate the optimized LLM model for RyzenAI EP ```bash -# Phi-4-mini-instruct +# Phi-4-mini-instruct (token_fusion) olive run --config Phi-4-mini-instruct_quark_ryzenai_llm.json + +# Phi-4-mini-instruct (full_fusion) *MLPerf +olive run --config Phi-4-mini-instruct_quark_ryzenai_llm_full_fusion.json + +# Phi-4-mini-instruct (hybrid) *MLPerf +olive run --config Phi-4-mini-instruct_quark_ryzenai_llm_hybrid.json ``` -✅ Optimized model saved in: `models/Phi-4-mini-instruct-rai/` +✅ Optimized model saved in: `models/Phi-4-mini-instruct-rai/` (and `-full_fusion-rai` / `-hybrid-rai` for the MLPerf recipes) > **Note:** Output model is saved in `output_dir` mentioned in the json files. diff --git a/microsoft-Phi-4-mini-instruct/RyzenAI/info.yaml b/microsoft-Phi-4-mini-instruct/RyzenAI/info.yaml index 9e93d64b6..3724a23bc 100644 --- a/microsoft-Phi-4-mini-instruct/RyzenAI/info.yaml +++ b/microsoft-Phi-4-mini-instruct/RyzenAI/info.yaml @@ -4,3 +4,13 @@ recipes: file: Phi-4-mini-instruct_quark_ryzenai_llm.json devices: npu eps: RyzenAIExecutionProvider + + - name: Phi-4-mini-instruct_RyzenAI (full_fusion) + file: Phi-4-mini-instruct_quark_ryzenai_llm_full_fusion.json + devices: npu + eps: RyzenAIExecutionProvider + + - name: Phi-4-mini-instruct_RyzenAI (hybrid) + file: Phi-4-mini-instruct_quark_ryzenai_llm_hybrid.json + devices: [npu, gpu] + eps: RyzenAIExecutionProvider diff --git a/microsoft-Phi-4-mini-instruct/RyzenAI/requirements_ryzenai_llm.txt b/microsoft-Phi-4-mini-instruct/RyzenAI/requirements_ryzenai_llm.txt index 53ba97fc9..22bd7b945 100644 --- a/microsoft-Phi-4-mini-instruct/RyzenAI/requirements_ryzenai_llm.txt +++ b/microsoft-Phi-4-mini-instruct/RyzenAI/requirements_ryzenai_llm.txt @@ -1,22 +1,22 @@ ---extra-index-url=https://pypi.amd.com/olive/1.7.1-6D/simple/ +--extra-index-url=https://pypi.amd.com/olive/1.8.0-8D/simple/ # AMD model generation --extra-index-url=https://pypi.amd.com/simple accelerate # Quark -amd-quark==0.11 +amd-quark==0.12.post1 datasets evaluate model-generate nltk -numpy==1.26.4 +numpy==2.5.1 # Pin onnx version -onnx==1.18.0 +onnx==1.22.0 onnxruntime -onnxruntime-genai +onnxruntime-genai==0.14.0 onnxsim optimum diff --git a/microsoft-Phi-4-mini-reasoning/RyzenAI/Phi-4-mini-reasoning_quark_ryzenai_llm.json b/microsoft-Phi-4-mini-reasoning/RyzenAI/Phi-4-mini-reasoning_quark_ryzenai_llm.json index 7dd9f0e5e..9efd9762a 100644 --- a/microsoft-Phi-4-mini-reasoning/RyzenAI/Phi-4-mini-reasoning_quark_ryzenai_llm.json +++ b/microsoft-Phi-4-mini-reasoning/RyzenAI/Phi-4-mini-reasoning_quark_ryzenai_llm.json @@ -13,7 +13,10 @@ }, "mg": { "type": "RyzenGenerateModelLLM", - "recipe": "token_fusion" + "recipe": "token_fusion", + "extra_options": { + "max_seq_len": "16384" + } } }, "log_severity_level": 1, diff --git a/microsoft-Phi-4-mini-reasoning/RyzenAI/requirements_ryzenai_llm.txt b/microsoft-Phi-4-mini-reasoning/RyzenAI/requirements_ryzenai_llm.txt index 53ba97fc9..22bd7b945 100644 --- a/microsoft-Phi-4-mini-reasoning/RyzenAI/requirements_ryzenai_llm.txt +++ b/microsoft-Phi-4-mini-reasoning/RyzenAI/requirements_ryzenai_llm.txt @@ -1,22 +1,22 @@ ---extra-index-url=https://pypi.amd.com/olive/1.7.1-6D/simple/ +--extra-index-url=https://pypi.amd.com/olive/1.8.0-8D/simple/ # AMD model generation --extra-index-url=https://pypi.amd.com/simple accelerate # Quark -amd-quark==0.11 +amd-quark==0.12.post1 datasets evaluate model-generate nltk -numpy==1.26.4 +numpy==2.5.1 # Pin onnx version -onnx==1.18.0 +onnx==1.22.0 onnxruntime -onnxruntime-genai +onnxruntime-genai==0.14.0 onnxsim optimum diff --git a/mistralai-Mistral-7B-Instruct-v0.1/RyzenAI/Mistral-7B-Instruct-v0.1_quark_ryzenai_llm.json b/mistralai-Mistral-7B-Instruct-v0.1/RyzenAI/Mistral-7B-Instruct-v0.1_quark_ryzenai_llm.json index b2cd6878e..57ffee626 100644 --- a/mistralai-Mistral-7B-Instruct-v0.1/RyzenAI/Mistral-7B-Instruct-v0.1_quark_ryzenai_llm.json +++ b/mistralai-Mistral-7B-Instruct-v0.1/RyzenAI/Mistral-7B-Instruct-v0.1_quark_ryzenai_llm.json @@ -13,7 +13,10 @@ }, "mg": { "type": "RyzenGenerateModelLLM", - "recipe": "token_fusion" + "recipe": "token_fusion", + "extra_options": { + "max_seq_len": "16384" + } } }, "log_severity_level": 1, diff --git a/mistralai-Mistral-7B-Instruct-v0.1/RyzenAI/requirements_ryzenai_llm.txt b/mistralai-Mistral-7B-Instruct-v0.1/RyzenAI/requirements_ryzenai_llm.txt index 53ba97fc9..22bd7b945 100644 --- a/mistralai-Mistral-7B-Instruct-v0.1/RyzenAI/requirements_ryzenai_llm.txt +++ b/mistralai-Mistral-7B-Instruct-v0.1/RyzenAI/requirements_ryzenai_llm.txt @@ -1,22 +1,22 @@ ---extra-index-url=https://pypi.amd.com/olive/1.7.1-6D/simple/ +--extra-index-url=https://pypi.amd.com/olive/1.8.0-8D/simple/ # AMD model generation --extra-index-url=https://pypi.amd.com/simple accelerate # Quark -amd-quark==0.11 +amd-quark==0.12.post1 datasets evaluate model-generate nltk -numpy==1.26.4 +numpy==2.5.1 # Pin onnx version -onnx==1.18.0 +onnx==1.22.0 onnxruntime -onnxruntime-genai +onnxruntime-genai==0.14.0 onnxsim optimum diff --git a/mistralai-Mistral-7B-Instruct-v0.2/RyzenAI/Mistral-7B-Instruct-v0.2_quark_ryzenai_llm.json b/mistralai-Mistral-7B-Instruct-v0.2/RyzenAI/Mistral-7B-Instruct-v0.2_quark_ryzenai_llm.json index 08f8287c6..530dc78a1 100644 --- a/mistralai-Mistral-7B-Instruct-v0.2/RyzenAI/Mistral-7B-Instruct-v0.2_quark_ryzenai_llm.json +++ b/mistralai-Mistral-7B-Instruct-v0.2/RyzenAI/Mistral-7B-Instruct-v0.2_quark_ryzenai_llm.json @@ -13,7 +13,10 @@ }, "mg": { "type": "RyzenGenerateModelLLM", - "recipe": "token_fusion" + "recipe": "token_fusion", + "extra_options": { + "max_seq_len": "16384" + } } }, "log_severity_level": 1, diff --git a/mistralai-Mistral-7B-Instruct-v0.2/RyzenAI/requirements_ryzenai_llm.txt b/mistralai-Mistral-7B-Instruct-v0.2/RyzenAI/requirements_ryzenai_llm.txt index 53ba97fc9..22bd7b945 100644 --- a/mistralai-Mistral-7B-Instruct-v0.2/RyzenAI/requirements_ryzenai_llm.txt +++ b/mistralai-Mistral-7B-Instruct-v0.2/RyzenAI/requirements_ryzenai_llm.txt @@ -1,22 +1,22 @@ ---extra-index-url=https://pypi.amd.com/olive/1.7.1-6D/simple/ +--extra-index-url=https://pypi.amd.com/olive/1.8.0-8D/simple/ # AMD model generation --extra-index-url=https://pypi.amd.com/simple accelerate # Quark -amd-quark==0.11 +amd-quark==0.12.post1 datasets evaluate model-generate nltk -numpy==1.26.4 +numpy==2.5.1 # Pin onnx version -onnx==1.18.0 +onnx==1.22.0 onnxruntime -onnxruntime-genai +onnxruntime-genai==0.14.0 onnxsim optimum diff --git a/mistralai-Mistral-7B-Instruct-v0.3/RyzenAI/Mistral-7B-Instruct-v0.3_quark_ryzenai_llm.json b/mistralai-Mistral-7B-Instruct-v0.3/RyzenAI/Mistral-7B-Instruct-v0.3_quark_ryzenai_llm.json index 1f1c74d25..f159e3ac6 100644 --- a/mistralai-Mistral-7B-Instruct-v0.3/RyzenAI/Mistral-7B-Instruct-v0.3_quark_ryzenai_llm.json +++ b/mistralai-Mistral-7B-Instruct-v0.3/RyzenAI/Mistral-7B-Instruct-v0.3_quark_ryzenai_llm.json @@ -13,7 +13,10 @@ }, "mg": { "type": "RyzenGenerateModelLLM", - "recipe": "token_fusion" + "recipe": "token_fusion", + "extra_options": { + "max_seq_len": "16384" + } } }, "log_severity_level": 1, diff --git a/mistralai-Mistral-7B-Instruct-v0.3/RyzenAI/requirements_ryzenai_llm.txt b/mistralai-Mistral-7B-Instruct-v0.3/RyzenAI/requirements_ryzenai_llm.txt index 53ba97fc9..22bd7b945 100644 --- a/mistralai-Mistral-7B-Instruct-v0.3/RyzenAI/requirements_ryzenai_llm.txt +++ b/mistralai-Mistral-7B-Instruct-v0.3/RyzenAI/requirements_ryzenai_llm.txt @@ -1,22 +1,22 @@ ---extra-index-url=https://pypi.amd.com/olive/1.7.1-6D/simple/ +--extra-index-url=https://pypi.amd.com/olive/1.8.0-8D/simple/ # AMD model generation --extra-index-url=https://pypi.amd.com/simple accelerate # Quark -amd-quark==0.11 +amd-quark==0.12.post1 datasets evaluate model-generate nltk -numpy==1.26.4 +numpy==2.5.1 # Pin onnx version -onnx==1.18.0 +onnx==1.22.0 onnxruntime -onnxruntime-genai +onnxruntime-genai==0.14.0 onnxsim optimum diff --git a/mistralai-Mistral-7B-v0.3/RyzenAI/Mistral-7B-v0.3_quark_ryzenai_llm.json b/mistralai-Mistral-7B-v0.3/RyzenAI/Mistral-7B-v0.3_quark_ryzenai_llm.json index d666992d1..7119d27dc 100644 --- a/mistralai-Mistral-7B-v0.3/RyzenAI/Mistral-7B-v0.3_quark_ryzenai_llm.json +++ b/mistralai-Mistral-7B-v0.3/RyzenAI/Mistral-7B-v0.3_quark_ryzenai_llm.json @@ -13,7 +13,10 @@ }, "mg": { "type": "RyzenGenerateModelLLM", - "recipe": "token_fusion" + "recipe": "token_fusion", + "extra_options": { + "max_seq_len": "16384" + } } }, "log_severity_level": 1, diff --git a/mistralai-Mistral-7B-v0.3/RyzenAI/requirements_ryzenai_llm.txt b/mistralai-Mistral-7B-v0.3/RyzenAI/requirements_ryzenai_llm.txt index 53ba97fc9..22bd7b945 100644 --- a/mistralai-Mistral-7B-v0.3/RyzenAI/requirements_ryzenai_llm.txt +++ b/mistralai-Mistral-7B-v0.3/RyzenAI/requirements_ryzenai_llm.txt @@ -1,22 +1,22 @@ ---extra-index-url=https://pypi.amd.com/olive/1.7.1-6D/simple/ +--extra-index-url=https://pypi.amd.com/olive/1.8.0-8D/simple/ # AMD model generation --extra-index-url=https://pypi.amd.com/simple accelerate # Quark -amd-quark==0.11 +amd-quark==0.12.post1 datasets evaluate model-generate nltk -numpy==1.26.4 +numpy==2.5.1 # Pin onnx version -onnx==1.18.0 +onnx==1.22.0 onnxruntime -onnxruntime-genai +onnxruntime-genai==0.14.0 onnxsim optimum