From 1fcd2822814a8471881dd141bd1d44cadc5ce513 Mon Sep 17 00:00:00 2001 From: Prathik Rao Date: Tue, 6 Oct 2026 16:16:52 -0700 Subject: [PATCH 1/2] optim --- .../webgpu/Qwen-Qwen3.8-27B_webgpu_int4.json | 104 ++++++++++++++++++ Qwen-Qwen3.8-27B/webgpu/README.md | 28 +++++ Qwen-Qwen3.8-27B/webgpu/info.yml | 11 ++ 3 files changed, 143 insertions(+) create mode 100644 Qwen-Qwen3.8-27B/webgpu/Qwen-Qwen3.8-27B_webgpu_int4.json create mode 100644 Qwen-Qwen3.8-27B/webgpu/README.md create mode 100644 Qwen-Qwen3.8-27B/webgpu/info.yml diff --git a/Qwen-Qwen3.8-27B/webgpu/Qwen-Qwen3.8-27B_webgpu_int4.json b/Qwen-Qwen3.8-27B/webgpu/Qwen-Qwen3.8-27B_webgpu_int4.json new file mode 100644 index 00000000..1d334922 --- /dev/null +++ b/Qwen-Qwen3.8-27B/webgpu/Qwen-Qwen3.8-27B_webgpu_int4.json @@ -0,0 +1,104 @@ +{ + "input_model": { + "type": "HfModel", + "model_path": "Qwen/Qwen3.8-27B", + "load_kwargs": { + "torch_dtype": "float16" + } + }, + "systems": { + "local_system": { + "type": "LocalSystem", + "accelerators": [ + { + "device": "gpu", + "execution_providers": [ + "WebGpuExecutionProvider" + ] + } + ] + } + }, + "passes": { + "builder_int4_webgpu": { + "type": "ModelBuilder", + "precision": "int4", + "builder_config_version": 2, + "extra_options": { + "prune_lm_head": true, + "hf_token": false, + "exclude_mtp": true + }, + "target_options": { + "quant_config": { + "io_dtype": "fp16", + "weights": { + "type": "int4", + "block_size": 32, + "method": "default", + "symmetric": true, + "op_types": [ + "MatMul" + ] + }, + "format": { + "use_qdq": false, + "matmulnbits_weights_prepacked": 1 + } + }, + "attention": { + "implementation": "paged", + "paged": { + "block_size": 256 + } + }, + "optimizations": { + "fuse_mlp_gate_up": true + } + }, + "runtime_config": { + "engine": { + "dynamic_batching": { + "block_size": 256, + "num_blocks": 8, + "max_batch_size": 1, + "max_scheduled_tokens": 64 + } + }, + "search": { + "chunk_size": 256, + "max_length": 262144 + } + } + }, + "runtime_profiles": { + "type": "GenAIModelRuntimeProfiles", + "runtime_profiles": [ + { + "id": "40-to-less-than-60-gib", + "eligibility": { + "minimum_total_device_memory_bytes": 42949672960, + "maximum_total_device_memory_bytes": 64424509439 + }, + "overlay": { + "engine": { + "dynamic_batching": { + "num_blocks": 8, + "max_batch_size": 1, + "max_scheduled_tokens": 256 + } + }, + "search": { + "chunk_size": 256 + } + } + } + ] + } + }, + "target": "local_system", + "log_severity_level": 0, + "output_dir": "qwen_3.8_27b_webgpu_int4_paged", + "cache_dir": "cache_webgpu_int4_paged", + "no_artifacts": false +} diff --git a/Qwen-Qwen3.8-27B/webgpu/README.md b/Qwen-Qwen3.8-27B/webgpu/README.md new file mode 100644 index 00000000..5356c017 --- /dev/null +++ b/Qwen-Qwen3.8-27B/webgpu/README.md @@ -0,0 +1,28 @@ +# Qwen3.8-27B INT4 (WebGPU) + +This recipe exports [`Qwen/Qwen3.8-27B`](https://huggingface.co/Qwen/Qwen3.8-27B) +as an ONNX Runtime GenAI package for the WebGPU execution provider. It uses: + +- symmetric INT4 target-model weights with block size 32; +- pruned language-model head; +- paged attention with a 256-token block size; and +- memory-selected runtime profile for GPUs with 40 to 60 GiB of device memory. + +## Export + +```bash +olive run --config Qwen-Qwen3.8-27B_webgpu_int4.json +``` + +The exported package is written under `qwen_3.8_27b_webgpu_int4_paged/`. + +## Runtime profiles + +The generated package selects a runtime configuration from total GPU memory: + +| Total GPU memory | KV blocks | Maximum batch size | Scheduled/chunk tokens | +|---|---:|---:|---:| +| 40 to less than 60 GiB | 8 | 1 | 256 | + +The base configuration uses 8 KV blocks, batch size 1, a 64-token schedule, +and chunk size 256. It is the fallback when the profile above is not eligible. diff --git a/Qwen-Qwen3.8-27B/webgpu/info.yml b/Qwen-Qwen3.8-27B/webgpu/info.yml new file mode 100644 index 00000000..290c34a1 --- /dev/null +++ b/Qwen-Qwen3.8-27B/webgpu/info.yml @@ -0,0 +1,11 @@ +keywords: + - olive-ai + - int4 + - paged-attention + - text-generation +arch: qwen3_5_text +recipes: + - name: Qwen3.8-27B_WebGPU_INT4_Paged + file: Qwen-Qwen3.8-27B_webgpu_int4.json + devices: gpu + eps: WebGpuExecutionProvider From 7e7243569aa54b8be2d610fc84a7abb9e838fa6e Mon Sep 17 00:00:00 2001 From: Prathik Rao Date: Thu, 8 Oct 2026 13:30:33 -0700 Subject: [PATCH 2/2] hardcode --- .../webgpu/Qwen-Qwen3.8-27B_webgpu_int4.json | 26 +------------------ 1 file changed, 1 insertion(+), 25 deletions(-) diff --git a/Qwen-Qwen3.8-27B/webgpu/Qwen-Qwen3.8-27B_webgpu_int4.json b/Qwen-Qwen3.8-27B/webgpu/Qwen-Qwen3.8-27B_webgpu_int4.json index 1d334922..95ba3c47 100644 --- a/Qwen-Qwen3.8-27B/webgpu/Qwen-Qwen3.8-27B_webgpu_int4.json +++ b/Qwen-Qwen3.8-27B/webgpu/Qwen-Qwen3.8-27B_webgpu_int4.json @@ -62,7 +62,7 @@ "block_size": 256, "num_blocks": 8, "max_batch_size": 1, - "max_scheduled_tokens": 64 + "max_scheduled_tokens": 256 } }, "search": { @@ -70,30 +70,6 @@ "max_length": 262144 } } - }, - "runtime_profiles": { - "type": "GenAIModelRuntimeProfiles", - "runtime_profiles": [ - { - "id": "40-to-less-than-60-gib", - "eligibility": { - "minimum_total_device_memory_bytes": 42949672960, - "maximum_total_device_memory_bytes": 64424509439 - }, - "overlay": { - "engine": { - "dynamic_batching": { - "num_blocks": 8, - "max_batch_size": 1, - "max_scheduled_tokens": 256 - } - }, - "search": { - "chunk_size": 256 - } - } - } - ] } }, "target": "local_system",