diff --git a/Qwen-Qwen3.8-27B/webgpu/Qwen-Qwen3.8-27B_webgpu_int4.json b/Qwen-Qwen3.8-27B/webgpu/Qwen-Qwen3.8-27B_webgpu_int4.json new file mode 100644 index 00000000..95ba3c47 --- /dev/null +++ b/Qwen-Qwen3.8-27B/webgpu/Qwen-Qwen3.8-27B_webgpu_int4.json @@ -0,0 +1,80 @@ +{ + "input_model": { + "type": "HfModel", + "model_path": "Qwen/Qwen3.8-27B", + "load_kwargs": { + "torch_dtype": "float16" + } + }, + "systems": { + "local_system": { + "type": "LocalSystem", + "accelerators": [ + { + "device": "gpu", + "execution_providers": [ + "WebGpuExecutionProvider" + ] + } + ] + } + }, + "passes": { + "builder_int4_webgpu": { + "type": "ModelBuilder", + "precision": "int4", + "builder_config_version": 2, + "extra_options": { + "prune_lm_head": true, + "hf_token": false, + "exclude_mtp": true + }, + "target_options": { + "quant_config": { + "io_dtype": "fp16", + "weights": { + "type": "int4", + "block_size": 32, + "method": "default", + "symmetric": true, + "op_types": [ + "MatMul" + ] + }, + "format": { + "use_qdq": false, + "matmulnbits_weights_prepacked": 1 + } + }, + "attention": { + "implementation": "paged", + "paged": { + "block_size": 256 + } + }, + "optimizations": { + "fuse_mlp_gate_up": true + } + }, + "runtime_config": { + "engine": { + "dynamic_batching": { + "block_size": 256, + "num_blocks": 8, + "max_batch_size": 1, + "max_scheduled_tokens": 256 + } + }, + "search": { + "chunk_size": 256, + "max_length": 262144 + } + } + } + }, + "target": "local_system", + "log_severity_level": 0, + "output_dir": "qwen_3.8_27b_webgpu_int4_paged", + "cache_dir": "cache_webgpu_int4_paged", + "no_artifacts": false +} diff --git a/Qwen-Qwen3.8-27B/webgpu/README.md b/Qwen-Qwen3.8-27B/webgpu/README.md new file mode 100644 index 00000000..5356c017 --- /dev/null +++ b/Qwen-Qwen3.8-27B/webgpu/README.md @@ -0,0 +1,28 @@ +# Qwen3.8-27B INT4 (WebGPU) + +This recipe exports [`Qwen/Qwen3.8-27B`](https://huggingface.co/Qwen/Qwen3.8-27B) +as an ONNX Runtime GenAI package for the WebGPU execution provider. It uses: + +- symmetric INT4 target-model weights with block size 32; +- pruned language-model head; +- paged attention with a 256-token block size; and +- memory-selected runtime profile for GPUs with 40 to 60 GiB of device memory. + +## Export + +```bash +olive run --config Qwen-Qwen3.8-27B_webgpu_int4.json +``` + +The exported package is written under `qwen_3.8_27b_webgpu_int4_paged/`. + +## Runtime profiles + +The generated package selects a runtime configuration from total GPU memory: + +| Total GPU memory | KV blocks | Maximum batch size | Scheduled/chunk tokens | +|---|---:|---:|---:| +| 40 to less than 60 GiB | 8 | 1 | 256 | + +The base configuration uses 8 KV blocks, batch size 1, a 64-token schedule, +and chunk size 256. It is the fallback when the profile above is not eligible. diff --git a/Qwen-Qwen3.8-27B/webgpu/info.yml b/Qwen-Qwen3.8-27B/webgpu/info.yml new file mode 100644 index 00000000..290c34a1 --- /dev/null +++ b/Qwen-Qwen3.8-27B/webgpu/info.yml @@ -0,0 +1,11 @@ +keywords: + - olive-ai + - int4 + - paged-attention + - text-generation +arch: qwen3_5_text +recipes: + - name: Qwen3.8-27B_WebGPU_INT4_Paged + file: Qwen-Qwen3.8-27B_webgpu_int4.json + devices: gpu + eps: WebGpuExecutionProvider