Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
80 changes: 80 additions & 0 deletions Qwen-Qwen3.8-27B/webgpu/Qwen-Qwen3.8-27B_webgpu_int4.json
Original file line number Diff line number Diff line change
@@ -0,0 +1,80 @@
{
"input_model": {
"type": "HfModel",
"model_path": "Qwen/Qwen3.8-27B",
"load_kwargs": {
"torch_dtype": "float16"
}
},
"systems": {
"local_system": {
"type": "LocalSystem",
"accelerators": [
{
"device": "gpu",
"execution_providers": [
"WebGpuExecutionProvider"
]
}
]
}
},
"passes": {
"builder_int4_webgpu": {
"type": "ModelBuilder",
"precision": "int4",
"builder_config_version": 2,
"extra_options": {
"prune_lm_head": true,
"hf_token": false,
"exclude_mtp": true
},
"target_options": {
"quant_config": {
"io_dtype": "fp16",
"weights": {
"type": "int4",
"block_size": 32,
"method": "default",
"symmetric": true,
"op_types": [
"MatMul"
]
},
"format": {
"use_qdq": false,
"matmulnbits_weights_prepacked": 1
}
},
"attention": {
"implementation": "paged",
"paged": {
"block_size": 256
}
},
"optimizations": {
"fuse_mlp_gate_up": true
}
},
"runtime_config": {
"engine": {
"dynamic_batching": {
"block_size": 256,
"num_blocks": 8,
"max_batch_size": 1,
"max_scheduled_tokens": 256
}
},
"search": {
"chunk_size": 256,
"max_length": 262144
}
}
}
},
"target": "local_system",
"log_severity_level": 0,
"output_dir": "qwen_3.8_27b_webgpu_int4_paged",
"cache_dir": "cache_webgpu_int4_paged",
"no_artifacts": false
}
28 changes: 28 additions & 0 deletions Qwen-Qwen3.8-27B/webgpu/README.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,28 @@
# Qwen3.8-27B INT4 (WebGPU)

This recipe exports [`Qwen/Qwen3.8-27B`](https://huggingface.co/Qwen/Qwen3.8-27B)
as an ONNX Runtime GenAI package for the WebGPU execution provider. It uses:

- symmetric INT4 target-model weights with block size 32;
- pruned language-model head;
- paged attention with a 256-token block size; and
- memory-selected runtime profile for GPUs with 40 to 60 GiB of device memory.

## Export

```bash
olive run --config Qwen-Qwen3.8-27B_webgpu_int4.json
```

The exported package is written under `qwen_3.8_27b_webgpu_int4_paged/`.

## Runtime profiles

The generated package selects a runtime configuration from total GPU memory:

| Total GPU memory | KV blocks | Maximum batch size | Scheduled/chunk tokens |
|---|---:|---:|---:|
| 40 to less than 60 GiB | 8 | 1 | 256 |

The base configuration uses 8 KV blocks, batch size 1, a 64-token schedule,
and chunk size 256. It is the fallback when the profile above is not eligible.
11 changes: 11 additions & 0 deletions Qwen-Qwen3.8-27B/webgpu/info.yml
Original file line number Diff line number Diff line change
@@ -0,0 +1,11 @@
keywords:
- olive-ai
- int4
- paged-attention
- text-generation
arch: qwen3_5_text
recipes:
- name: Qwen3.8-27B_WebGPU_INT4_Paged
file: Qwen-Qwen3.8-27B_webgpu_int4.json
devices: gpu
eps: WebGpuExecutionProvider
Loading