Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
92 changes: 55 additions & 37 deletions _static/models.json
Original file line number Diff line number Diff line change
@@ -1,7 +1,7 @@
[
{
"model": {
"id": "openai/gpt-oss-120b",
"id": "zai-org/GLM-5.3-Flash",
"owned_by": "Admin",
"permissions": [],
"object": "model",
Expand All @@ -11,15 +11,15 @@
"messages"
]
},
"name": "openai/gpt-oss-120b",
"description": "OpenAIs gpt-oss-120B model. Context length 131072 tokens (on demand)",
"prompt_cost": 1,
"completion_cost": 1,
"cached_token_cost": 0.5
"name": "zai-org/GLM-5.3-Flash",
"description": "GLM 5.3 Flash Model on triton. Context length 1048576 tokens ( Always On)",
"prompt_cost": 0.1,
"completion_cost": 0.5,
"cached_token_cost": 0.01
},
{
"model": {
"id": "google/codegemma-7b-it",
"id": "Qwen/Qwen3.8-27B-FP8",
"owned_by": "Admin",
"permissions": [],
"object": "model",
Expand All @@ -29,15 +29,15 @@
"messages"
]
},
"name": "google/codegemma-7b-it",
"description": "Google codegemma 7B it model. Context length 64000 tokens (on demand)",
"prompt_cost": 0.1,
"name": "Qwen/Qwen3.8-27B-FP8",
"description": "Qwen3.8 27B model with fp8 quantization. Context length 262144 tokens ( Always On )",
"prompt_cost": 0.01,
"completion_cost": 0.1,
"cached_token_cost": 0.05
},
{
"model": {
"id": "Qwen/Qwen3.8-27B-FP8",
"id": "RedHatAI/gemma-4-31B-it-FP8-Dynamic",
"owned_by": "Admin",
"permissions": [],
"object": "model",
Expand All @@ -47,15 +47,15 @@
"messages"
]
},
"name": "Qwen/Qwen3.8-27B-FP8",
"description": "Qwen3.8 27B model with fp8 quantization. Context length 262144 tokens ( Always On )",
"prompt_cost": 0.1,
"name": "RedHatAI/gemma-4-31B-it-FP8-Dynamic",
"description": "Google Gemma 4 31B IT model Q8 Quantized RedHatAI/gemma-4-31B-it-FP8-Dynamic. Context length 32768 tokens (Always On)",
"prompt_cost": 0.01,
"completion_cost": 0.1,
"cached_token_cost": 0.05
},
{
"model": {
"id": "google/gemma-4-E4B-it",
"id": "google/codegemma-7b-it",
"owned_by": "Admin",
"permissions": [],
"object": "model",
Expand All @@ -65,9 +65,27 @@
"messages"
]
},
"name": "google/gemma-4-E4B-it",
"description": "Google Gemma 4 E4B IT model. Context length 128000 tokens (on demand)",
"prompt_cost": 0.1,
"name": "google/codegemma-7b-it",
"description": "Google codegemma 7B it model. Context length 64000 tokens (on demand)",
"prompt_cost": 0.01,
"completion_cost": 0.1,
"cached_token_cost": 0.05
},
{
"model": {
"id": "Qwen/Qwen3-VL-30B-A3B-Thinking-FP8",
"owned_by": "Admin",
"permissions": [],
"object": "model",
"type": [
"chat",
"responses",
"messages"
]
},
"name": "Qwen/Qwen3-VL-30B-A3B-Thinking-FP8",
"description": "Qwen3 VL 30B thinking model with FP8 quantization. Context length 31228 tokens (on demand)",
"prompt_cost": 0.01,
"completion_cost": 0.1,
"cached_token_cost": 0.05
},
Expand All @@ -85,13 +103,13 @@
},
"name": "Qwen/Qwen3-30B-A3B-Instruct-2507-FP8",
"description": "Qwen3 30B instruct model with fp8 quantization. Context length 15664 tokens ( On demand )",
"prompt_cost": 0.1,
"prompt_cost": 0.01,
"completion_cost": 0.1,
"cached_token_cost": 0.05
},
{
"model": {
"id": "Qwen/Qwen3-VL-30B-A3B-Instruct-FP8",
"id": "google/gemma-4-E4B-it",
"owned_by": "Admin",
"permissions": [],
"object": "model",
Expand All @@ -101,15 +119,15 @@
"messages"
]
},
"name": "Qwen/Qwen3-VL-30B-A3B-Instruct-FP8",
"description": "Qwen3 VL 30B instruct model with FP8 quantization. Context length 31228 tokens (on demand)",
"prompt_cost": 0.1,
"name": "google/gemma-4-E4B-it",
"description": "Google Gemma 4 E4B IT model. Context length 128000 tokens (on demand)",
"prompt_cost": 0.01,
"completion_cost": 0.1,
"cached_token_cost": 0.05
},
{
"model": {
"id": "RedHatAI/gemma-4-31B-it-FP8-Dynamic",
"id": "Qwen/Qwen3-VL-30B-A3B-Instruct-FP8",
"owned_by": "Admin",
"permissions": [],
"object": "model",
Expand All @@ -119,15 +137,15 @@
"messages"
]
},
"name": "RedHatAI/gemma-4-31B-it-FP8-Dynamic",
"description": "Google Gemma 4 31B IT model Q8 Quantized RedHatAI/gemma-4-31B-it-FP8-Dynamic. Context length 32768 tokens (Always On)",
"prompt_cost": 0.1,
"name": "Qwen/Qwen3-VL-30B-A3B-Instruct-FP8",
"description": "Qwen3 VL 30B instruct model with FP8 quantization. Context length 31228 tokens (on demand)",
"prompt_cost": 0.01,
"completion_cost": 0.1,
"cached_token_cost": 0.05
},
{
"model": {
"id": "Qwen/Qwen3-Coder-30B-A3B-Instruct-FP8",
"id": "openai/gpt-oss-120b",
"owned_by": "Admin",
"permissions": [],
"object": "model",
Expand All @@ -137,15 +155,15 @@
"messages"
]
},
"name": "Qwen/Qwen3-Coder-30B-A3B-Instruct-FP8",
"description": "Qwen3 30B coder instruct model with fp8 quantization. Context length 32768 tokens (on demand)",
"name": "openai/gpt-oss-120b",
"description": "OpenAIs gpt-oss-120B model. Context length 131072 tokens (on demand)",
"prompt_cost": 0.1,
"completion_cost": 0.1,
"cached_token_cost": 0.05
"completion_cost": 0.5,
"cached_token_cost": 0.5
},
{
"model": {
"id": "Qwen/Qwen3-VL-30B-A3B-Thinking-FP8",
"id": "Qwen/Qwen3-Coder-30B-A3B-Instruct-FP8",
"owned_by": "Admin",
"permissions": [],
"object": "model",
Expand All @@ -155,10 +173,10 @@
"messages"
]
},
"name": "Qwen/Qwen3-VL-30B-A3B-Thinking-FP8",
"description": "Qwen3 VL 30B thinking model with FP8 quantization. Context length 31228 tokens (on demand)",
"prompt_cost": 0.1,
"name": "Qwen/Qwen3-Coder-30B-A3B-Instruct-FP8",
"description": "Qwen3 30B coder instruct model with fp8 quantization. Context length 32768 tokens (on demand)",
"prompt_cost": 0.01,
"completion_cost": 0.1,
"cached_token_cost": 0.05
}
]
]
8 changes: 6 additions & 2 deletions aalto/llm-web-apis.rst
Original file line number Diff line number Diff line change
Expand Up @@ -69,7 +69,7 @@ On-demand models are labelled as such in the model overview on the gateway front
Given the limited resources (at time of writing the whole supporting infrastructure has 8 L40s cards)
not all models can run at the same time and it is entirely possible that a model
will not spin up after a request because resources are in use.
If you are unsure which one to pick as a start, pick "RedHatAI/gemma-4-31B-it-FP8-Dynamic".
If you are unsure which one to pick as a start, pick "Qwen/Qwen3.8-27B-FP8".

Python quickstart
-----------------
Expand Down Expand Up @@ -98,7 +98,7 @@ Replace ``YOURKEYGOESHERE`` with the key you created above.
base_url="https://llm-gateway.k8s.aalto.fi/api/v1"
)
completion = client.chat.completions.create(
model="RedHatAI/gemma-4-31B-it-FP8-Dynamic",
model="Qwen/Qwen3.8-27B-FP8",
messages=[
{"role": "system", "content": "Helpful assistant that writes python for research."},
{"role": "user", "content": "I need a python script to load a csv."}
Expand Down Expand Up @@ -170,6 +170,10 @@ about whether the endpoint is the right tool for what you have in mind.
- No
- Deploying tools to end users (students, staff, public) makes you an AI system
provider under the EU AI Act, which comes with obligations we can't support here. Just get in touch so we can chat about your idea.
* - Are there legal restrictions on the models that you provide?
- Each model comes with its own license. We try to host only models with permissive licenses (MIT, Apache 2.0) but sometimes licenses
change in time and restrictions that did not exist might suddenly appear. If you are unsure, let's have a chat during our daily zoom
and, depending on the case, we can then escalate to our legal experts.

If you're not sure whether your use case fits, just ask: drop a message in the
``#llms`` stream on :ref:`chat` or email ``rse@aalto.fi`` and we'll help you figure it out. If you are making any type of production system that is not for research, it might have extra legal requirements. Using this platform does not give you any compliance towards these rules.
Expand Down
Loading