diff --git a/_static/models.json b/_static/models.json index 9fa6b845c..b6157a6b1 100644 --- a/_static/models.json +++ b/_static/models.json @@ -1,7 +1,7 @@ [ { "model": { - "id": "openai/gpt-oss-120b", + "id": "zai-org/GLM-5.3-Flash", "owned_by": "Admin", "permissions": [], "object": "model", @@ -11,15 +11,15 @@ "messages" ] }, - "name": "openai/gpt-oss-120b", - "description": "OpenAIs gpt-oss-120B model. Context length 131072 tokens (on demand)", - "prompt_cost": 1, - "completion_cost": 1, - "cached_token_cost": 0.5 + "name": "zai-org/GLM-5.3-Flash", + "description": "GLM 5.3 Flash Model on triton. Context length 1048576 tokens ( Always On)", + "prompt_cost": 0.1, + "completion_cost": 0.5, + "cached_token_cost": 0.01 }, { "model": { - "id": "google/codegemma-7b-it", + "id": "Qwen/Qwen3.8-27B-FP8", "owned_by": "Admin", "permissions": [], "object": "model", @@ -29,15 +29,15 @@ "messages" ] }, - "name": "google/codegemma-7b-it", - "description": "Google codegemma 7B it model. Context length 64000 tokens (on demand)", - "prompt_cost": 0.1, + "name": "Qwen/Qwen3.8-27B-FP8", + "description": "Qwen3.8 27B model with fp8 quantization. Context length 262144 tokens ( Always On )", + "prompt_cost": 0.01, "completion_cost": 0.1, "cached_token_cost": 0.05 }, { "model": { - "id": "Qwen/Qwen3.8-27B-FP8", + "id": "RedHatAI/gemma-4-31B-it-FP8-Dynamic", "owned_by": "Admin", "permissions": [], "object": "model", @@ -47,15 +47,15 @@ "messages" ] }, - "name": "Qwen/Qwen3.8-27B-FP8", - "description": "Qwen3.8 27B model with fp8 quantization. Context length 262144 tokens ( Always On )", - "prompt_cost": 0.1, + "name": "RedHatAI/gemma-4-31B-it-FP8-Dynamic", + "description": "Google Gemma 4 31B IT model Q8 Quantized RedHatAI/gemma-4-31B-it-FP8-Dynamic. Context length 32768 tokens (Always On)", + "prompt_cost": 0.01, "completion_cost": 0.1, "cached_token_cost": 0.05 }, { "model": { - "id": "google/gemma-4-E4B-it", + "id": "google/codegemma-7b-it", "owned_by": "Admin", "permissions": [], "object": "model", @@ -65,9 +65,27 @@ "messages" ] }, - "name": "google/gemma-4-E4B-it", - "description": "Google Gemma 4 E4B IT model. Context length 128000 tokens (on demand)", - "prompt_cost": 0.1, + "name": "google/codegemma-7b-it", + "description": "Google codegemma 7B it model. Context length 64000 tokens (on demand)", + "prompt_cost": 0.01, + "completion_cost": 0.1, + "cached_token_cost": 0.05 + }, + { + "model": { + "id": "Qwen/Qwen3-VL-30B-A3B-Thinking-FP8", + "owned_by": "Admin", + "permissions": [], + "object": "model", + "type": [ + "chat", + "responses", + "messages" + ] + }, + "name": "Qwen/Qwen3-VL-30B-A3B-Thinking-FP8", + "description": "Qwen3 VL 30B thinking model with FP8 quantization. Context length 31228 tokens (on demand)", + "prompt_cost": 0.01, "completion_cost": 0.1, "cached_token_cost": 0.05 }, @@ -85,13 +103,13 @@ }, "name": "Qwen/Qwen3-30B-A3B-Instruct-2507-FP8", "description": "Qwen3 30B instruct model with fp8 quantization. Context length 15664 tokens ( On demand )", - "prompt_cost": 0.1, + "prompt_cost": 0.01, "completion_cost": 0.1, "cached_token_cost": 0.05 }, { "model": { - "id": "Qwen/Qwen3-VL-30B-A3B-Instruct-FP8", + "id": "google/gemma-4-E4B-it", "owned_by": "Admin", "permissions": [], "object": "model", @@ -101,15 +119,15 @@ "messages" ] }, - "name": "Qwen/Qwen3-VL-30B-A3B-Instruct-FP8", - "description": "Qwen3 VL 30B instruct model with FP8 quantization. Context length 31228 tokens (on demand)", - "prompt_cost": 0.1, + "name": "google/gemma-4-E4B-it", + "description": "Google Gemma 4 E4B IT model. Context length 128000 tokens (on demand)", + "prompt_cost": 0.01, "completion_cost": 0.1, "cached_token_cost": 0.05 }, { "model": { - "id": "RedHatAI/gemma-4-31B-it-FP8-Dynamic", + "id": "Qwen/Qwen3-VL-30B-A3B-Instruct-FP8", "owned_by": "Admin", "permissions": [], "object": "model", @@ -119,15 +137,15 @@ "messages" ] }, - "name": "RedHatAI/gemma-4-31B-it-FP8-Dynamic", - "description": "Google Gemma 4 31B IT model Q8 Quantized RedHatAI/gemma-4-31B-it-FP8-Dynamic. Context length 32768 tokens (Always On)", - "prompt_cost": 0.1, + "name": "Qwen/Qwen3-VL-30B-A3B-Instruct-FP8", + "description": "Qwen3 VL 30B instruct model with FP8 quantization. Context length 31228 tokens (on demand)", + "prompt_cost": 0.01, "completion_cost": 0.1, "cached_token_cost": 0.05 }, { "model": { - "id": "Qwen/Qwen3-Coder-30B-A3B-Instruct-FP8", + "id": "openai/gpt-oss-120b", "owned_by": "Admin", "permissions": [], "object": "model", @@ -137,15 +155,15 @@ "messages" ] }, - "name": "Qwen/Qwen3-Coder-30B-A3B-Instruct-FP8", - "description": "Qwen3 30B coder instruct model with fp8 quantization. Context length 32768 tokens (on demand)", + "name": "openai/gpt-oss-120b", + "description": "OpenAIs gpt-oss-120B model. Context length 131072 tokens (on demand)", "prompt_cost": 0.1, - "completion_cost": 0.1, - "cached_token_cost": 0.05 + "completion_cost": 0.5, + "cached_token_cost": 0.5 }, { "model": { - "id": "Qwen/Qwen3-VL-30B-A3B-Thinking-FP8", + "id": "Qwen/Qwen3-Coder-30B-A3B-Instruct-FP8", "owned_by": "Admin", "permissions": [], "object": "model", @@ -155,10 +173,10 @@ "messages" ] }, - "name": "Qwen/Qwen3-VL-30B-A3B-Thinking-FP8", - "description": "Qwen3 VL 30B thinking model with FP8 quantization. Context length 31228 tokens (on demand)", - "prompt_cost": 0.1, + "name": "Qwen/Qwen3-Coder-30B-A3B-Instruct-FP8", + "description": "Qwen3 30B coder instruct model with fp8 quantization. Context length 32768 tokens (on demand)", + "prompt_cost": 0.01, "completion_cost": 0.1, "cached_token_cost": 0.05 } -] \ No newline at end of file +] diff --git a/aalto/llm-web-apis.rst b/aalto/llm-web-apis.rst index a961672b4..251d1b5c9 100644 --- a/aalto/llm-web-apis.rst +++ b/aalto/llm-web-apis.rst @@ -69,7 +69,7 @@ On-demand models are labelled as such in the model overview on the gateway front Given the limited resources (at time of writing the whole supporting infrastructure has 8 L40s cards) not all models can run at the same time and it is entirely possible that a model will not spin up after a request because resources are in use. -If you are unsure which one to pick as a start, pick "RedHatAI/gemma-4-31B-it-FP8-Dynamic". +If you are unsure which one to pick as a start, pick "Qwen/Qwen3.8-27B-FP8". Python quickstart ----------------- @@ -98,7 +98,7 @@ Replace ``YOURKEYGOESHERE`` with the key you created above. base_url="https://llm-gateway.k8s.aalto.fi/api/v1" ) completion = client.chat.completions.create( - model="RedHatAI/gemma-4-31B-it-FP8-Dynamic", + model="Qwen/Qwen3.8-27B-FP8", messages=[ {"role": "system", "content": "Helpful assistant that writes python for research."}, {"role": "user", "content": "I need a python script to load a csv."} @@ -170,6 +170,10 @@ about whether the endpoint is the right tool for what you have in mind. - No - Deploying tools to end users (students, staff, public) makes you an AI system provider under the EU AI Act, which comes with obligations we can't support here. Just get in touch so we can chat about your idea. + * - Are there legal restrictions on the models that you provide? + - Each model comes with its own license. We try to host only models with permissive licenses (MIT, Apache 2.0) but sometimes licenses + change in time and restrictions that did not exist might suddenly appear. If you are unsure, let's have a chat during our daily zoom + and, depending on the case, we can then escalate to our legal experts. If you're not sure whether your use case fits, just ask: drop a message in the ``#llms`` stream on :ref:`chat` or email ``rse@aalto.fi`` and we'll help you figure it out. If you are making any type of production system that is not for research, it might have extra legal requirements. Using this platform does not give you any compliance towards these rules.