diff --git a/models/alibaba/qwen-audio-3.0-tts-plus.toml b/models/alibaba/qwen-audio-3.0-tts-plus.toml new file mode 100644 index 00000000000..aae03164ee3 --- /dev/null +++ b/models/alibaba/qwen-audio-3.0-tts-plus.toml @@ -0,0 +1,17 @@ +name = "Qwen-Audio-3.0-TTS-Plus" +description = "Qwen-Audio-3.0-TTS-Plus is Alibaba's higher-quality text-to-speech model, generating spoken audio from text via the DashScope Speech Synthesizer API." +release_date = "2026-08-05" +last_updated = "2026-08-05" +attachment = false +reasoning = false +temperature = false +tool_call = false +open_weights = false + +[limit] +context = 0 +output = 0 + +[modalities] +input = ["text"] +output = ["audio"] diff --git a/models/alibaba/qwen-image-2.0-pro.toml b/models/alibaba/qwen-image-2.0-pro.toml new file mode 100644 index 00000000000..285a450761c --- /dev/null +++ b/models/alibaba/qwen-image-2.0-pro.toml @@ -0,0 +1,18 @@ +# https://www.alibabacloud.com/help/en/model-studio/qwen-image-api +name = "Qwen-Image-2.0-Pro" +description = "Qwen-Image-2.0-Pro is an image generation model launched by Qwen AI." +release_date = "2026-03-03" +last_updated = "2026-03-03" +attachment = true +reasoning = false +temperature = false +tool_call = false +open_weights = false + +[limit] +context = 1_300 +output = 0 + +[modalities] +input = ["text", "image"] +output = ["image"] diff --git a/models/alibaba/qwen-image-2.0.toml b/models/alibaba/qwen-image-2.0.toml new file mode 100644 index 00000000000..9d89f4d8957 --- /dev/null +++ b/models/alibaba/qwen-image-2.0.toml @@ -0,0 +1,18 @@ +# https://www.alibabacloud.com/help/en/model-studio/qwen-image-api +name = "Qwen-Image-2.0" +description = "Qwen-Image-2.0 is an image generation model launched by Qwen AI." +release_date = "2026-03-03" +last_updated = "2026-03-03" +attachment = true +reasoning = false +temperature = false +tool_call = false +open_weights = false + +[limit] +context = 1_300 +output = 0 + +[modalities] +input = ["text", "image"] +output = ["image"] diff --git a/models/alibaba/qwen-image-3.0-pro.toml b/models/alibaba/qwen-image-3.0-pro.toml new file mode 100644 index 00000000000..156aaa994b3 --- /dev/null +++ b/models/alibaba/qwen-image-3.0-pro.toml @@ -0,0 +1,18 @@ +# https://www.alibabacloud.com/help/en/model-studio/qwen-image-generation-and-editing-api-reference +name = "Qwen-Image-3.0-Pro" +description = "Rich content: Supports input of up to 4.5k tokens and dense information layout with images-within-images, enabling complex layouts like newspapers, storyboards, menus, and exam papers to be generated in a single pass. Authentic detail: Supports precise rendering of text as small as 10px, and vividly reproduces fine details such as micro-expressions, pores, and individual strands of hair—approaching the quality of real photography. Deep knowledge: Supports native rendering of 12 languages and 20+ fonts, realistic simulation of mainstream interfaces such as web pages, games, and live streams, fully incorporating external knowledge. Qwen-Image-3.0-Pro isn't just pursuing \"good looks\"—it's pursuing \"usefulness\", making image generation a truly deployable productivity tool." +release_date = "2026-08-05" +last_updated = "2026-08-05" +attachment = true +reasoning = false +temperature = false +tool_call = false +open_weights = false + +[limit] +context = 4_500 +output = 0 + +[modalities] +input = ["text", "image"] +output = ["image"] diff --git a/models/alibaba/qwen-image-3.0.toml b/models/alibaba/qwen-image-3.0.toml new file mode 100644 index 00000000000..84ba68f4c6b --- /dev/null +++ b/models/alibaba/qwen-image-3.0.toml @@ -0,0 +1,18 @@ +# https://www.alibabacloud.com/help/en/model-studio/qwen-image-generation-and-editing-api-reference +name = "Qwen-Image-3.0" +description = "Accurate Prompt Understanding: Supports up to 4.5k token inputs, accurately interpreting complex text-and-image prompts and generating dense layouts in one pass. Reliable Text Rendering: Delivers crisp 10px text across 12 languages and 20+ fonts, making infographics and interfaces ready to use. Efficient Batch Production: Scales everyday tasks like posters, web pages, and UI screens with better cost efficiency, keeping creative output sustainable. Qwen-Image-3.0 Standard is built not just for visual quality, but for smooth, reliable daily creation—turning image generation into sustainable content productivity." +release_date = "2026-08-05" +last_updated = "2026-08-05" +attachment = true +reasoning = false +temperature = false +tool_call = false +open_weights = false + +[limit] +context = 4_500 +output = 0 + +[modalities] +input = ["text", "image"] +output = ["image"] diff --git a/models/alibaba/qwen3-14b.toml b/models/alibaba/qwen3-14b.toml new file mode 100644 index 00000000000..fd5ef6377ea --- /dev/null +++ b/models/alibaba/qwen3-14b.toml @@ -0,0 +1,24 @@ +name = "Qwen3 14B" +description = "Qwen instruction model for multilingual chat, reasoning, and tool use" +family = "qwen" +release_date = "2025-04-28" +last_updated = "2025-04-28" +attachment = false +reasoning = true +temperature = true +tool_call = true +knowledge = "2025-04" +open_weights = true +license = "Apache 2.0" + +[limit] +context = 131_072 +output = 8_192 + +[modalities] +input = ["text"] +output = ["text"] + +[[weights]] +label = "Hugging Face" +url = "https://huggingface.co/Qwen/Qwen3-14B" diff --git a/models/alibaba/qwen3-235b-a22b-thinking-2507.toml b/models/alibaba/qwen3-235b-a22b-thinking-2507.toml new file mode 100644 index 00000000000..d6ccc62a6e4 --- /dev/null +++ b/models/alibaba/qwen3-235b-a22b-thinking-2507.toml @@ -0,0 +1,23 @@ +# https://huggingface.co/Qwen/Qwen3-235B-A22B-Thinking-2507 +name = "Qwen3 235B A22B Thinking 2507" +description = "Qwen3-235B-A22B-Thinking-2507 is a high-performance, open-weight Mixture-of-Experts (MoE) language model optimized for complex reasoning tasks. It activates 22B of its 235B parameters per forward pass and natively supports up to 262,144 tokens of context. This \"thinking-only\" variant enhances structured logical reasoning, mathematics, science, and long-form generation, showing strong benchmark performance across AIME, SuperGPQA, LiveCodeBench, and MMLU-Redux. It enforces a special reasoning mode () and is designed for high-token outputs (up to 81,920 tokens) in challenging domains.\n\nThe model is instruction-tuned and excels at step-by-step reasoning, tool use, agentic workflows, and multilingual tasks. This release represents the most capable open-source variant in the Qwen3-235B series, surpassing many closed models in structured reasoning use cases." +release_date = "2025-07-21" +last_updated = "2025-07-21" +attachment = false +reasoning = true +temperature = true +tool_call = true +open_weights = true +license = "Apache 2.0" + +[limit] +context = 262_144 +output = 81_920 + +[modalities] +input = ["text"] +output = ["text"] + +[[weights]] +label = "Hugging Face" +url = "https://huggingface.co/Qwen/Qwen3-235B-A22B-Thinking-2507" diff --git a/models/alibaba/qwen3-asr-flash.toml b/models/alibaba/qwen3-asr-flash.toml new file mode 100644 index 00000000000..266d0372f58 --- /dev/null +++ b/models/alibaba/qwen3-asr-flash.toml @@ -0,0 +1,19 @@ +name = "Qwen3-ASR Flash" +description = "Speech transcription model for accurate audio-to-text and captioning workflows" +family = "qwen" +release_date = "2025-09-08" +last_updated = "2025-09-08" +attachment = false +reasoning = false +temperature = false +tool_call = false +knowledge = "2024-04" +open_weights = false + +[limit] +context = 53_248 +output = 4_096 + +[modalities] +input = ["audio"] +output = ["text"] diff --git a/models/alibaba/qwen3-max-2026-01-23.toml b/models/alibaba/qwen3-max-2026-01-23.toml new file mode 100644 index 00000000000..8f63001edd2 --- /dev/null +++ b/models/alibaba/qwen3-max-2026-01-23.toml @@ -0,0 +1,25 @@ +# https://docs.modelstudio.console.alibabacloud.com/en/model-studio/model-qwen3-max +# https://www.alibabacloud.com/blog/pushing-qwen3-max-thinking-beyond-its-limits_602834 +name = "Qwen3 Max 2026-01-23" +description = "Qwen3 Max snapshot integrating thinking and non-thinking modes for complex reasoning, instruction following, multilingual tasks, retrieval, and tool use" +family = "qwen" +release_date = "2026-01-23" +last_updated = "2026-01-23" +attachment = false +reasoning = true +temperature = true +tool_call = true +open_weights = false + +[limit] +context = 262_144 +output = 65_536 + +[modalities] +input = ["text"] +output = ["text"] + +[[links]] +label = "Alibaba Cloud Model Studio" +url = "https://docs.modelstudio.console.alibabacloud.com/en/model-studio/model-qwen3-max" +type = "docs" diff --git a/models/alibaba/qwen3-rerank.toml b/models/alibaba/qwen3-rerank.toml new file mode 100644 index 00000000000..3c1da5d60ee --- /dev/null +++ b/models/alibaba/qwen3-rerank.toml @@ -0,0 +1,17 @@ +name = "Qwen3 Rerank" +description = "Qwen3 Rerank is a new proprietary model of the Qwen model family. Specifically designed for reranking tasks, built on the Qwen3 foundation model. Leveraging Qwen3’s robust multilingual text understanding capabilities, the series achieves state-of-the-art performance across multiple benchmarks for text embedding and reranking tasks." +release_date = "2026-01-08" +last_updated = "2026-01-08" +attachment = false +reasoning = false +temperature = false +tool_call = false +open_weights = false + +[limit] +context = 8_192 +output = 8_192 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/models/alibaba/qwen3-vl-embedding.toml b/models/alibaba/qwen3-vl-embedding.toml new file mode 100644 index 00000000000..a3048f96de5 --- /dev/null +++ b/models/alibaba/qwen3-vl-embedding.toml @@ -0,0 +1,17 @@ +name = "Qwen3 VL Embedding" +description = "The Qwen3-VL-Embedding model series are the latest additions to the Qwen family, built upon the recently open-sourced and powerful Qwen3-VL foundation model. Specifically designed for multimodal information retrieval and cross-modal understanding, this suite accepts diverse inputs including text, images, screenshots, and videos, as well as inputs containing a mixture of these modalities." +release_date = "2026-01-08" +last_updated = "2026-01-08" +attachment = true +reasoning = false +temperature = false +tool_call = false +open_weights = false + +[limit] +context = 8_192 +output = 8_192 + +[modalities] +input = ["text", "video", "image"] +output = ["text"] diff --git a/models/alibaba/qwen3-vl-rerank.toml b/models/alibaba/qwen3-vl-rerank.toml new file mode 100644 index 00000000000..c15ef8c3ab8 --- /dev/null +++ b/models/alibaba/qwen3-vl-rerank.toml @@ -0,0 +1,18 @@ +# https://docs.modelstudio.console.alibabacloud.com/en/model-studio/qwen3-vl-rerank +name = "Qwen3 VL Rerank" +description = "The Qwen3-VL-Rerank model series are the latest additions to the Qwen family, built upon the recently open-sourced and powerful Qwen3-VL foundation model. Specifically designed for multimodal information retrieval and cross-modal understanding, this suite accepts diverse inputs including text, images, screenshots, and videos, as well as inputs containing a mixture of these modalities." +release_date = "2026-01-08" +last_updated = "2026-01-08" +attachment = true +reasoning = false +temperature = false +tool_call = false +open_weights = false + +[limit] +context = 120_000 +output = 0 + +[modalities] +input = ["text", "image", "video"] +output = ["text"] diff --git a/providers/zenmux/models/qwen/qwen-audio-3.0-tts-plus.toml b/providers/zenmux/models/qwen/qwen-audio-3.0-tts-plus.toml new file mode 100644 index 00000000000..d777ab2af16 --- /dev/null +++ b/providers/zenmux/models/qwen/qwen-audio-3.0-tts-plus.toml @@ -0,0 +1,3 @@ +# ZenMux pricing is non-token-denominated for this route and is intentionally omitted. +# https://zenmux.ai/docs/api/openai/openai-list-models.html +base_model = "alibaba/qwen-audio-3.0-tts-plus" diff --git a/providers/zenmux/models/qwen/qwen-image-2.0-pro.toml b/providers/zenmux/models/qwen/qwen-image-2.0-pro.toml new file mode 100644 index 00000000000..b88491e9195 --- /dev/null +++ b/providers/zenmux/models/qwen/qwen-image-2.0-pro.toml @@ -0,0 +1,7 @@ +# ZenMux pricing is non-token-denominated for this route and is intentionally omitted. +# Host prompt limit: the 2026-09-19 ZenMux page reports context_length=1024. +# https://zenmux.ai/docs/api/openai/openai-list-models.html +base_model = "alibaba/qwen-image-2.0-pro" + +[limit] +context = 1_024 diff --git a/providers/zenmux/models/qwen/qwen-image-2.0.toml b/providers/zenmux/models/qwen/qwen-image-2.0.toml new file mode 100644 index 00000000000..5e791df4085 --- /dev/null +++ b/providers/zenmux/models/qwen/qwen-image-2.0.toml @@ -0,0 +1,7 @@ +# ZenMux pricing is non-token-denominated for this route and is intentionally omitted. +# Host prompt limit: the 2026-09-19 ZenMux page reports context_length=1024. +# https://zenmux.ai/docs/api/openai/openai-list-models.html +base_model = "alibaba/qwen-image-2.0" + +[limit] +context = 1_024 diff --git a/providers/zenmux/models/qwen/qwen-image-3.0-pro.toml b/providers/zenmux/models/qwen/qwen-image-3.0-pro.toml new file mode 100644 index 00000000000..ea890d3d706 --- /dev/null +++ b/providers/zenmux/models/qwen/qwen-image-3.0-pro.toml @@ -0,0 +1,7 @@ +# ZenMux pricing is non-token-denominated for this route and is intentionally omitted. +# Host prompt limit: the 2026-09-19 ZenMux page reports context_length=1024. +# https://zenmux.ai/docs/api/openai/openai-list-models.html +base_model = "alibaba/qwen-image-3.0-pro" + +[limit] +context = 1_024 diff --git a/providers/zenmux/models/qwen/qwen-image-3.0.toml b/providers/zenmux/models/qwen/qwen-image-3.0.toml new file mode 100644 index 00000000000..030aa99f9bb --- /dev/null +++ b/providers/zenmux/models/qwen/qwen-image-3.0.toml @@ -0,0 +1,7 @@ +# ZenMux pricing is non-token-denominated for this route and is intentionally omitted. +# Host prompt limit: the 2026-09-19 ZenMux page reports context_length=1024. +# https://zenmux.ai/docs/api/openai/openai-list-models.html +base_model = "alibaba/qwen-image-3.0" + +[limit] +context = 1_024 diff --git a/providers/zenmux/models/qwen/qwen3-14b.toml b/providers/zenmux/models/qwen/qwen3-14b.toml new file mode 100644 index 00000000000..7691a66fcb7 --- /dev/null +++ b/providers/zenmux/models/qwen/qwen3-14b.toml @@ -0,0 +1,21 @@ +# Toggle: reasoning.enabled = true|false +# Budget: reasoning.max_tokens (integer reasoning tokens) +# https://zenmux.ai/docs/api/openai/openai-list-models.html +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "alibaba/qwen3-14b" +name = "Qwen3-14B" + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0.14 +output = 1.4 +cache_read = 0.028 + +[limit] +context = 32_000 +output = 32_000 diff --git a/providers/zenmux/models/qwen/qwen3-235b-a22b-2507.toml b/providers/zenmux/models/qwen/qwen3-235b-a22b-2507.toml new file mode 100644 index 00000000000..bbff8c04049 --- /dev/null +++ b/providers/zenmux/models/qwen/qwen3-235b-a22b-2507.toml @@ -0,0 +1,14 @@ +# Host limits: the 2026-09-19 ZenMux page reports context_length=256000 and max_completion_tokens=128000. +# https://zenmux.ai/docs/api/openai/openai-list-models.html +base_model = "alibaba/qwen3-235b-a22b-instruct-2507" +name = "Qwen3 235B A22B Instruct 2507" +structured_output = true + +[cost] +input = 0.28 +output = 1.11 +cache_read = 0.056 + +[limit] +context = 256_000 +output = 128_000 diff --git a/providers/zenmux/models/qwen/qwen3-235b-a22b-thinking-2507.toml b/providers/zenmux/models/qwen/qwen3-235b-a22b-thinking-2507.toml new file mode 100644 index 00000000000..17f14e62fbb --- /dev/null +++ b/providers/zenmux/models/qwen/qwen3-235b-a22b-thinking-2507.toml @@ -0,0 +1,18 @@ +# Host limits: the 2026-09-19 ZenMux page reports context_length=256000 and max_completion_tokens=128000. +# https://zenmux.ai/docs/api/openai/openai-list-models.html +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "alibaba/qwen3-235b-a22b-thinking-2507" +structured_output = true +reasoning_options = [] + +[interleaved] +field = "reasoning_content" + +[cost] +input = 0.28 +output = 2.78 +cache_read = 0.056 + +[limit] +context = 256_000 +output = 128_000 diff --git a/providers/zenmux/models/qwen/qwen3-asr-flash.toml b/providers/zenmux/models/qwen/qwen3-asr-flash.toml new file mode 100644 index 00000000000..6acef26a6a5 --- /dev/null +++ b/providers/zenmux/models/qwen/qwen3-asr-flash.toml @@ -0,0 +1,13 @@ +# ZenMux's list flags this route as reasoning-capable, but the transcription API +# exposes neither a reasoning request control nor a reasoning response channel; +# lab and first-party metadata both classify Qwen3 ASR Flash as non-reasoning. +# Host limits: page max_context=1000000 and max_completion_tokens=65536; +# OpenAI list independently reports context_length=1000000. +# ZenMux pricing is non-token-denominated for this route and is intentionally omitted. +# https://zenmux.ai/docs/api/openai/openai-list-models.html +base_model = "alibaba/qwen3-asr-flash" +name = "Qwen3 ASR Flash" + +[limit] +context = 1_000_000 +output = 65_536 diff --git a/providers/zenmux/models/qwen/qwen3-coder-plus.toml b/providers/zenmux/models/qwen/qwen3-coder-plus.toml index 2b327c36d7b..86a58aadea9 100644 --- a/providers/zenmux/models/qwen/qwen3-coder-plus.toml +++ b/providers/zenmux/models/qwen/qwen3-coder-plus.toml @@ -1,24 +1,34 @@ +# https://zenmux.ai/docs/api/openai/openai-list-models.html +base_model = "alibaba/qwen3-coder-plus" name = "Qwen3-Coder-Plus" -description = "Qwen coding model for software agents, repository edits, and code reasoning" -release_date = "2025-07-23" -last_updated = "2025-07-23" -attachment = false -reasoning = false -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false +structured_output = true [cost] -input = 1.00 -output = 5.00 -cache_read = 0.10 +input = 1 +output = 5 +cache_read = 0.1 cache_write = 1.25 -[limit] -context = 1000_000 -output = 64_000 +[[cost.tiers]] +tier = { type = "context", size = 32_000 } +input = 1.8 +output = 9 +cache_read = 0.18 +cache_write = 2.25 + +[[cost.tiers]] +tier = { type = "context", size = 128_000 } +input = 3 +output = 15 +cache_read = 0.3 +cache_write = 3.75 -[modalities] -input = ["text"] -output = ["text"] +[[cost.tiers]] +tier = { type = "context", size = 256_000 } +input = 6 +output = 60 +cache_read = 0.6 +cache_write = 7.5 + +[limit] +context = 1_000_000 diff --git a/providers/zenmux/models/qwen/qwen3-coder.toml b/providers/zenmux/models/qwen/qwen3-coder.toml new file mode 100644 index 00000000000..661da6b6538 --- /dev/null +++ b/providers/zenmux/models/qwen/qwen3-coder.toml @@ -0,0 +1,14 @@ +# Host limits: the 2026-09-19 ZenMux page reports context_length=256000 and max_completion_tokens=128000. +# https://zenmux.ai/docs/api/openai/openai-list-models.html +base_model = "alibaba/qwen3-coder-480b-a35b-instruct" +name = "Qwen3-Coder" +structured_output = true + +[cost] +input = 1.25 +output = 5.01 +cache_read = 0.25 + +[limit] +context = 256_000 +output = 128_000 diff --git a/providers/zenmux/models/qwen/qwen3-max.toml b/providers/zenmux/models/qwen/qwen3-max.toml index 4687663742c..5ebb5d782a0 100644 --- a/providers/zenmux/models/qwen/qwen3-max.toml +++ b/providers/zenmux/models/qwen/qwen3-max.toml @@ -1,23 +1,38 @@ +# Toggle: reasoning.enabled = true|false +# Budget: reasoning.max_tokens (integer reasoning tokens) +# Host limits: the 2026-09-19 ZenMux page reports context_length=256000 and max_completion_tokens=32000. +# https://zenmux.ai/docs/api/openai/openai-list-models.html +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "alibaba/qwen3-max-2026-01-23" name = "Qwen3-Max-Thinking" -description = "Qwen reasoning model for deliberate problem solving, math, and coding" -release_date = "2026-01-23" -last_updated = "2026-01-23" -attachment = false -reasoning = true -reasoning_options = [] -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false +structured_output = true + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "budget_tokens" [cost] -input = 1.20 -output = 6.00 +input = 1.2 +output = 6 +cache_read = 0.12 +cache_write = 1.5 + +[[cost.tiers]] +tier = { type = "context", size = 32_000 } +input = 2.4 +output = 12 +cache_read = 0.24 +cache_write = 3 + +[[cost.tiers]] +tier = { type = "context", size = 128_000 } +input = 3 +output = 15 +cache_read = 0.3 +cache_write = 3.75 [limit] context = 256_000 -output = 64_000 - -[modalities] -input = ["text"] -output = ["text"] +output = 32_000 diff --git a/providers/zenmux/models/qwen/qwen3-rerank.toml b/providers/zenmux/models/qwen/qwen3-rerank.toml new file mode 100644 index 00000000000..01088e0fecd --- /dev/null +++ b/providers/zenmux/models/qwen/qwen3-rerank.toml @@ -0,0 +1,6 @@ +# https://zenmux.ai/docs/api/openai/openai-list-models.html +base_model = "alibaba/qwen3-rerank" + +[cost] +input = 0.1 +output = 0 diff --git a/providers/zenmux/models/qwen/qwen3-vl-embedding.toml b/providers/zenmux/models/qwen/qwen3-vl-embedding.toml new file mode 100644 index 00000000000..593f756a189 --- /dev/null +++ b/providers/zenmux/models/qwen/qwen3-vl-embedding.toml @@ -0,0 +1,6 @@ +# https://zenmux.ai/docs/api/openai/openai-list-models.html +base_model = "alibaba/qwen3-vl-embedding" + +[cost] +input = 0.1 +output = 0 diff --git a/providers/zenmux/models/qwen/qwen3-vl-plus.toml b/providers/zenmux/models/qwen/qwen3-vl-plus.toml new file mode 100644 index 00000000000..84d1677542c --- /dev/null +++ b/providers/zenmux/models/qwen/qwen3-vl-plus.toml @@ -0,0 +1,38 @@ +# Toggle: reasoning.enabled = true|false +# Budget: reasoning.max_tokens (integer reasoning tokens) +# Host modalities: the 2026-09-19 ZenMux page reports text, image, and video input. +# https://zenmux.ai/docs/api/openai/openai-list-models.html +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "alibaba/qwen3-vl-plus" +name = "Qwen3-VL-Plus" +attachment = true +structured_output = true + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0.2 +output = 1.6 +cache_read = 0.02 +cache_write = 0.25 + +[[cost.tiers]] +tier = { type = "context", size = 32_000 } +input = 0.3 +output = 2.4 +cache_read = 0.03 +cache_write = 0.375 + +[[cost.tiers]] +tier = { type = "context", size = 128_000 } +input = 0.6 +output = 4.8 +cache_read = 0.06 +cache_write = 0.75 + +[modalities] +input = ["text", "image", "video"] diff --git a/providers/zenmux/models/qwen/qwen3-vl-rerank.toml b/providers/zenmux/models/qwen/qwen3-vl-rerank.toml new file mode 100644 index 00000000000..e88dfc4bd9a --- /dev/null +++ b/providers/zenmux/models/qwen/qwen3-vl-rerank.toml @@ -0,0 +1,10 @@ +# Host limit: the 2026-09-19 ZenMux page reports context_length=1000000. +# https://zenmux.ai/docs/api/openai/openai-list-models.html +base_model = "alibaba/qwen3-vl-rerank" + +[cost] +input = 0.1 +output = 0 + +[limit] +context = 1_000_000 diff --git a/providers/zenmux/models/qwen/qwen3.5-flash.toml b/providers/zenmux/models/qwen/qwen3.5-flash.toml index c863eed08b1..7babd4e43d7 100644 --- a/providers/zenmux/models/qwen/qwen3.5-flash.toml +++ b/providers/zenmux/models/qwen/qwen3.5-flash.toml @@ -1,22 +1,26 @@ -name = "Qwen3.5 Flash" -description = "Qwen vision-language model for visual reasoning, documents, and agent tasks" -release_date = "2026-03-20" -last_updated = "2026-03-20" -attachment = true -reasoning = false -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false +# Toggle: reasoning.enabled = true|false +# Budget: reasoning.max_tokens (integer reasoning tokens) +# Host limits: the 2026-09-19 ZenMux page reports context_length=1024000 and max_completion_tokens=1024000. +# https://zenmux.ai/docs/api/openai/openai-list-models.html +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "alibaba/qwen3.5-flash" +name = "Qwen3.5-Flash" + +[interleaved] +field = "reasoning_content" + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "budget_tokens" [cost] -input = 0.10 -output = 0.40 +input = 0.1 +output = 0.4 +cache_read = 0.01 +cache_write = 0.125 [limit] -context = 1_020_000 -output = 1_020_000 - -[modalities] -input = ["text", "image"] -output = ["text"] +context = 1_024_000 +output = 1_024_000 diff --git a/providers/zenmux/models/qwen/qwen3.5-plus.toml b/providers/zenmux/models/qwen/qwen3.5-plus.toml index 0cdf87788f3..09c1e88bbf2 100644 --- a/providers/zenmux/models/qwen/qwen3.5-plus.toml +++ b/providers/zenmux/models/qwen/qwen3.5-plus.toml @@ -1,23 +1,30 @@ -name = "Qwen3.5 Plus" -description = "Qwen vision-language model for visual reasoning, documents, and agent tasks" -release_date = "2026-03-20" -last_updated = "2026-03-20" +# Toggle: reasoning.enabled = true|false +# Budget: reasoning.max_tokens (integer reasoning tokens) +# https://zenmux.ai/docs/api/openai/openai-list-models.html +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "alibaba/qwen3.5-plus" +name = "Qwen3.5-Plus" attachment = true -reasoning = true -reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false +structured_output = true + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "budget_tokens" [cost] -input = 0.80 -output = 4.80 +input = 0.4 +output = 2.4 +cache_read = 0.04 +cache_write = 0.5 + +[[cost.tiers]] +tier = { type = "context", size = 256_000 } +input = 0.5 +output = 3 +cache_read = 0.05 +cache_write = 0.625 [limit] -context = 1_000_000 output = 64_000 - -[modalities] -input = ["text", "image"] -output = ["text"] diff --git a/providers/zenmux/models/qwen/qwen3.6-flash.toml b/providers/zenmux/models/qwen/qwen3.6-flash.toml new file mode 100644 index 00000000000..b18da445fdb --- /dev/null +++ b/providers/zenmux/models/qwen/qwen3.6-flash.toml @@ -0,0 +1,24 @@ +# Toggle: reasoning.enabled = true|false +# Budget: reasoning.max_tokens (integer reasoning tokens) +# https://zenmux.ai/docs/api/openai/openai-list-models.html +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "alibaba/qwen3.6-flash" + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0.25 +output = 1.5 +cache_read = 0.025 +cache_write = 0.3125 + +[[cost.tiers]] +tier = { type = "context", size = 256_000 } +input = 1 +output = 4 +cache_read = 0.1 +cache_write = 1.25 diff --git a/providers/zenmux/models/qwen/qwen3.6-max-preview.toml b/providers/zenmux/models/qwen/qwen3.6-max-preview.toml new file mode 100644 index 00000000000..11ce923426e --- /dev/null +++ b/providers/zenmux/models/qwen/qwen3.6-max-preview.toml @@ -0,0 +1,25 @@ +# Toggle: reasoning.enabled = true|false +# Budget: reasoning.max_tokens (integer reasoning tokens) +# https://zenmux.ai/docs/api/openai/openai-list-models.html +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "alibaba/qwen3.6-max-preview" +structured_output = true + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 1.3 +output = 7.8 +cache_read = 0.13 +cache_write = 1.625 + +[[cost.tiers]] +tier = { type = "context", size = 128_000 } +input = 2 +output = 12 +cache_read = 0.2 +cache_write = 2.5 diff --git a/providers/zenmux/models/qwen/qwen3.6-plus.toml b/providers/zenmux/models/qwen/qwen3.6-plus.toml index 6233134b5ac..525cfb1b421 100644 --- a/providers/zenmux/models/qwen/qwen3.6-plus.toml +++ b/providers/zenmux/models/qwen/qwen3.6-plus.toml @@ -1,31 +1,34 @@ +# Toggle: reasoning.enabled = true|false +# Budget: reasoning.max_tokens (integer reasoning tokens) +# ZenMux's 2026-09-19 page and both live OpenAI-/Anthropic-compatible lists +# explicitly include file input; the catalog represents that host modality as PDF. +# https://zenmux.ai/docs/api/openai/openai-list-models.html +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "alibaba/qwen3.6-plus" name = "Qwen3.6-Plus" -description = "Qwen instruction model for multilingual chat, reasoning, and tool use" -release_date = "2026-03-30" -last_updated = "2026-03-30" -attachment = false -reasoning = true -reasoning_options = [{ type = "toggle" }] -temperature = true -tool_call = true -open_weights = false +structured_output = true + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "budget_tokens" [cost] -input = 0.50 -output = 3.00 +input = 0.5 +output = 3 cache_read = 0.05 cache_write = 0.625 [[cost.tiers]] -tier = { size = 256_000 } -input = 2.00 -output = 6.00 -cache_read = 0.20 -cache_write = 2.50 +tier = { type = "context", size = 256_000 } +input = 2 +output = 6 +cache_read = 0.2 +cache_write = 2.5 [limit] -context = 1_000_000 output = 64_000 [modalities] -input = ["text"] -output = ["text"] +input = ["text", "image", "pdf", "video"] diff --git a/providers/zenmux/models/qwen/qwen3.7-flash.toml b/providers/zenmux/models/qwen/qwen3.7-flash.toml new file mode 100644 index 00000000000..6cc899e503a --- /dev/null +++ b/providers/zenmux/models/qwen/qwen3.7-flash.toml @@ -0,0 +1,35 @@ +# Toggle: reasoning.enabled = true|false +# Budget: reasoning.max_tokens (integer reasoning tokens) +# https://zenmux.ai/docs/api/openai/openai-list-models.html +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "alibaba/qwen3.7-flash" +name = "Qwen3.7-Flash" + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "budget_tokens" + +[cost] +input = 0.03 +output = 0.13 +cache_read = 0.003 +cache_write = 0.038 + +[[cost.tiers]] +tier = { type = "context", size = 32_000 } +input = 0.1 +output = 0.4 +cache_read = 0.01 +cache_write = 0.125 + +[[cost.tiers]] +tier = { type = "context", size = 256_000 } +input = 0.2 +output = 0.8 +cache_read = 0.02 +cache_write = 0.25 + +[limit] +output = 64_000 diff --git a/providers/zenmux/models/qwen/qwen3.7-max.toml b/providers/zenmux/models/qwen/qwen3.7-max.toml index 7a085d238f1..34e397e0ba7 100644 --- a/providers/zenmux/models/qwen/qwen3.7-max.toml +++ b/providers/zenmux/models/qwen/qwen3.7-max.toml @@ -1,8 +1,21 @@ +# Toggle: reasoning.enabled = true|false +# Budget: reasoning.max_tokens (integer reasoning tokens) +# https://zenmux.ai/docs/api/openai/openai-list-models.html +# https://zenmux.ai/docs/guide/advanced/reasoning.html base_model = "alibaba/qwen3.7-max" -reasoning_options = [{ type = "toggle" }] +name = "Qwen3.7-Max" + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "budget_tokens" [cost] -input = 2.50 -output = 7.50 -cache_read = 0.50 -cache_write = 3.125 +input = 1.25 +output = 3.75 +cache_read = 0.125 +cache_write = 1.5625 + +[limit] +output = 64_000 diff --git a/providers/zenmux/models/qwen/qwen3.7-plus.toml b/providers/zenmux/models/qwen/qwen3.7-plus.toml index 661f1915cc1..64bb828e576 100644 --- a/providers/zenmux/models/qwen/qwen3.7-plus.toml +++ b/providers/zenmux/models/qwen/qwen3.7-plus.toml @@ -1,15 +1,25 @@ +# Toggle: reasoning.enabled = true|false +# Budget: reasoning.max_tokens (integer reasoning tokens) +# https://zenmux.ai/docs/api/openai/openai-list-models.html +# https://zenmux.ai/docs/guide/advanced/reasoning.html base_model = "alibaba/qwen3.7-plus" -reasoning_options = [{ type = "toggle" }] +name = "Qwen3.7-Plus" + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "budget_tokens" [cost] -input = 0.40 -output = 1.60 -cache_read = 0.08 -cache_write = 0.50 +input = 0.4 +output = 1.6 +cache_read = 0.04 +cache_write = 0.5 [[cost.tiers]] -tier = { size = 256_000 } -input = 1.20 -output = 4.80 -cache_read = 0.24 -cache_write = 1.50 +tier = { type = "context", size = 256_000 } +input = 1.2 +output = 4.8 +cache_read = 0.12 +cache_write = 1.5 diff --git a/providers/zenmux/models/qwen/qwen3.8-2.4t-a95b.toml b/providers/zenmux/models/qwen/qwen3.8-2.4t-a95b.toml new file mode 100644 index 00000000000..fd60945a2cc --- /dev/null +++ b/providers/zenmux/models/qwen/qwen3.8-2.4t-a95b.toml @@ -0,0 +1,19 @@ +# Effort: reasoning_effort = low|medium|xhigh +# https://zenmux.ai/docs/api/openai/openai-list-models.html +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "alibaba/qwen3.8-2.4t-a95b" +name = "Qwen3.8-2.4T-A95B" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "xhigh"] + +[cost] +input = 2 +output = 6 +cache_read = 0.17 +cache_write = 2.5 + +[limit] +context = 1_000_000 +output = 64_000 diff --git a/providers/zenmux/models/qwen/qwen3.8-27b.toml b/providers/zenmux/models/qwen/qwen3.8-27b.toml new file mode 100644 index 00000000000..c574b301e05 --- /dev/null +++ b/providers/zenmux/models/qwen/qwen3.8-27b.toml @@ -0,0 +1,23 @@ +# Toggle: reasoning.enabled = true|false +# Effort: reasoning_effort = low|medium|xhigh +# https://zenmux.ai/docs/api/openai/openai-list-models.html +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "alibaba/qwen3.8-27b" +name = "Qwen3.8-27B" + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "xhigh"] + +[cost] +input = 0.5 +output = 3 +cache_read = 0.05 +cache_write = 0.625 + +[limit] +context = 1_000_000 +output = 131_072 diff --git a/providers/zenmux/models/qwen/qwen3.8-flash.toml b/providers/zenmux/models/qwen/qwen3.8-flash.toml new file mode 100644 index 00000000000..32dbb1e3a6e --- /dev/null +++ b/providers/zenmux/models/qwen/qwen3.8-flash.toml @@ -0,0 +1,32 @@ +# Toggle: reasoning.enabled = true|false +# Effort: reasoning_effort = low|medium|xhigh +# Budget: reasoning.max_tokens (integer reasoning tokens) +# ZenMux's 2026-09-19 page explicitly reports max_completion_tokens=1000000 for +# this route; the 1M output limit is therefore an intentional host override. +# https://zenmux.ai/docs/api/openai/openai-list-models.html +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "alibaba/qwen3.8-flash" +name = "Qwen3.8-Flash" + +[interleaved] +field = "reasoning_content" + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "xhigh"] + +[[reasoning_options]] +type = "budget_tokens" +max = 262_144 + +[cost] +input = 0.16 +output = 0.47 +cache_read = 0.016 +cache_write = 0.2 + +[limit] +output = 1_000_000 diff --git a/providers/zenmux/models/qwen/qwen3.8-max-0902.toml b/providers/zenmux/models/qwen/qwen3.8-max-0902.toml new file mode 100644 index 00000000000..4b710b76a1c --- /dev/null +++ b/providers/zenmux/models/qwen/qwen3.8-max-0902.toml @@ -0,0 +1,33 @@ +# Toggle: reasoning.enabled = true|false +# Effort: reasoning_effort = low|medium|xhigh +# Budget: reasoning.max_tokens (integer reasoning tokens) +# ZenMux's 2026-09-19 page and both live OpenAI-/Anthropic-compatible lists +# advertise text, image, and video input only; PDF is intentionally omitted. +# https://zenmux.ai/docs/api/openai/openai-list-models.html +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "alibaba/qwen3.8-max-0902" +name = "Qwen3.8-Max-0902" + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "xhigh"] + +[[reasoning_options]] +type = "budget_tokens" +min = 0 +max = 262_144 + +[cost] +input = 2 +output = 6 +cache_read = 0.17 +cache_write = 2.5 + +[limit] +output = 128_000 + +[modalities] +input = ["text", "image", "video"] diff --git a/providers/zenmux/models/qwen/qwen3.8-max.toml b/providers/zenmux/models/qwen/qwen3.8-max.toml new file mode 100644 index 00000000000..7c5111f7b93 --- /dev/null +++ b/providers/zenmux/models/qwen/qwen3.8-max.toml @@ -0,0 +1,36 @@ +# Toggle: reasoning.enabled = true|false +# Effort: reasoning_effort = low|medium|xhigh +# Budget: reasoning.max_tokens (integer reasoning tokens) +# ZenMux's 2026-09-19 page and both live OpenAI-/Anthropic-compatible lists +# advertise text, image, and video input only; PDF is intentionally omitted. +# https://zenmux.ai/docs/api/openai/openai-list-models.html +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "alibaba/qwen3.8-max" +name = "Qwen3.8-Max" + +[interleaved] +field = "reasoning_content" + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "xhigh"] + +[[reasoning_options]] +type = "budget_tokens" +min = 0 +max = 262_144 + +[cost] +input = 1.4 +output = 4.2 +cache_read = 0.119 +cache_write = 1.75 + +[limit] +output = 64_000 + +[modalities] +input = ["text", "image", "video"]