From 545c00e6c5bb5e294c408306b383c794da4e2f55 Mon Sep 17 00:00:00 2001 From: Alcatraz-Zhang Date: Sat, 19 Sep 2026 06:13:39 +0800 Subject: [PATCH 1/5] chore(sync): update core ZenMux model families --- models/deepseek/deepseek-r1-0528.toml | 29 +++++++++++++ models/deepseek/deepseek-v3.2-exp.toml | 30 ++++++++++++++ .../google/gemini-omni-1.1-flash-preview.toml | 25 +++++++++++ models/google/veo-3.1-fast-generate-001.toml | 25 +++++++++++ models/google/veo-3.1-generate-001.toml | 25 +++++++++++ models/google/veo-3.1-lite-generate-001.toml | 25 +++++++++++ models/openai/gpt-transcribe.toml | 25 +++++++++++ models/openai/text-embedding-3-large.toml | 25 +++++++++++ models/openai/text-embedding-3-small.toml | 25 +++++++++++ .../models/anthropic/claude-3.5-haiku.toml | 29 ------------- .../models/anthropic/claude-3.7-sonnet.toml | 30 -------------- .../models/anthropic/claude-fable-5.1.toml | 17 ++++++++ .../models/anthropic/claude-fable-5.toml | 15 ++++--- .../models/anthropic/claude-haiku-4.5.toml | 31 +++++--------- .../models/anthropic/claude-opus-4.1.toml | 32 +++++---------- .../models/anthropic/claude-opus-4.5.toml | 36 ++++++++-------- .../models/anthropic/claude-opus-4.6.toml | 37 +++++++---------- .../models/anthropic/claude-opus-4.7.toml | 33 +++++---------- .../models/anthropic/claude-opus-4.8.toml | 13 ++++-- .../models/anthropic/claude-opus-4.toml | 29 ------------- .../models/anthropic/claude-opus-5.toml | 17 ++++++++ .../models/anthropic/claude-sonnet-4.5.toml | 38 ++++++++--------- .../models/anthropic/claude-sonnet-4.6.toml | 38 ++++++++--------- .../models/anthropic/claude-sonnet-4.toml | 29 ------------- .../anthropic/claude-sonnet-5-free.toml | 13 ------ .../models/anthropic/claude-sonnet-5.toml | 22 +++++++--- .../models/deepseek/deepseek-chat-v3.1.toml | 13 ++++++ .../zenmux/models/deepseek/deepseek-chat.toml | 24 ----------- .../models/deepseek/deepseek-r1-0528.toml | 12 ++++++ .../models/deepseek/deepseek-v3.2-exp.toml | 29 ++++--------- .../zenmux/models/deepseek/deepseek-v3.2.toml | 29 +++++-------- .../models/deepseek/deepseek-v4-flash.toml | 13 +++++- .../models/deepseek/deepseek-v4-pro.toml | 13 +++++- .../models/deepseek/deepseek-v4.1-flash.toml | 11 +++++ .../models/google/gemini-2.5-flash-image.toml | 12 ++++++ .../models/google/gemini-2.5-flash-lite.toml | 38 ++++++++--------- .../models/google/gemini-2.5-flash.toml | 40 ++++++++---------- .../zenmux/models/google/gemini-2.5-pro.toml | 41 +++++++++---------- .../models/google/gemini-3-flash-preview.toml | 30 +++++--------- .../models/google/gemini-3-pro-image.toml | 10 +++++ .../models/google/gemini-3.1-flash-image.toml | 18 ++++++++ .../google/gemini-3.1-flash-lite-image.toml | 15 +++++++ .../google/gemini-3.1-flash-lite-preview.toml | 21 ---------- .../models/google/gemini-3.1-flash-lite.toml | 12 +++++- .../google/gemini-3.1-flash-tts-preview.toml | 8 ++++ .../models/google/gemini-3.1-pro-preview.toml | 40 +++++++++--------- .../models/google/gemini-3.5-flash-lite.toml | 15 +++++++ .../models/google/gemini-3.5-flash.toml | 15 +++++-- .../models/google/gemini-3.6-flash.toml | 15 +++++++ .../models/google/gemini-3.7-flash.toml | 15 +++++++ .../models/google/gemini-3.8-flash.toml | 12 ++++++ .../models/google/gemini-embedding-2.toml | 8 ++++ .../google/gemini-omni-1.1-flash-preview.toml | 6 +++ .../google/gemini-omni-flash-preview.toml | 12 ++++++ .../models/google/gemma-4-26b-a4b-it.toml | 13 ++++++ .../zenmux/models/google/gemma-4-31b-it.toml | 8 ++++ .../google/veo-3.1-fast-generate-001.toml | 4 ++ .../models/google/veo-3.1-generate-001.toml | 7 ++++ .../google/veo-3.1-lite-generate-001.toml | 4 ++ .../models/moonshotai/kimi-k2-0905.toml | 24 ----------- .../moonshotai/kimi-k2-thinking-turbo.toml | 25 ----------- .../models/moonshotai/kimi-k2-thinking.toml | 25 ----------- .../zenmux/models/moonshotai/kimi-k2.5.toml | 27 ------------ .../zenmux/models/moonshotai/kimi-k2.6.toml | 31 ++++---------- .../moonshotai/kimi-k2.7-code-free.toml | 8 ---- .../moonshotai/kimi-k2.7-code-highspeed.toml | 8 ++++ .../models/moonshotai/kimi-k2.7-code.toml | 4 +- .../models/moonshotai/kimi-k2.8-preview.toml | 19 +++++++++ .../models/moonshotai/kimi-k3-free.toml | 21 ---------- .../zenmux/models/moonshotai/kimi-k3.toml | 21 ++++++---- ...{gpt-5.5-instant.toml => chat-latest.toml} | 3 +- .../zenmux/models/openai/gpt-4.1-mini.toml | 7 ++++ .../zenmux/models/openai/gpt-4.1-nano.toml | 10 +++++ providers/zenmux/models/openai/gpt-4.1.toml | 6 +++ .../zenmux/models/openai/gpt-4o-mini.toml | 7 ++++ providers/zenmux/models/openai/gpt-4o.toml | 6 +++ .../zenmux/models/openai/gpt-5-codex.toml | 28 ++++--------- .../zenmux/models/openai/gpt-5-mini.toml | 15 +++++++ .../zenmux/models/openai/gpt-5-nano.toml | 15 +++++++ providers/zenmux/models/openai/gpt-5-pro.toml | 17 ++++++++ .../zenmux/models/openai/gpt-5.1-chat.toml | 29 ------------- .../models/openai/gpt-5.1-codex-mini.toml | 28 +++++-------- .../zenmux/models/openai/gpt-5.1-codex.toml | 29 ++++--------- providers/zenmux/models/openai/gpt-5.1.toml | 27 ++++-------- .../zenmux/models/openai/gpt-5.2-codex.toml | 29 ++++--------- .../zenmux/models/openai/gpt-5.2-pro.toml | 29 +++++-------- providers/zenmux/models/openai/gpt-5.2.toml | 27 ++++-------- .../zenmux/models/openai/gpt-5.3-chat.toml | 26 ------------ .../zenmux/models/openai/gpt-5.3-codex.toml | 30 +++++--------- .../zenmux/models/openai/gpt-5.4-mini.toml | 26 +++++------- .../zenmux/models/openai/gpt-5.4-nano.toml | 26 +++++------- .../zenmux/models/openai/gpt-5.4-pro.toml | 32 +++++++-------- providers/zenmux/models/openai/gpt-5.4.toml | 35 +++++++--------- .../zenmux/models/openai/gpt-5.5-pro.toml | 7 +++- providers/zenmux/models/openai/gpt-5.5.toml | 7 +++- .../zenmux/models/openai/gpt-5.6-luna.toml | 25 ++++++----- .../zenmux/models/openai/gpt-5.6-sol.toml | 25 ++++++----- .../zenmux/models/openai/gpt-5.6-terra.toml | 25 ++++++----- providers/zenmux/models/openai/gpt-5.toml | 27 ++++-------- .../zenmux/models/openai/gpt-6-astra.toml | 20 +++++++++ .../zenmux/models/openai/gpt-image-1.5.toml | 12 ++++++ .../models/openai/gpt-image-2.5-flare.toml | 10 +++++ .../models/openai/gpt-image-2.5-sunburst.toml | 10 +++++ .../zenmux/models/openai/gpt-image-2.toml | 9 ++++ .../zenmux/models/openai/gpt-transcribe.toml | 1 + providers/zenmux/models/openai/o4-mini.toml | 16 ++++++++ .../models/openai/text-embedding-3-large.toml | 10 +++++ .../models/openai/text-embedding-3-small.toml | 10 +++++ 108 files changed, 1188 insertions(+), 982 deletions(-) create mode 100644 models/deepseek/deepseek-r1-0528.toml create mode 100644 models/deepseek/deepseek-v3.2-exp.toml create mode 100644 models/google/gemini-omni-1.1-flash-preview.toml create mode 100644 models/google/veo-3.1-fast-generate-001.toml create mode 100644 models/google/veo-3.1-generate-001.toml create mode 100644 models/google/veo-3.1-lite-generate-001.toml create mode 100644 models/openai/gpt-transcribe.toml create mode 100644 models/openai/text-embedding-3-large.toml create mode 100644 models/openai/text-embedding-3-small.toml delete mode 100644 providers/zenmux/models/anthropic/claude-3.5-haiku.toml delete mode 100644 providers/zenmux/models/anthropic/claude-3.7-sonnet.toml create mode 100644 providers/zenmux/models/anthropic/claude-fable-5.1.toml delete mode 100644 providers/zenmux/models/anthropic/claude-opus-4.toml create mode 100644 providers/zenmux/models/anthropic/claude-opus-5.toml delete mode 100644 providers/zenmux/models/anthropic/claude-sonnet-4.toml delete mode 100644 providers/zenmux/models/anthropic/claude-sonnet-5-free.toml create mode 100644 providers/zenmux/models/deepseek/deepseek-chat-v3.1.toml delete mode 100644 providers/zenmux/models/deepseek/deepseek-chat.toml create mode 100644 providers/zenmux/models/deepseek/deepseek-r1-0528.toml create mode 100644 providers/zenmux/models/deepseek/deepseek-v4.1-flash.toml create mode 100644 providers/zenmux/models/google/gemini-2.5-flash-image.toml create mode 100644 providers/zenmux/models/google/gemini-3-pro-image.toml create mode 100644 providers/zenmux/models/google/gemini-3.1-flash-image.toml create mode 100644 providers/zenmux/models/google/gemini-3.1-flash-lite-image.toml delete mode 100644 providers/zenmux/models/google/gemini-3.1-flash-lite-preview.toml create mode 100644 providers/zenmux/models/google/gemini-3.1-flash-tts-preview.toml create mode 100644 providers/zenmux/models/google/gemini-3.5-flash-lite.toml create mode 100644 providers/zenmux/models/google/gemini-3.6-flash.toml create mode 100644 providers/zenmux/models/google/gemini-3.7-flash.toml create mode 100644 providers/zenmux/models/google/gemini-3.8-flash.toml create mode 100644 providers/zenmux/models/google/gemini-embedding-2.toml create mode 100644 providers/zenmux/models/google/gemini-omni-1.1-flash-preview.toml create mode 100644 providers/zenmux/models/google/gemini-omni-flash-preview.toml create mode 100644 providers/zenmux/models/google/gemma-4-26b-a4b-it.toml create mode 100644 providers/zenmux/models/google/gemma-4-31b-it.toml create mode 100644 providers/zenmux/models/google/veo-3.1-fast-generate-001.toml create mode 100644 providers/zenmux/models/google/veo-3.1-generate-001.toml create mode 100644 providers/zenmux/models/google/veo-3.1-lite-generate-001.toml delete mode 100644 providers/zenmux/models/moonshotai/kimi-k2-0905.toml delete mode 100644 providers/zenmux/models/moonshotai/kimi-k2-thinking-turbo.toml delete mode 100644 providers/zenmux/models/moonshotai/kimi-k2-thinking.toml delete mode 100644 providers/zenmux/models/moonshotai/kimi-k2.5.toml delete mode 100644 providers/zenmux/models/moonshotai/kimi-k2.7-code-free.toml create mode 100644 providers/zenmux/models/moonshotai/kimi-k2.7-code-highspeed.toml create mode 100644 providers/zenmux/models/moonshotai/kimi-k2.8-preview.toml delete mode 100644 providers/zenmux/models/moonshotai/kimi-k3-free.toml rename providers/zenmux/models/openai/{gpt-5.5-instant.toml => chat-latest.toml} (52%) create mode 100644 providers/zenmux/models/openai/gpt-4.1-mini.toml create mode 100644 providers/zenmux/models/openai/gpt-4.1-nano.toml create mode 100644 providers/zenmux/models/openai/gpt-4.1.toml create mode 100644 providers/zenmux/models/openai/gpt-4o-mini.toml create mode 100644 providers/zenmux/models/openai/gpt-4o.toml create mode 100644 providers/zenmux/models/openai/gpt-5-mini.toml create mode 100644 providers/zenmux/models/openai/gpt-5-nano.toml create mode 100644 providers/zenmux/models/openai/gpt-5-pro.toml delete mode 100644 providers/zenmux/models/openai/gpt-5.1-chat.toml delete mode 100644 providers/zenmux/models/openai/gpt-5.3-chat.toml create mode 100644 providers/zenmux/models/openai/gpt-6-astra.toml create mode 100644 providers/zenmux/models/openai/gpt-image-1.5.toml create mode 100644 providers/zenmux/models/openai/gpt-image-2.5-flare.toml create mode 100644 providers/zenmux/models/openai/gpt-image-2.5-sunburst.toml create mode 100644 providers/zenmux/models/openai/gpt-image-2.toml create mode 100644 providers/zenmux/models/openai/gpt-transcribe.toml create mode 100644 providers/zenmux/models/openai/o4-mini.toml create mode 100644 providers/zenmux/models/openai/text-embedding-3-large.toml create mode 100644 providers/zenmux/models/openai/text-embedding-3-small.toml diff --git a/models/deepseek/deepseek-r1-0528.toml b/models/deepseek/deepseek-r1-0528.toml new file mode 100644 index 00000000000..2eb59f40062 --- /dev/null +++ b/models/deepseek/deepseek-r1-0528.toml @@ -0,0 +1,29 @@ +# Sources: +# - https://huggingface.co/deepseek-ai/DeepSeek-R1-0528 +# - https://zenmux.ai/models +# Accessed 2026-09-19. +name = "DeepSeek R1 0528" +description = "May 2025 update to DeepSeek R1 with stronger reasoning, math, coding, and tool use" +family = "deepseek-thinking" +release_date = "2025-05-28" +last_updated = "2025-05-28" +attachment = false +reasoning = true +temperature = true +tool_call = true +structured_output = true +knowledge = "2024-07" +open_weights = true +license = "MIT" + +[limit] +context = 128_000 +output = 32_768 + +[modalities] +input = ["text"] +output = ["text"] + +[[weights]] +label = "Hugging Face" +url = "https://huggingface.co/deepseek-ai/DeepSeek-R1-0528" diff --git a/models/deepseek/deepseek-v3.2-exp.toml b/models/deepseek/deepseek-v3.2-exp.toml new file mode 100644 index 00000000000..756435a6304 --- /dev/null +++ b/models/deepseek/deepseek-v3.2-exp.toml @@ -0,0 +1,30 @@ +# Sources: +# - https://zenmux.ai/models +# - https://zenmux.ai/docs/api/openai/openai-list-models.html +# - https://zenmux.ai/docs/api/anthropic/anthropic-list-models.html +# - https://zenmux.ai/docs/api/vertexai/google-list-models.html +# - https://huggingface.co/deepseek-ai/DeepSeek-V3.2-Exp +# Accessed 2026-09-19. +name = "DeepSeek-V3.2-Exp" +description = "DeepSeek-V3.2-Exp is an experimental model version, serving as an intermediate step toward the next-generation architecture. Built on the foundation of V3.1-Terminus, it introduces the DeepSeek Sparse Attention (DSA) mechanism—a sparse attention mechanism designed to explore and validate the optimization of training and inference efficiency in long-context scenarios. This experimental version represents the team's continuous research on more efficient Transformer architectures, with a specific focus on improving computational efficiency when processing long text sequences. For the first time, DSA enables fine-grained sparse attention, which significantly enhances the efficiency of long-context training and inference while maintaining almost unchanged model output quality." +family = "deepseek" +release_date = "2025-09-29" +last_updated = "2025-09-29" +attachment = false +reasoning = true +temperature = true +tool_call = true +structured_output = true +open_weights = true + +[limit] +context = 163_840 +output = 65_536 + +[modalities] +input = ["text"] +output = ["text"] + +[[weights]] +label = "Hugging Face" +url = "https://huggingface.co/deepseek-ai/DeepSeek-V3.2-Exp" diff --git a/models/google/gemini-omni-1.1-flash-preview.toml b/models/google/gemini-omni-1.1-flash-preview.toml new file mode 100644 index 00000000000..08075b5e5d5 --- /dev/null +++ b/models/google/gemini-omni-1.1-flash-preview.toml @@ -0,0 +1,25 @@ +# Sources: +# - https://zenmux.ai/models +# - https://zenmux.ai/docs/api/openai/openai-list-models.html +# - https://zenmux.ai/docs/api/anthropic/anthropic-list-models.html +# - https://zenmux.ai/docs/api/vertexai/google-list-models.html +# Accessed 2026-09-19. +name = "Gemini Omni 1.1 Flash Preview" +description = "Gemini Omni 1.1 Flash (Preview) is a multimodal model designed for video, image, and text tasks. It is optimized for video generation, offering video output alongside text responses in a single model." +family = "gemini" +release_date = "2026-08-31" +last_updated = "2026-08-31" +attachment = true +reasoning = true +temperature = false +tool_call = false +structured_output = false +open_weights = false + +[limit] +context = 131_072 +output = 57_920 + +[modalities] +input = ["text", "image", "video"] +output = ["text", "video"] diff --git a/models/google/veo-3.1-fast-generate-001.toml b/models/google/veo-3.1-fast-generate-001.toml new file mode 100644 index 00000000000..9811836e922 --- /dev/null +++ b/models/google/veo-3.1-fast-generate-001.toml @@ -0,0 +1,25 @@ +# Sources: +# - https://zenmux.ai/models +# - https://zenmux.ai/docs/api/openai/openai-list-models.html +# - https://zenmux.ai/docs/api/anthropic/anthropic-list-models.html +# - https://zenmux.ai/docs/api/vertexai/google-list-models.html +# Accessed 2026-09-19. +name = "Veo 3.1 Fast" +description = "Veo 3.1 Fast is a speed-optimized variant of Google DeepMind's flagship video generation model. It is designed to generate high-quality video significantly faster and at a lower cost than the standard Veo 3.1 Quality model, making it ideal for rapid prototyping and high-volume content creation." +family = "veo" +release_date = "2025-10-15" +last_updated = "2026-01-01" +attachment = true +reasoning = false +temperature = false +tool_call = false +structured_output = false +open_weights = false + +[limit] +context = 1_024 +output = 0 + +[modalities] +input = ["text", "image", "video"] +output = ["video"] diff --git a/models/google/veo-3.1-generate-001.toml b/models/google/veo-3.1-generate-001.toml new file mode 100644 index 00000000000..0eaa229aa72 --- /dev/null +++ b/models/google/veo-3.1-generate-001.toml @@ -0,0 +1,25 @@ +# Sources: +# - https://zenmux.ai/models +# - https://zenmux.ai/docs/api/openai/openai-list-models.html +# - https://zenmux.ai/docs/api/anthropic/anthropic-list-models.html +# - https://zenmux.ai/docs/api/vertexai/google-list-models.html +# Accessed 2026-09-19. +name = "Veo 3.1" +description = "Veo 3.1 is Google's state-of-the-art model for generating high-fidelity, 8-second 720p, 1080p or 4k videos featuring stunning realism and natively generated audio. You can access this model programmatically using the Gemini API. To learn more about the available Veo model variants, see the Model Versions section." +family = "veo" +release_date = "2025-10-15" +last_updated = "2026-01" +attachment = true +reasoning = false +temperature = false +tool_call = false +structured_output = false +open_weights = false + +[limit] +context = 1_024 +output = 0 + +[modalities] +input = ["text", "image"] +output = ["video"] diff --git a/models/google/veo-3.1-lite-generate-001.toml b/models/google/veo-3.1-lite-generate-001.toml new file mode 100644 index 00000000000..3212a720350 --- /dev/null +++ b/models/google/veo-3.1-lite-generate-001.toml @@ -0,0 +1,25 @@ +# Sources: +# - https://zenmux.ai/models +# - https://zenmux.ai/docs/api/openai/openai-list-models.html +# - https://zenmux.ai/docs/api/anthropic/anthropic-list-models.html +# - https://zenmux.ai/docs/api/vertexai/google-list-models.html +# Accessed 2026-09-19. +name = "Veo 3.1 Lite" +description = "Veo 3.1 Lite is Google DeepMind's most cost-efficient AI video generation model, released on March 31, 2026. It is designed to provide professional-grade video capabilities at a significantly lower price point, making it ideal for developers and content teams who need to scale high-volume video production" +family = "veo" +release_date = "2026-03-31" +last_updated = "2026-03-31" +attachment = true +reasoning = false +temperature = false +tool_call = false +structured_output = false +open_weights = false + +[limit] +context = 1_024 +output = 0 + +[modalities] +input = ["text", "image"] +output = ["video"] diff --git a/models/openai/gpt-transcribe.toml b/models/openai/gpt-transcribe.toml new file mode 100644 index 00000000000..52686fb5a2b --- /dev/null +++ b/models/openai/gpt-transcribe.toml @@ -0,0 +1,25 @@ +# Sources: +# - https://zenmux.ai/models +# - https://zenmux.ai/docs/api/openai/openai-list-models.html +# - https://zenmux.ai/docs/api/anthropic/anthropic-list-models.html +# - https://zenmux.ai/docs/api/vertexai/google-list-models.html +# Accessed 2026-09-19. +name = "GPT Transcribe" +description = "GPT Transcribe is a high-accuracy speech-to-text model from OpenAI. It is suited for recorded audio, streamed file transcription, and committed Realtime turns, with free-form context, keyword hints, and multiple language hints for specialized terms and multilingual speech." +family = "gpt" +release_date = "2026-08-06" +last_updated = "2026-08-06" +attachment = true +reasoning = false +temperature = false +tool_call = false +structured_output = false +open_weights = false + +[limit] +context = 15_000 +output = 15_000 + +[modalities] +input = ["audio"] +output = ["text"] diff --git a/models/openai/text-embedding-3-large.toml b/models/openai/text-embedding-3-large.toml new file mode 100644 index 00000000000..943de42b401 --- /dev/null +++ b/models/openai/text-embedding-3-large.toml @@ -0,0 +1,25 @@ +# Sources: +# - https://zenmux.ai/models +# - https://zenmux.ai/docs/api/openai/openai-list-models.html +# - https://zenmux.ai/docs/api/anthropic/anthropic-list-models.html +# - https://zenmux.ai/docs/api/vertexai/google-list-models.html +# Accessed 2026-09-19. +name = "text-embedding-3-large" +description = "Embedding model for semantic search, retrieval, clustering, and ranking pipelines" +family = "text-embedding" +release_date = "2024-01-25" +last_updated = "2024-01-25" +attachment = false +reasoning = false +temperature = false +tool_call = false +knowledge = "2024-01" +open_weights = false + +[limit] +context = 8_191 +output = 3_072 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/models/openai/text-embedding-3-small.toml b/models/openai/text-embedding-3-small.toml new file mode 100644 index 00000000000..932c390266a --- /dev/null +++ b/models/openai/text-embedding-3-small.toml @@ -0,0 +1,25 @@ +# Sources: +# - https://zenmux.ai/models +# - https://zenmux.ai/docs/api/openai/openai-list-models.html +# - https://zenmux.ai/docs/api/anthropic/anthropic-list-models.html +# - https://zenmux.ai/docs/api/vertexai/google-list-models.html +# Accessed 2026-09-19. +name = "text-embedding-3-small" +description = "Embedding model for semantic search, retrieval, clustering, and ranking pipelines" +family = "text-embedding" +release_date = "2024-01-25" +last_updated = "2024-01-25" +attachment = false +reasoning = false +temperature = false +tool_call = false +knowledge = "2024-01" +open_weights = false + +[limit] +context = 8_191 +output = 1_536 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/zenmux/models/anthropic/claude-3.5-haiku.toml b/providers/zenmux/models/anthropic/claude-3.5-haiku.toml deleted file mode 100644 index c3b0a3607c8..00000000000 --- a/providers/zenmux/models/anthropic/claude-3.5-haiku.toml +++ /dev/null @@ -1,29 +0,0 @@ -name = "Claude 3.5 Haiku" -description = "Fast Claude model for responsive assistance, classification, and lightweight agents" -release_date = "2024-11-04" -last_updated = "2024-11-04" -attachment = true -reasoning = false -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false -status = "deprecated" - -[cost] -input = 0.80 -output = 4.00 -cache_read = 0.08 -cache_write = 1.00 - -[limit] -context = 200_000 -output = 64_000 - -[modalities] -input = ["text", "image"] -output = ["text"] - -[provider] -npm = "@ai-sdk/anthropic" -api = "https://zenmux.ai/api/anthropic/v1" diff --git a/providers/zenmux/models/anthropic/claude-3.7-sonnet.toml b/providers/zenmux/models/anthropic/claude-3.7-sonnet.toml deleted file mode 100644 index e1b9a104ca2..00000000000 --- a/providers/zenmux/models/anthropic/claude-3.7-sonnet.toml +++ /dev/null @@ -1,30 +0,0 @@ -name = "Claude 3.7 Sonnet" -description = "Balanced Claude model for coding, analysis, agent workflows, and cost control" -release_date = "2025-02-24" -last_updated = "2025-02-24" -attachment = true -reasoning = true -reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false -status = "deprecated" - -[cost] -input = 3.00 -output = 15.00 -cache_read = 0.30 -cache_write = 3.75 - -[limit] -context = 200_000 -output = 64_000 - -[modalities] -input = ["text", "image", "pdf"] -output = ["text"] - -[provider] -npm = "@ai-sdk/anthropic" -api = "https://zenmux.ai/api/anthropic/v1" diff --git a/providers/zenmux/models/anthropic/claude-fable-5.1.toml b/providers/zenmux/models/anthropic/claude-fable-5.1.toml new file mode 100644 index 00000000000..0673940f3a3 --- /dev/null +++ b/providers/zenmux/models/anthropic/claude-fable-5.1.toml @@ -0,0 +1,17 @@ +# Effort: reasoning_effort = low|medium|high|xhigh|max +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "anthropic/claude-fable-5-1" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high", "xhigh", "max"] + +[cost] +input = 10 +output = 50 +cache_read = 0.25 +cache_write = 12.5 + +[provider] +npm = "@ai-sdk/anthropic" +api = "https://zenmux.ai/api/anthropic/v1" diff --git a/providers/zenmux/models/anthropic/claude-fable-5.toml b/providers/zenmux/models/anthropic/claude-fable-5.toml index 16e066bd842..5f1b50e3e5a 100644 --- a/providers/zenmux/models/anthropic/claude-fable-5.toml +++ b/providers/zenmux/models/anthropic/claude-fable-5.toml @@ -1,11 +1,16 @@ +# Effort: reasoning_effort = low|medium|high|xhigh|max +# https://zenmux.ai/docs/guide/advanced/reasoning.html base_model = "anthropic/claude-fable-5" -reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high", "xhigh", "max"] [cost] -input = 10.00 -output = 50.00 -cache_read = 1.00 -cache_write = 12.50 +input = 10 +output = 50 +cache_read = 1 +cache_write = 12.5 [provider] npm = "@ai-sdk/anthropic" diff --git a/providers/zenmux/models/anthropic/claude-haiku-4.5.toml b/providers/zenmux/models/anthropic/claude-haiku-4.5.toml index d3b93b2f05f..d4578029bb8 100644 --- a/providers/zenmux/models/anthropic/claude-haiku-4.5.toml +++ b/providers/zenmux/models/anthropic/claude-haiku-4.5.toml @@ -1,28 +1,19 @@ +# Budget: thinking.budget_tokens (integer) +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "anthropic/claude-haiku-4-5" name = "Claude Haiku 4.5" -description = "Fast Claude model for responsive assistance, classification, and lightweight agents" -release_date = "2025-10-15" -last_updated = "2025-10-15" -attachment = true -reasoning = false -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false +structured_output = true + +[[reasoning_options]] +type = "budget_tokens" +min = 1_024 [cost] -input = 1.00 -output = 5.00 -cache_read = 0.10 +input = 1 +output = 5 +cache_read = 0.1 cache_write = 1.25 -[limit] -context = 200_000 -output = 64_000 - -[modalities] -input = ["image", "text"] -output = ["text"] - [provider] npm = "@ai-sdk/anthropic" api = "https://zenmux.ai/api/anthropic/v1" diff --git a/providers/zenmux/models/anthropic/claude-opus-4.1.toml b/providers/zenmux/models/anthropic/claude-opus-4.1.toml index 1852583315e..024c7ebddf8 100644 --- a/providers/zenmux/models/anthropic/claude-opus-4.1.toml +++ b/providers/zenmux/models/anthropic/claude-opus-4.1.toml @@ -1,29 +1,19 @@ +# Budget: thinking.budget_tokens (integer) +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "anthropic/claude-opus-4-1" name = "Claude Opus 4.1" -description = "Flagship Claude model for deep reasoning, coding, and long-horizon agents" -release_date = "2025-08-05" -last_updated = "2025-08-05" -attachment = true -reasoning = true -reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false +structured_output = true + +[[reasoning_options]] +type = "budget_tokens" +min = 1_024 [cost] -input = 15.00 -output = 75.00 -cache_read = 1.50 +input = 15 +output = 75 +cache_read = 1.5 cache_write = 18.75 -[limit] -context = 200_000 -output = 64_000 - -[modalities] -input = ["image", "text", "pdf"] -output = ["text"] - [provider] npm = "@ai-sdk/anthropic" api = "https://zenmux.ai/api/anthropic/v1" diff --git a/providers/zenmux/models/anthropic/claude-opus-4.5.toml b/providers/zenmux/models/anthropic/claude-opus-4.5.toml index 5f09cfcad6a..eac91936470 100644 --- a/providers/zenmux/models/anthropic/claude-opus-4.5.toml +++ b/providers/zenmux/models/anthropic/claude-opus-4.5.toml @@ -1,28 +1,26 @@ +# Effort: reasoning_effort = low|medium|high +# Budget: thinking.budget_tokens (integer) +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "anthropic/claude-opus-4-5" name = "Claude Opus 4.5" -description = "Flagship Claude model for deep reasoning, coding, and long-horizon agents" -release_date = "2025-11-24" -last_updated = "2025-11-24" -attachment = true -reasoning = true -reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false +structured_output = true + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high"] + +[[reasoning_options]] +type = "budget_tokens" +min = 1_024 [cost] -input = 5.00 -output = 25.00 -cache_read = 0.50 +input = 5 +output = 25 +cache_read = 0.5 cache_write = 6.25 [limit] -context = 200_000 -output = 64_000 - -[modalities] -input = ["pdf", "image", "text"] -output = ["text"] +output = 32_000 [provider] npm = "@ai-sdk/anthropic" diff --git a/providers/zenmux/models/anthropic/claude-opus-4.6.toml b/providers/zenmux/models/anthropic/claude-opus-4.6.toml index 19c0e670677..2452e86cd51 100644 --- a/providers/zenmux/models/anthropic/claude-opus-4.6.toml +++ b/providers/zenmux/models/anthropic/claude-opus-4.6.toml @@ -1,28 +1,21 @@ -name = "Claude Opus 4.6" -description = "Flagship Claude model for deep reasoning, coding, and long-horizon agents" -release_date = "2026-02-06" -last_updated = "2026-02-06" -attachment = true -reasoning = true -reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] -temperature = true -tool_call = true -knowledge = "2025-05-31" -open_weights = false +# Effort: reasoning_effort = low|medium|high|max +# Budget: thinking.budget_tokens (integer) +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "anthropic/claude-opus-4-6" -[cost] -input = 5.00 -output = 25.00 -cache_read = 0.50 -cache_write = 6.25 +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high", "max"] -[limit] -context = 1_000_000 -output = 128_000 +[[reasoning_options]] +type = "budget_tokens" +min = 1_024 -[modalities] -input = ["image", "text"] -output = ["text"] +[cost] +input = 5 +output = 25 +cache_read = 0.5 +cache_write = 6.25 [provider] npm = "@ai-sdk/anthropic" diff --git a/providers/zenmux/models/anthropic/claude-opus-4.7.toml b/providers/zenmux/models/anthropic/claude-opus-4.7.toml index 59aff8efcf6..c75963a946b 100644 --- a/providers/zenmux/models/anthropic/claude-opus-4.7.toml +++ b/providers/zenmux/models/anthropic/claude-opus-4.7.toml @@ -1,30 +1,17 @@ -name = "Claude Opus 4.7" -description = "Flagship Claude model for deep reasoning, coding, and long-horizon agents" -release_date = "2026-04-16" -last_updated = "2026-04-16" -attachment = true -reasoning = true -reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] -temperature = false -tool_call = true -knowledge = "2026-01-31" -open_weights = false +# Effort: reasoning_effort = low|medium|high|xhigh|max +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "anthropic/claude-opus-4-7" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high", "xhigh", "max"] [cost] -input = 5.00 -output = 25.00 -cache_read = 0.50 +input = 5 +output = 25 +cache_read = 0.5 cache_write = 6.25 -[limit] -context = 1_000_000 -output = 128_000 - -[modalities] -input = ["text", "image", "pdf"] -output = ["text"] - - [provider] npm = "@ai-sdk/anthropic" api = "https://zenmux.ai/api/anthropic/v1" diff --git a/providers/zenmux/models/anthropic/claude-opus-4.8.toml b/providers/zenmux/models/anthropic/claude-opus-4.8.toml index 6fbaf57cd45..9cc519f01d3 100644 --- a/providers/zenmux/models/anthropic/claude-opus-4.8.toml +++ b/providers/zenmux/models/anthropic/claude-opus-4.8.toml @@ -1,10 +1,15 @@ +# Effort: reasoning_effort = low|medium|high|xhigh|max +# https://zenmux.ai/docs/guide/advanced/reasoning.html base_model = "anthropic/claude-opus-4-8" -reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high", "xhigh", "max"] [cost] -input = 5.00 -output = 25.00 -cache_read = 0.50 +input = 5 +output = 25 +cache_read = 0.5 cache_write = 6.25 [provider] diff --git a/providers/zenmux/models/anthropic/claude-opus-4.toml b/providers/zenmux/models/anthropic/claude-opus-4.toml deleted file mode 100644 index 48c39f7c072..00000000000 --- a/providers/zenmux/models/anthropic/claude-opus-4.toml +++ /dev/null @@ -1,29 +0,0 @@ -name = "Claude Opus 4" -description = "Flagship Claude model for deep reasoning, coding, and long-horizon agents" -release_date = "2025-05-22" -last_updated = "2025-05-22" -attachment = true -reasoning = true -reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false - -[cost] -input = 15.00 -output = 75.00 -cache_read = 1.50 -cache_write = 18.75 - -[limit] -context = 200_000 -output = 32_000 - -[modalities] -input = ["image", "text", "pdf"] -output = ["text"] - -[provider] -npm = "@ai-sdk/anthropic" -api = "https://zenmux.ai/api/anthropic/v1" diff --git a/providers/zenmux/models/anthropic/claude-opus-5.toml b/providers/zenmux/models/anthropic/claude-opus-5.toml new file mode 100644 index 00000000000..2be7ab604c2 --- /dev/null +++ b/providers/zenmux/models/anthropic/claude-opus-5.toml @@ -0,0 +1,17 @@ +# Effort: reasoning_effort = low|medium|high|xhigh|max +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "anthropic/claude-opus-5" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high", "xhigh", "max"] + +[cost] +input = 5 +output = 25 +cache_read = 0.5 +cache_write = 6.25 + +[provider] +npm = "@ai-sdk/anthropic" +api = "https://zenmux.ai/api/anthropic/v1" diff --git a/providers/zenmux/models/anthropic/claude-sonnet-4.5.toml b/providers/zenmux/models/anthropic/claude-sonnet-4.5.toml index bca5d46dbc5..adb2642adad 100644 --- a/providers/zenmux/models/anthropic/claude-sonnet-4.5.toml +++ b/providers/zenmux/models/anthropic/claude-sonnet-4.5.toml @@ -1,28 +1,28 @@ +# Budget: thinking.budget_tokens (integer) +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "anthropic/claude-sonnet-4-5" name = "Claude Sonnet 4.5" -description = "Balanced Claude model for coding, analysis, agent workflows, and cost control" -release_date = "2025-09-29" -last_updated = "2025-09-29" -attachment = true -reasoning = true -reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false +structured_output = true + +[[reasoning_options]] +type = "budget_tokens" +min = 1_024 [cost] -input = 3.00 -output = 15.00 -cache_read = 0.30 +input = 3 +output = 15 +cache_read = 0.3 cache_write = 3.75 -[limit] -context = 1000_000 -output = 64_000 +[[cost.tiers]] +tier = { type = "context", size = 200_000 } +input = 6 +output = 22.5 +cache_read = 0.6 +cache_write = 7.5 -[modalities] -input = ["text", "image", "pdf"] -output = ["text"] +[limit] +context = 1_000_000 [provider] npm = "@ai-sdk/anthropic" diff --git a/providers/zenmux/models/anthropic/claude-sonnet-4.6.toml b/providers/zenmux/models/anthropic/claude-sonnet-4.6.toml index c69459b62d6..e31dfca3f91 100644 --- a/providers/zenmux/models/anthropic/claude-sonnet-4.6.toml +++ b/providers/zenmux/models/anthropic/claude-sonnet-4.6.toml @@ -1,28 +1,22 @@ -name = "Claude Sonnet 4.6" -description = "Balanced Claude model for coding, analysis, agent workflows, and cost control" -release_date = "2026-02-18" -last_updated = "2026-02-18" -attachment = true -reasoning = true -reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] -temperature = true -tool_call = true -knowledge = "2025-08-31" -open_weights = false +# Effort: reasoning_effort = low|medium|high|max +# Budget: thinking.budget_tokens (integer) +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "anthropic/claude-sonnet-4-6" +structured_output = true -[cost] -input = 3.00 -output = 15.00 -cache_read = 0.30 -cache_write = 3.75 +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high", "max"] -[limit] -context = 1_000_000 -output = 64_000 +[[reasoning_options]] +type = "budget_tokens" +min = 1_024 -[modalities] -input = ["text", "image"] -output = ["text"] +[cost] +input = 3 +output = 15 +cache_read = 0.3 +cache_write = 3.75 [provider] npm = "@ai-sdk/anthropic" diff --git a/providers/zenmux/models/anthropic/claude-sonnet-4.toml b/providers/zenmux/models/anthropic/claude-sonnet-4.toml deleted file mode 100644 index 3bd6222f82d..00000000000 --- a/providers/zenmux/models/anthropic/claude-sonnet-4.toml +++ /dev/null @@ -1,29 +0,0 @@ -name = "Claude Sonnet 4" -description = "Balanced Claude model for coding, analysis, agent workflows, and cost control" -release_date = "2025-05-22" -last_updated = "2025-05-22" -attachment = true -reasoning = true -reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false - -[cost] -input = 3.00 -output = 15.00 -cache_read = 0.30 -cache_write = 3.75 - -[limit] -context = 1000_000 -output = 64_000 - -[modalities] -input = ["image", "text", "pdf"] -output = ["text"] - -[provider] -npm = "@ai-sdk/anthropic" -api = "https://zenmux.ai/api/anthropic/v1" diff --git a/providers/zenmux/models/anthropic/claude-sonnet-5-free.toml b/providers/zenmux/models/anthropic/claude-sonnet-5-free.toml deleted file mode 100644 index 97f3f99e91b..00000000000 --- a/providers/zenmux/models/anthropic/claude-sonnet-5-free.toml +++ /dev/null @@ -1,13 +0,0 @@ -base_model = "anthropic/claude-sonnet-5" -name = "Claude Sonnet 5 (Free)" -reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] - -[cost] -input = 0.00 -output = 0.00 -cache_read = 0.00 -cache_write = 0.00 - -[provider] -npm = "@ai-sdk/anthropic" -api = "https://zenmux.ai/api/anthropic/v1" diff --git a/providers/zenmux/models/anthropic/claude-sonnet-5.toml b/providers/zenmux/models/anthropic/claude-sonnet-5.toml index 693121528d8..265d0e0b7ea 100644 --- a/providers/zenmux/models/anthropic/claude-sonnet-5.toml +++ b/providers/zenmux/models/anthropic/claude-sonnet-5.toml @@ -1,11 +1,23 @@ +# Toggle: thinking.type = enabled|disabled +# Effort: reasoning_effort = low|medium|high|xhigh|max +# https://zenmux.ai/docs/guide/advanced/reasoning.html base_model = "anthropic/claude-sonnet-5" -reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high", "xhigh", "max"] [cost] -input = 2.00 -output = 10.00 -cache_read = 0.20 -cache_write = 4.00 +input = 2 +output = 10 +cache_read = 0.2 +cache_write = 2.5 + +[limit] +output = 64_000 [provider] npm = "@ai-sdk/anthropic" diff --git a/providers/zenmux/models/deepseek/deepseek-chat-v3.1.toml b/providers/zenmux/models/deepseek/deepseek-chat-v3.1.toml new file mode 100644 index 00000000000..b18cf81d588 --- /dev/null +++ b/providers/zenmux/models/deepseek/deepseek-chat-v3.1.toml @@ -0,0 +1,13 @@ +base_model = "deepseek/deepseek-v3.1" +name = "DeepSeek V3.1" +reasoning = false +structured_output = true + +[cost] +input = 0.28 +output = 1.11 +cache_read = 0.056 + +[limit] +context = 128_000 +output = 65_536 diff --git a/providers/zenmux/models/deepseek/deepseek-chat.toml b/providers/zenmux/models/deepseek/deepseek-chat.toml deleted file mode 100644 index b437a255b41..00000000000 --- a/providers/zenmux/models/deepseek/deepseek-chat.toml +++ /dev/null @@ -1,24 +0,0 @@ -name = "DeepSeek-V3.2 (Non-thinking Mode)" -description = "DeepSeek chat model for instruction following, coding, and analysis" -release_date = "2025-12-01" -last_updated = "2025-12-01" -attachment = false -reasoning = false -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false -status = "deprecated" - -[cost] -input = 0.28 -output = 0.42 -cache_read = 0.03 - -[limit] -context = 128_000 -output = 64_000 - -[modalities] -input = ["text"] -output = ["text"] diff --git a/providers/zenmux/models/deepseek/deepseek-r1-0528.toml b/providers/zenmux/models/deepseek/deepseek-r1-0528.toml new file mode 100644 index 00000000000..d3bb3ba1179 --- /dev/null +++ b/providers/zenmux/models/deepseek/deepseek-r1-0528.toml @@ -0,0 +1,12 @@ +base_model = "deepseek/deepseek-r1-0528" +structured_output = true +reasoning_options = [] + +[cost] +input = 0.56 +output = 2.23 +cache_read = 0.112 + +[limit] +context = 64_000 +output = 64_000 diff --git a/providers/zenmux/models/deepseek/deepseek-v3.2-exp.toml b/providers/zenmux/models/deepseek/deepseek-v3.2-exp.toml index 2580aef1fb7..d922c63e17a 100644 --- a/providers/zenmux/models/deepseek/deepseek-v3.2-exp.toml +++ b/providers/zenmux/models/deepseek/deepseek-v3.2-exp.toml @@ -1,23 +1,10 @@ -name = "DeepSeek-V3.2-Exp" -description = "DeepSeek chat model for instruction following, coding, and analysis" -release_date = "2025-09-29" -last_updated = "2025-09-29" -attachment = false -reasoning = true -reasoning_options = [{ type = "toggle" }] -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false +# Toggle: thinking.type = enabled|disabled +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "deepseek/deepseek-v3.2-exp" -[cost] -input = 0.22 -output = 0.33 - -[limit] -context = 163_000 -output = 64_000 +[[reasoning_options]] +type = "toggle" -[modalities] -input = ["text"] -output = ["text"] +[cost] +input = 0.216 +output = 0.328 diff --git a/providers/zenmux/models/deepseek/deepseek-v3.2.toml b/providers/zenmux/models/deepseek/deepseek-v3.2.toml index 63edf977fda..e22843debba 100644 --- a/providers/zenmux/models/deepseek/deepseek-v3.2.toml +++ b/providers/zenmux/models/deepseek/deepseek-v3.2.toml @@ -1,23 +1,14 @@ -name = "DeepSeek V3.2" -description = "DeepSeek chat model for instruction following, coding, and analysis" -release_date = "2025-12-05" -last_updated = "2025-12-05" -attachment = false -reasoning = true -reasoning_options = [{ type = "toggle" }] -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false +# Toggle: thinking.type = enabled|disabled +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "deepseek/deepseek-v3.2" + +[[reasoning_options]] +type = "toggle" [cost] -input = 0.28 -output = 0.43 +input = 0.293 +output = 0.4395 +cache_read = 0.0293 [limit] -context = 128_000 -output = 64_000 - -[modalities] -input = ["text"] -output = ["text"] +output = 8_000 diff --git a/providers/zenmux/models/deepseek/deepseek-v4-flash.toml b/providers/zenmux/models/deepseek/deepseek-v4-flash.toml index 45ee40683c5..6c27fb4c758 100644 --- a/providers/zenmux/models/deepseek/deepseek-v4-flash.toml +++ b/providers/zenmux/models/deepseek/deepseek-v4-flash.toml @@ -1,9 +1,18 @@ -base_model = "deepseek/deepseek-v4-flash" -reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high"] }] +# Toggle: thinking.type = enabled|disabled +# Effort: reasoning_effort = low|high|max +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "deepseek/deepseek-v4-flash-0731" [interleaved] field = "reasoning_content" +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["low", "high", "max"] + [cost] input = 0.14 output = 0.28 diff --git a/providers/zenmux/models/deepseek/deepseek-v4-pro.toml b/providers/zenmux/models/deepseek/deepseek-v4-pro.toml index 5e110965760..6026f1c7ac2 100644 --- a/providers/zenmux/models/deepseek/deepseek-v4-pro.toml +++ b/providers/zenmux/models/deepseek/deepseek-v4-pro.toml @@ -1,9 +1,18 @@ -base_model = "deepseek/deepseek-v4-pro" -reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high"] }] +# Toggle: thinking.type = enabled|disabled +# Effort: reasoning_effort = high|max +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "deepseek/deepseek-v4-pro-0813" [interleaved] field = "reasoning_content" +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["high", "max"] + [cost] input = 0.435 output = 0.87 diff --git a/providers/zenmux/models/deepseek/deepseek-v4.1-flash.toml b/providers/zenmux/models/deepseek/deepseek-v4.1-flash.toml new file mode 100644 index 00000000000..95992ef7deb --- /dev/null +++ b/providers/zenmux/models/deepseek/deepseek-v4.1-flash.toml @@ -0,0 +1,11 @@ +# Toggle: thinking.type = enabled|disabled +# Effort: reasoning_effort = low|high|max +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "deepseek/deepseek-v4.1-flash" + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["low", "high", "max"] diff --git a/providers/zenmux/models/google/gemini-2.5-flash-image.toml b/providers/zenmux/models/google/gemini-2.5-flash-image.toml new file mode 100644 index 00000000000..28e69e2e250 --- /dev/null +++ b/providers/zenmux/models/google/gemini-2.5-flash-image.toml @@ -0,0 +1,12 @@ +base_model = "google/gemini-2.5-flash-image" +name = "Gemini 2.5 Flash Image (Nano Banana)" +reasoning = false +structured_output = true + +[cost] +input = 0.3 +output = 30 +cache_read = 0.03 + +[limit] +output = 8_192 diff --git a/providers/zenmux/models/google/gemini-2.5-flash-lite.toml b/providers/zenmux/models/google/gemini-2.5-flash-lite.toml index f83ea6a084c..395ce356f55 100644 --- a/providers/zenmux/models/google/gemini-2.5-flash-lite.toml +++ b/providers/zenmux/models/google/gemini-2.5-flash-lite.toml @@ -1,26 +1,22 @@ -# Vertex `thinkingBudget`: -1 (dynamic) or 512..24576; no disable sentinel is listed. -# https://docs.zenmux.ai/api/vertexai/generate-content.html (accessed 2026-06-25) +# Toggle: reasoning.enabled = true|false +# Budget: reasoning.max_tokens (integer) +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "google/gemini-2.5-flash-lite" name = "Gemini 2.5 Flash Lite" -description = "Low-latency Gemini model for high-volume multimodal and agent workloads" -release_date = "2025-07-22" -last_updated = "2025-07-22" -attachment = true -reasoning = false -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "budget_tokens" +min = 512 +max = 24_576 [cost] -input = 0.10 -output = 0.40 -cache_read = 0.03 -cache_write = 1.00 +input = 0.1 +output = 0.4 +cache_read = 0.01 +cache_write = 1 [limit] -context = 1048_000 -output = 64_000 - -[modalities] -input = ["pdf", "image", "text", "audio"] -output = ["text"] +output = 65_535 diff --git a/providers/zenmux/models/google/gemini-2.5-flash.toml b/providers/zenmux/models/google/gemini-2.5-flash.toml index 41e23a328c3..ecea4b32b36 100644 --- a/providers/zenmux/models/google/gemini-2.5-flash.toml +++ b/providers/zenmux/models/google/gemini-2.5-flash.toml @@ -1,27 +1,21 @@ -# Vertex `thinkingBudget`: -1 (dynamic) or 0..24576; 0 disables thinking. -# https://docs.zenmux.ai/api/vertexai/generate-content.html (accessed 2026-06-25) -name = "Gemini 2.5 Flash" -description = "Fast Gemini model balancing multimodal reasoning, tool use, and cost" -release_date = "2025-06-17" -last_updated = "2025-06-17" -attachment = true -reasoning = true -reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false +# Toggle: reasoning.enabled = true|false +# Budget: reasoning.max_tokens (integer) +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "google/gemini-2.5-flash" + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "budget_tokens" +min = 0 +max = 24_576 [cost] -input = 0.30 -output = 2.50 -cache_read = 0.07 -cache_write = 1.00 +input = 0.3 +output = 2.5 +cache_read = 0.03 +cache_write = 1 [limit] -context = 1048_000 -output = 64_000 - -[modalities] -input = ["pdf", "image", "text", "audio"] -output = ["text"] +output = 65_535 diff --git a/providers/zenmux/models/google/gemini-2.5-pro.toml b/providers/zenmux/models/google/gemini-2.5-pro.toml index 44216397c1e..f3f65b21018 100644 --- a/providers/zenmux/models/google/gemini-2.5-pro.toml +++ b/providers/zenmux/models/google/gemini-2.5-pro.toml @@ -1,27 +1,24 @@ -# Vertex `thinkingBudget`: -1 (dynamic) or 128..32768; thinking cannot be disabled. -# https://docs.zenmux.ai/api/vertexai/generate-content.html (accessed 2026-06-25) -name = "Gemini 2.5 Pro" -description = "Advanced Gemini model for complex reasoning, coding, and multimodal analysis" -release_date = "2025-06-17" -last_updated = "2025-06-17" -attachment = true -reasoning = true -reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false +# Budget: reasoning.max_tokens (integer) +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "google/gemini-2.5-pro" + +[[reasoning_options]] +type = "budget_tokens" +min = 128 +max = 32_768 [cost] input = 1.25 -output = 10.00 -cache_read = 0.31 -cache_write = 4.50 +output = 10 +cache_read = 0.13 +cache_write = 4.5 -[limit] -context = 1048_000 -output = 64_000 +[[cost.tiers]] +tier = { type = "context", size = 200_000 } +input = 2.5 +output = 15 +cache_read = 0.25 +cache_write = 4.5 -[modalities] -input = ["pdf", "image", "text", "audio", "video"] -output = ["text"] +[limit] +output = 65_535 diff --git a/providers/zenmux/models/google/gemini-3-flash-preview.toml b/providers/zenmux/models/google/gemini-3-flash-preview.toml index 8dd6f77cc68..dd8846f7703 100644 --- a/providers/zenmux/models/google/gemini-3-flash-preview.toml +++ b/providers/zenmux/models/google/gemini-3-flash-preview.toml @@ -1,25 +1,15 @@ -name = "Gemini 3 Flash Preview" -description = "Fast Gemini model balancing multimodal reasoning, tool use, and cost" -release_date = "2025-12-17" -last_updated = "2025-12-17" -attachment = true -reasoning = true -reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false +# Effort: reasoning_effort = minimal|low|medium|high +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "google/gemini-3-flash-preview" + +[[reasoning_options]] +type = "effort" +values = ["minimal", "low", "medium", "high"] [cost] -input = 0.50 -output = 3.00 +input = 0.5 +output = 3 cache_read = 0.05 -cache_write = 1.00 [limit] -context = 1048_000 -output = 64_000 - -[modalities] -input = ["text", "image", "pdf", "audio"] -output = ["text"] +output = 65_535 diff --git a/providers/zenmux/models/google/gemini-3-pro-image.toml b/providers/zenmux/models/google/gemini-3-pro-image.toml new file mode 100644 index 00000000000..9c317943c12 --- /dev/null +++ b/providers/zenmux/models/google/gemini-3-pro-image.toml @@ -0,0 +1,10 @@ +base_model = "google/gemini-3-pro-image" +name = "Nano Banana Pro (Gemini 3 Pro Image)" +reasoning = false + +[cost] +input = 2 +output = 120 + +[limit] +output = 8_192 diff --git a/providers/zenmux/models/google/gemini-3.1-flash-image.toml b/providers/zenmux/models/google/gemini-3.1-flash-image.toml new file mode 100644 index 00000000000..5052f88acef --- /dev/null +++ b/providers/zenmux/models/google/gemini-3.1-flash-image.toml @@ -0,0 +1,18 @@ +# Effort: reasoning_effort = minimal|high +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "google/gemini-3.1-flash-image" +name = "Nano Banana 2 (Gemini 3.1 Flash Image)" + +[[reasoning_options]] +type = "effort" +values = ["minimal", "high"] + +[cost] +input = 0.5 +output = 60 + +[limit] +context = 65_536 + +[modalities] +input = ["image", "text", "video"] diff --git a/providers/zenmux/models/google/gemini-3.1-flash-lite-image.toml b/providers/zenmux/models/google/gemini-3.1-flash-lite-image.toml new file mode 100644 index 00000000000..f36b469aec8 --- /dev/null +++ b/providers/zenmux/models/google/gemini-3.1-flash-lite-image.toml @@ -0,0 +1,15 @@ +# Effort: reasoning_effort = minimal|high +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "google/gemini-3.1-flash-lite-image" +name = "Nano Banana 2 Lite (Gemini 3.1 Flash-Lite Image)" + +[[reasoning_options]] +type = "effort" +values = ["minimal", "high"] + +[cost] +input = 0.25 +output = 30 + +[limit] +output = 32_768 diff --git a/providers/zenmux/models/google/gemini-3.1-flash-lite-preview.toml b/providers/zenmux/models/google/gemini-3.1-flash-lite-preview.toml deleted file mode 100644 index f6c3fa28eaf..00000000000 --- a/providers/zenmux/models/google/gemini-3.1-flash-lite-preview.toml +++ /dev/null @@ -1,21 +0,0 @@ -name = "Gemini 3.1 Flash Lite Preview" -description = "Low-latency Gemini model for high-volume multimodal and agent workloads" -release_date = "2025-03-20" -last_updated = "2025-03-20" -attachment = true -reasoning = false -temperature = true -tool_call = true -open_weights = false - -[cost] -input = 0.25 -output = 1.50 - -[limit] -context = 1050000 -output = 65530 - -[modalities] -input = ["text", "image", "audio", "video"] -output = ["text"] diff --git a/providers/zenmux/models/google/gemini-3.1-flash-lite.toml b/providers/zenmux/models/google/gemini-3.1-flash-lite.toml index 8e69ceae67f..d8eacb4c721 100644 --- a/providers/zenmux/models/google/gemini-3.1-flash-lite.toml +++ b/providers/zenmux/models/google/gemini-3.1-flash-lite.toml @@ -1,7 +1,15 @@ +# Effort: reasoning_effort = minimal|low|medium|high +# https://zenmux.ai/docs/guide/advanced/reasoning.html base_model = "google/gemini-3.1-flash-lite" -reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] + +[[reasoning_options]] +type = "effort" +values = ["minimal", "low", "medium", "high"] [cost] input = 0.25 -output = 1.50 +output = 1.5 cache_read = 0.025 + +[limit] +output = 65_535 diff --git a/providers/zenmux/models/google/gemini-3.1-flash-tts-preview.toml b/providers/zenmux/models/google/gemini-3.1-flash-tts-preview.toml new file mode 100644 index 00000000000..3e9d7912df9 --- /dev/null +++ b/providers/zenmux/models/google/gemini-3.1-flash-tts-preview.toml @@ -0,0 +1,8 @@ +base_model = "google/gemini-3.1-flash-tts-preview" + +[cost] +input = 1 +output = 20 + +[limit] +output = 8_192 diff --git a/providers/zenmux/models/google/gemini-3.1-pro-preview.toml b/providers/zenmux/models/google/gemini-3.1-pro-preview.toml index d614e3d8f8b..573cf6a8cfd 100644 --- a/providers/zenmux/models/google/gemini-3.1-pro-preview.toml +++ b/providers/zenmux/models/google/gemini-3.1-pro-preview.toml @@ -1,25 +1,23 @@ -name = "Gemini 3.1 Pro Preview" -description = "Advanced Gemini model for complex reasoning, coding, and multimodal analysis" -release_date = "2026-02-19" -last_updated = "2026-02-19" -attachment = true -reasoning = true -reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] -temperature = true -tool_call = true -knowledge = "2026-02-19" -open_weights = false +# Effort: reasoning_effort = low|medium|high +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "google/gemini-3.1-pro-preview" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high"] [cost] -input = 2.00 -output = 12.00 -cache_read = 0.20 -cache_write = 4.50 +input = 2 +output = 12 +cache_read = 0.2 +cache_write = 4.5 -[limit] -context = 1048_000 -output = 64_000 +[[cost.tiers]] +tier = { type = "context", size = 200_000 } +input = 4 +output = 18 +cache_read = 0.4 +cache_write = 4.5 -[modalities] -input = ["text", "image", "pdf", "audio", "video"] -output = ["text"] +[limit] +output = 65_535 diff --git a/providers/zenmux/models/google/gemini-3.5-flash-lite.toml b/providers/zenmux/models/google/gemini-3.5-flash-lite.toml new file mode 100644 index 00000000000..4f2913c2fe5 --- /dev/null +++ b/providers/zenmux/models/google/gemini-3.5-flash-lite.toml @@ -0,0 +1,15 @@ +# Effort: reasoning_effort = minimal|low|medium|high +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "google/gemini-3.5-flash-lite" + +[[reasoning_options]] +type = "effort" +values = ["minimal", "low", "medium", "high"] + +[cost] +input = 0.3 +output = 2.5 +cache_read = 0.03 + +[limit] +output = 65_535 diff --git a/providers/zenmux/models/google/gemini-3.5-flash.toml b/providers/zenmux/models/google/gemini-3.5-flash.toml index 53b64265d5b..370405193ea 100644 --- a/providers/zenmux/models/google/gemini-3.5-flash.toml +++ b/providers/zenmux/models/google/gemini-3.5-flash.toml @@ -1,7 +1,16 @@ +# Effort: reasoning_effort = minimal|low|medium|high +# https://zenmux.ai/docs/guide/advanced/reasoning.html base_model = "google/gemini-3.5-flash" -reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] + +[[reasoning_options]] +type = "effort" +values = ["minimal", "low", "medium", "high"] [cost] -input = 1.50 -output = 9.00 +input = 1.5 +output = 9 cache_read = 0.15 +cache_write = 0 + +[limit] +output = 65_535 diff --git a/providers/zenmux/models/google/gemini-3.6-flash.toml b/providers/zenmux/models/google/gemini-3.6-flash.toml new file mode 100644 index 00000000000..824853d6b60 --- /dev/null +++ b/providers/zenmux/models/google/gemini-3.6-flash.toml @@ -0,0 +1,15 @@ +# Effort: reasoning_effort = minimal|low|medium|high +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "google/gemini-3.6-flash" + +[[reasoning_options]] +type = "effort" +values = ["minimal", "low", "medium", "high"] + +[cost] +input = 1.5 +output = 7.5 +cache_read = 0.15 + +[limit] +output = 65_535 diff --git a/providers/zenmux/models/google/gemini-3.7-flash.toml b/providers/zenmux/models/google/gemini-3.7-flash.toml new file mode 100644 index 00000000000..728500331f7 --- /dev/null +++ b/providers/zenmux/models/google/gemini-3.7-flash.toml @@ -0,0 +1,15 @@ +# Effort: reasoning_effort = low|medium|high +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "google/gemini-3.7-flash" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high"] + +[cost] +input = 1.5 +output = 7.5 +cache_read = 0.15 + +[limit] +output = 65_535 diff --git a/providers/zenmux/models/google/gemini-3.8-flash.toml b/providers/zenmux/models/google/gemini-3.8-flash.toml new file mode 100644 index 00000000000..54f4e11f662 --- /dev/null +++ b/providers/zenmux/models/google/gemini-3.8-flash.toml @@ -0,0 +1,12 @@ +# Effort: reasoning_effort = low|medium|high +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "google/gemini-3.8-flash" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high"] + +[cost] +input = 1.5 +output = 7.5 +cache_read = 0.15 diff --git a/providers/zenmux/models/google/gemini-embedding-2.toml b/providers/zenmux/models/google/gemini-embedding-2.toml new file mode 100644 index 00000000000..da5123f4089 --- /dev/null +++ b/providers/zenmux/models/google/gemini-embedding-2.toml @@ -0,0 +1,8 @@ +base_model = "google/gemini-embedding-2" + +[cost] +input = 0.2 +output = 0 + +[limit] +output = 0 diff --git a/providers/zenmux/models/google/gemini-omni-1.1-flash-preview.toml b/providers/zenmux/models/google/gemini-omni-1.1-flash-preview.toml new file mode 100644 index 00000000000..37b2a2874dc --- /dev/null +++ b/providers/zenmux/models/google/gemini-omni-1.1-flash-preview.toml @@ -0,0 +1,6 @@ +base_model = "google/gemini-omni-1.1-flash-preview" +reasoning_options = [] + +[cost] +input = 1.5 +output = 17.5 diff --git a/providers/zenmux/models/google/gemini-omni-flash-preview.toml b/providers/zenmux/models/google/gemini-omni-flash-preview.toml new file mode 100644 index 00000000000..cb5858ba1a0 --- /dev/null +++ b/providers/zenmux/models/google/gemini-omni-flash-preview.toml @@ -0,0 +1,12 @@ +base_model = "google/gemini-omni-flash-preview" +reasoning_options = [] + +[cost] +input = 1.5 +output = 17.5 + +[limit] +context = 131_072 + +[modalities] +output = ["text", "video"] diff --git a/providers/zenmux/models/google/gemma-4-26b-a4b-it.toml b/providers/zenmux/models/google/gemma-4-26b-a4b-it.toml new file mode 100644 index 00000000000..7583789d90d --- /dev/null +++ b/providers/zenmux/models/google/gemma-4-26b-a4b-it.toml @@ -0,0 +1,13 @@ +base_model = "google/gemma-4-26b-a4b-it" +reasoning = false + +[cost] +input = 0.15 +output = 0.6 +cache_read = 0.015 + +[limit] +output = 65_535 + +[modalities] +input = ["image", "video", "text"] diff --git a/providers/zenmux/models/google/gemma-4-31b-it.toml b/providers/zenmux/models/google/gemma-4-31b-it.toml new file mode 100644 index 00000000000..d8cb297dbed --- /dev/null +++ b/providers/zenmux/models/google/gemma-4-31b-it.toml @@ -0,0 +1,8 @@ +base_model = "google/gemma-4-31b-it" +reasoning = false + +[limit] +output = 65_535 + +[modalities] +input = ["image", "text", "video"] diff --git a/providers/zenmux/models/google/veo-3.1-fast-generate-001.toml b/providers/zenmux/models/google/veo-3.1-fast-generate-001.toml new file mode 100644 index 00000000000..9d0f5db6876 --- /dev/null +++ b/providers/zenmux/models/google/veo-3.1-fast-generate-001.toml @@ -0,0 +1,4 @@ +base_model = "google/veo-3.1-fast-generate-001" + +[limit] +context = 100_000 diff --git a/providers/zenmux/models/google/veo-3.1-generate-001.toml b/providers/zenmux/models/google/veo-3.1-generate-001.toml new file mode 100644 index 00000000000..bde05604591 --- /dev/null +++ b/providers/zenmux/models/google/veo-3.1-generate-001.toml @@ -0,0 +1,7 @@ +base_model = "google/veo-3.1-generate-001" + +[limit] +context = 100_000 + +[modalities] +input = ["text", "image", "video"] diff --git a/providers/zenmux/models/google/veo-3.1-lite-generate-001.toml b/providers/zenmux/models/google/veo-3.1-lite-generate-001.toml new file mode 100644 index 00000000000..440fc75aabf --- /dev/null +++ b/providers/zenmux/models/google/veo-3.1-lite-generate-001.toml @@ -0,0 +1,4 @@ +base_model = "google/veo-3.1-lite-generate-001" + +[limit] +context = 100_000 diff --git a/providers/zenmux/models/moonshotai/kimi-k2-0905.toml b/providers/zenmux/models/moonshotai/kimi-k2-0905.toml deleted file mode 100644 index ea07cdb73de..00000000000 --- a/providers/zenmux/models/moonshotai/kimi-k2-0905.toml +++ /dev/null @@ -1,24 +0,0 @@ -name = "Kimi K2 0905" -description = "Kimi model for long-context chat, coding, and agentic reasoning" -release_date = "2025-09-04" -last_updated = "2025-09-04" -attachment = false -reasoning = false -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false -status = "deprecated" - -[cost] -input = 0.60 -output = 2.50 -cache_read = 0.15 - -[limit] -context = 262_000 -output = 64_000 - -[modalities] -input = ["text"] -output = ["text"] diff --git a/providers/zenmux/models/moonshotai/kimi-k2-thinking-turbo.toml b/providers/zenmux/models/moonshotai/kimi-k2-thinking-turbo.toml deleted file mode 100644 index 8cd5c98b080..00000000000 --- a/providers/zenmux/models/moonshotai/kimi-k2-thinking-turbo.toml +++ /dev/null @@ -1,25 +0,0 @@ -name = "Kimi K2 Thinking Turbo" -description = "Kimi reasoning model for long-horizon research, planning, and tool use" -release_date = "2025-11-06" -last_updated = "2025-11-06" -attachment = false -reasoning = true -reasoning_options = [] -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false -status = "deprecated" - -[cost] -input = 1.15 -output = 8.00 -cache_read = 0.15 - -[limit] -context = 262_000 -output = 64_000 - -[modalities] -input = ["text"] -output = ["text"] diff --git a/providers/zenmux/models/moonshotai/kimi-k2-thinking.toml b/providers/zenmux/models/moonshotai/kimi-k2-thinking.toml deleted file mode 100644 index 7760c8ebb87..00000000000 --- a/providers/zenmux/models/moonshotai/kimi-k2-thinking.toml +++ /dev/null @@ -1,25 +0,0 @@ -name = "Kimi K2 Thinking" -description = "Kimi reasoning model for long-horizon research, planning, and tool use" -release_date = "2025-11-06" -last_updated = "2025-11-06" -attachment = false -reasoning = true -reasoning_options = [] -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false -status = "deprecated" - -[cost] -input = 0.60 -output = 2.50 -cache_read = 0.15 - -[limit] -context = 262_000 -output = 64_000 - -[modalities] -input = ["text"] -output = ["text"] diff --git a/providers/zenmux/models/moonshotai/kimi-k2.5.toml b/providers/zenmux/models/moonshotai/kimi-k2.5.toml deleted file mode 100644 index 417b6fef769..00000000000 --- a/providers/zenmux/models/moonshotai/kimi-k2.5.toml +++ /dev/null @@ -1,27 +0,0 @@ -name = "Kimi K2.5" -description = "Kimi multimodal agent model for visual understanding, coding, and planning" -release_date = "2026-01-27" -last_updated = "2026-01-27" -attachment = true -reasoning = true -reasoning_options = [{ type = "toggle" }] -temperature = false -tool_call = true -knowledge = "2025-01-01" -open_weights = false - -[cost] -input = 0.58 -output = 3.02 -cache_read = 0.10 - -[limit] -context = 262_000 -output = 64_000 - -[interleaved] -field = "reasoning_content" - -[modalities] -input = ["text", "image", "video"] -output = ["text"] diff --git a/providers/zenmux/models/moonshotai/kimi-k2.6.toml b/providers/zenmux/models/moonshotai/kimi-k2.6.toml index c0feb549167..e12b9307b15 100644 --- a/providers/zenmux/models/moonshotai/kimi-k2.6.toml +++ b/providers/zenmux/models/moonshotai/kimi-k2.6.toml @@ -1,27 +1,14 @@ -name = "Kimi K2.6" -description = "Kimi multimodal agent model for visual understanding, coding, and planning" -release_date = "2026-04-20" -last_updated = "2026-04-20" -attachment = true -reasoning = true -reasoning_options = [{ type = "toggle" }] -temperature = false -tool_call = true -knowledge = "2025-01-01" -open_weights = true +# Toggle: thinking.type = enabled|disabled|adaptive +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "moonshotai/kimi-k2.6" + +[interleaved] +field = "reasoning_content" + +[[reasoning_options]] +type = "toggle" [cost] input = 0.95 output = 4 cache_read = 0.16 - -[limit] -context = 262_140 -output = 262_140 - -[interleaved] -field = "reasoning_content" - -[modalities] -input = ["text", "image", "video"] -output = ["text"] diff --git a/providers/zenmux/models/moonshotai/kimi-k2.7-code-free.toml b/providers/zenmux/models/moonshotai/kimi-k2.7-code-free.toml deleted file mode 100644 index b8198b1b767..00000000000 --- a/providers/zenmux/models/moonshotai/kimi-k2.7-code-free.toml +++ /dev/null @@ -1,8 +0,0 @@ -base_model="moonshotai/kimi-k2.7-code" -reasoning_options = [] -name = "Kimi K2.7 Code (Free)" - -[cost] -input = 0 -output = 0 -cache_read = 0 diff --git a/providers/zenmux/models/moonshotai/kimi-k2.7-code-highspeed.toml b/providers/zenmux/models/moonshotai/kimi-k2.7-code-highspeed.toml new file mode 100644 index 00000000000..66c82f75088 --- /dev/null +++ b/providers/zenmux/models/moonshotai/kimi-k2.7-code-highspeed.toml @@ -0,0 +1,8 @@ +base_model = "moonshotai/kimi-k2.7-code-highspeed" +name = "Kimi K2.7 Code HighSpeed" +reasoning_options = [] + +[cost] +input = 1.9 +output = 8 +cache_read = 0.38 diff --git a/providers/zenmux/models/moonshotai/kimi-k2.7-code.toml b/providers/zenmux/models/moonshotai/kimi-k2.7-code.toml index 7e744c06b46..d7fe478db1a 100644 --- a/providers/zenmux/models/moonshotai/kimi-k2.7-code.toml +++ b/providers/zenmux/models/moonshotai/kimi-k2.7-code.toml @@ -1,7 +1,7 @@ -base_model="moonshotai/kimi-k2.7-code" +base_model = "moonshotai/kimi-k2.7-code" reasoning_options = [] [cost] input = 0.95 output = 4 -cache_read = 0.16 +cache_read = 0.19 diff --git a/providers/zenmux/models/moonshotai/kimi-k2.8-preview.toml b/providers/zenmux/models/moonshotai/kimi-k2.8-preview.toml new file mode 100644 index 00000000000..c4e4ca57134 --- /dev/null +++ b/providers/zenmux/models/moonshotai/kimi-k2.8-preview.toml @@ -0,0 +1,19 @@ +# Toggle: thinking.type = enabled|disabled|adaptive +# Effort: output_config.effort = low|high|max +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "moonshotai/kimi-k2.8-preview" + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["low", "high", "max"] + +[cost] +input = 1 +output = 4 +cache_read = 0.25 + +[limit] +output = 1_048_576 diff --git a/providers/zenmux/models/moonshotai/kimi-k3-free.toml b/providers/zenmux/models/moonshotai/kimi-k3-free.toml deleted file mode 100644 index 41c692633b2..00000000000 --- a/providers/zenmux/models/moonshotai/kimi-k3-free.toml +++ /dev/null @@ -1,21 +0,0 @@ -# API: thinking.type = "enabled" | "disabled" | "adaptive"; in adaptive mode the -# reasoning depth is set by output_config.effort, which only accepts "max". -# Pricing: free tier via ZenMux (accessed 2026-07-20). -# https://docs.zenmux.ai -base_model = "moonshotai/kimi-k3" -name = "Kimi K3 (Free)" - -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "effort" -values = ["max"] - -[interleaved] -field = "reasoning_content" - -[cost] -input = 0 -output = 0 -cache_read = 0 diff --git a/providers/zenmux/models/moonshotai/kimi-k3.toml b/providers/zenmux/models/moonshotai/kimi-k3.toml index 942632af006..b24ebb3223d 100644 --- a/providers/zenmux/models/moonshotai/kimi-k3.toml +++ b/providers/zenmux/models/moonshotai/kimi-k3.toml @@ -1,19 +1,22 @@ -# API: thinking.type = "enabled" | "disabled" | "adaptive"; in adaptive mode the -# reasoning depth is set by output_config.effort, which only accepts "max". -# Pricing: https://platform.kimi.ai/docs/pricing/chat-k3 (accessed 2026-07-17). +# Toggle: thinking.type = enabled|disabled|adaptive +# Effort: output_config.effort = low|high|max +# https://zenmux.ai/docs/guide/advanced/reasoning.html base_model = "moonshotai/kimi-k3" +[interleaved] +field = "reasoning_content" + [[reasoning_options]] type = "toggle" [[reasoning_options]] type = "effort" -values = ["max"] - -[interleaved] -field = "reasoning_content" +values = ["low", "high", "max"] [cost] -input = 3.0 -output = 15.0 +input = 3 +output = 15 cache_read = 0.3 + +[limit] +output = 262_144 diff --git a/providers/zenmux/models/openai/gpt-5.5-instant.toml b/providers/zenmux/models/openai/chat-latest.toml similarity index 52% rename from providers/zenmux/models/openai/gpt-5.5-instant.toml rename to providers/zenmux/models/openai/chat-latest.toml index ffcea046e7a..14bccd2647d 100644 --- a/providers/zenmux/models/openai/gpt-5.5-instant.toml +++ b/providers/zenmux/models/openai/chat-latest.toml @@ -1,5 +1,6 @@ base_model = "openai/gpt-5.5-instant" -reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] +name = "Chat Latest (GPT-5.5 Instant)" +reasoning = false [cost] input = 5 diff --git a/providers/zenmux/models/openai/gpt-4.1-mini.toml b/providers/zenmux/models/openai/gpt-4.1-mini.toml new file mode 100644 index 00000000000..a6b4371ba16 --- /dev/null +++ b/providers/zenmux/models/openai/gpt-4.1-mini.toml @@ -0,0 +1,7 @@ +base_model = "openai/gpt-4.1-mini" +name = "GPT-4.1 Mini" + +[cost] +input = 0.4 +output = 1.6 +cache_read = 0.1 diff --git a/providers/zenmux/models/openai/gpt-4.1-nano.toml b/providers/zenmux/models/openai/gpt-4.1-nano.toml new file mode 100644 index 00000000000..959cb906795 --- /dev/null +++ b/providers/zenmux/models/openai/gpt-4.1-nano.toml @@ -0,0 +1,10 @@ +base_model = "openai/gpt-4.1-nano" +name = "GPT-4.1 Nano" + +[cost] +input = 0.1 +output = 0.4 +cache_read = 0.025 + +[modalities] +input = ["image", "text", "pdf"] diff --git a/providers/zenmux/models/openai/gpt-4.1.toml b/providers/zenmux/models/openai/gpt-4.1.toml new file mode 100644 index 00000000000..1fdb518f628 --- /dev/null +++ b/providers/zenmux/models/openai/gpt-4.1.toml @@ -0,0 +1,6 @@ +base_model = "openai/gpt-4.1" + +[cost] +input = 2 +output = 8 +cache_read = 0.5 diff --git a/providers/zenmux/models/openai/gpt-4o-mini.toml b/providers/zenmux/models/openai/gpt-4o-mini.toml new file mode 100644 index 00000000000..800564f45fd --- /dev/null +++ b/providers/zenmux/models/openai/gpt-4o-mini.toml @@ -0,0 +1,7 @@ +base_model = "openai/gpt-4o-mini" +name = "GPT-4o-mini" + +[cost] +input = 0.15 +output = 0.6 +cache_read = 0.075 diff --git a/providers/zenmux/models/openai/gpt-4o.toml b/providers/zenmux/models/openai/gpt-4o.toml new file mode 100644 index 00000000000..1b57e520127 --- /dev/null +++ b/providers/zenmux/models/openai/gpt-4o.toml @@ -0,0 +1,6 @@ +base_model = "openai/gpt-4o" + +[cost] +input = 2.5 +output = 10 +cache_read = 1.25 diff --git a/providers/zenmux/models/openai/gpt-5-codex.toml b/providers/zenmux/models/openai/gpt-5-codex.toml index 4a42dd4975e..3f0d2687e68 100644 --- a/providers/zenmux/models/openai/gpt-5-codex.toml +++ b/providers/zenmux/models/openai/gpt-5-codex.toml @@ -1,27 +1,17 @@ +# Effort: reasoning_effort = low|medium|high +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "openai/gpt-5-codex" name = "GPT-5 Codex" -description = "Coding-optimized GPT model for repository edits, reviews, and agentic software work" -release_date = "2025-09-23" -last_updated = "2025-09-23" attachment = true -reasoning = true -reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high"] [cost] input = 1.25 -output = 10.00 -cache_read = 0.12 - -[limit] -context = 400_000 -output = 64_000 - -[modalities] -input = ["text", "image"] -output = ["text"] +output = 10 +cache_read = 0.125 [provider] npm = "@ai-sdk/openai" diff --git a/providers/zenmux/models/openai/gpt-5-mini.toml b/providers/zenmux/models/openai/gpt-5-mini.toml new file mode 100644 index 00000000000..882ad0ea7de --- /dev/null +++ b/providers/zenmux/models/openai/gpt-5-mini.toml @@ -0,0 +1,15 @@ +# Effort: reasoning_effort = minimal|low|medium|high +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "openai/gpt-5-mini" + +[[reasoning_options]] +type = "effort" +values = ["minimal", "low", "medium", "high"] + +[cost] +input = 0.25 +output = 2 +cache_read = 0.025 + +[modalities] +input = ["text", "image", "pdf"] diff --git a/providers/zenmux/models/openai/gpt-5-nano.toml b/providers/zenmux/models/openai/gpt-5-nano.toml new file mode 100644 index 00000000000..83e79ef5eb0 --- /dev/null +++ b/providers/zenmux/models/openai/gpt-5-nano.toml @@ -0,0 +1,15 @@ +# Effort: reasoning_effort = minimal|low|medium|high +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "openai/gpt-5-nano" + +[[reasoning_options]] +type = "effort" +values = ["minimal", "low", "medium", "high"] + +[cost] +input = 0.05 +output = 0.4 +cache_read = 0.005 + +[modalities] +input = ["text", "image", "pdf"] diff --git a/providers/zenmux/models/openai/gpt-5-pro.toml b/providers/zenmux/models/openai/gpt-5-pro.toml new file mode 100644 index 00000000000..33dc045f0ea --- /dev/null +++ b/providers/zenmux/models/openai/gpt-5-pro.toml @@ -0,0 +1,17 @@ +# Effort: reasoning_effort = high +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "openai/gpt-5-pro" + +[[reasoning_options]] +type = "effort" +values = ["high"] + +[cost] +input = 15 +output = 120 + +[limit] +output = 128_000 + +[modalities] +input = ["image", "text", "pdf"] diff --git a/providers/zenmux/models/openai/gpt-5.1-chat.toml b/providers/zenmux/models/openai/gpt-5.1-chat.toml deleted file mode 100644 index c35f6c1085e..00000000000 --- a/providers/zenmux/models/openai/gpt-5.1-chat.toml +++ /dev/null @@ -1,29 +0,0 @@ -name = "GPT-5.1 Chat" -description = "Chat-tuned GPT model for conversational assistance, writing, and tool workflows" -release_date = "2025-11-13" -last_updated = "2025-11-13" -attachment = true -reasoning = false -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false - -status = "deprecated" - -[cost] -input = 1.25 -output = 10.00 -cache_read = 0.12 - -[limit] -context = 128_000 -output = 64_000 - -[modalities] -input = ["pdf", "image", "text"] -output = ["text"] - -[provider] -npm = "@ai-sdk/openai" -api = "https://zenmux.ai/api/v1" diff --git a/providers/zenmux/models/openai/gpt-5.1-codex-mini.toml b/providers/zenmux/models/openai/gpt-5.1-codex-mini.toml index 323be60fb0d..b5b6086fd90 100644 --- a/providers/zenmux/models/openai/gpt-5.1-codex-mini.toml +++ b/providers/zenmux/models/openai/gpt-5.1-codex-mini.toml @@ -1,27 +1,19 @@ +# Effort: reasoning_effort = low|medium|high +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "openai/gpt-5.1-codex-mini" name = "GPT-5.1-Codex-Mini" -description = "Coding-optimized GPT model for repository edits, reviews, and agentic software work" -release_date = "2025-11-13" -last_updated = "2025-11-13" -attachment = true -reasoning = true -reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high"] [cost] input = 0.25 -output = 2.00 -cache_read = 0.03 +output = 2 +cache_read = 0.025 [limit] -context = 400_000 -output = 64_000 - -[modalities] -input = ["image", "text"] -output = ["text"] +output = 100_000 [provider] npm = "@ai-sdk/openai" diff --git a/providers/zenmux/models/openai/gpt-5.1-codex.toml b/providers/zenmux/models/openai/gpt-5.1-codex.toml index 716e439736d..523eb508304 100644 --- a/providers/zenmux/models/openai/gpt-5.1-codex.toml +++ b/providers/zenmux/models/openai/gpt-5.1-codex.toml @@ -1,27 +1,16 @@ +# Effort: reasoning_effort = low|medium|high +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "openai/gpt-5.1-codex" name = "GPT-5.1-Codex" -description = "Coding-optimized GPT model for repository edits, reviews, and agentic software work" -release_date = "2025-11-13" -last_updated = "2025-11-13" -attachment = true -reasoning = true -reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high"] [cost] input = 1.25 -output = 10.00 -cache_read = 0.12 - -[limit] -context = 400_000 -output = 64_000 - -[modalities] -input = ["text", "image"] -output = ["text"] +output = 10 +cache_read = 0.125 [provider] npm = "@ai-sdk/openai" diff --git a/providers/zenmux/models/openai/gpt-5.1.toml b/providers/zenmux/models/openai/gpt-5.1.toml index f4eee5a2053..90ba7e5bdd2 100644 --- a/providers/zenmux/models/openai/gpt-5.1.toml +++ b/providers/zenmux/models/openai/gpt-5.1.toml @@ -1,27 +1,18 @@ -name = "GPT-5.1" -description = "GPT model for general reasoning, writing, coding, and tool-assisted tasks" -release_date = "2025-11-13" -last_updated = "2025-11-13" -attachment = true -reasoning = true -reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false +# Effort: reasoning_effort = none|low|medium|high +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "openai/gpt-5.1" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high"] [cost] input = 1.25 -output = 10.00 -cache_read = 0.12 - -[limit] -context = 400_000 -output = 64_000 +output = 10 +cache_read = 0.125 [modalities] input = ["image", "text", "pdf"] -output = ["text"] [provider] npm = "@ai-sdk/openai" diff --git a/providers/zenmux/models/openai/gpt-5.2-codex.toml b/providers/zenmux/models/openai/gpt-5.2-codex.toml index f7ce8937dbd..3d30d33c8d0 100644 --- a/providers/zenmux/models/openai/gpt-5.2-codex.toml +++ b/providers/zenmux/models/openai/gpt-5.2-codex.toml @@ -1,27 +1,16 @@ +# Effort: reasoning_effort = low|medium|high +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "openai/gpt-5.2-codex" name = "GPT-5.2-Codex" -description = "Coding-optimized GPT model for repository edits, reviews, and agentic software work" -release_date = "2026-01-15" -last_updated = "2026-01-15" -attachment = true -reasoning = true -reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] -temperature = false -tool_call = true -knowledge = "2025-01-01" -open_weights = false + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high"] [cost] input = 1.75 -output = 14.00 -cache_read = 0.17 - -[limit] -context = 400_000 -output = 64_000 - -[modalities] -input = ["text", "image", "pdf"] -output = ["text"] +output = 14 +cache_read = 0.175 [provider] npm = "@ai-sdk/openai" diff --git a/providers/zenmux/models/openai/gpt-5.2-pro.toml b/providers/zenmux/models/openai/gpt-5.2-pro.toml index f535798c000..30cf6fc1f61 100644 --- a/providers/zenmux/models/openai/gpt-5.2-pro.toml +++ b/providers/zenmux/models/openai/gpt-5.2-pro.toml @@ -1,26 +1,17 @@ -name = "GPT-5.2-Pro" -description = "Frontier GPT model for professional reasoning, coding, and multimodal work" -release_date = "2025-12-11" -last_updated = "2025-12-11" -attachment = true -reasoning = true -reasoning_options = [{ type = "effort", values = ["medium", "high", "xhigh"] }] -temperature = false -tool_call = true -knowledge = "2025-08-31" -open_weights = false +# Effort: reasoning_effort = medium|high|xhigh +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "openai/gpt-5.2-pro" -[cost] -input = 21.00 -output = 168.00 +[[reasoning_options]] +type = "effort" +values = ["medium", "high", "xhigh"] -[limit] -context = 400_000 -output = 128_000 +[cost] +input = 21 +output = 168 [modalities] -input = ["text", "image", "pdf"] -output = ["text"] +input = ["image", "text", "pdf"] [provider] npm = "@ai-sdk/openai" diff --git a/providers/zenmux/models/openai/gpt-5.2.toml b/providers/zenmux/models/openai/gpt-5.2.toml index c0b1515bc5c..1f83b79beff 100644 --- a/providers/zenmux/models/openai/gpt-5.2.toml +++ b/providers/zenmux/models/openai/gpt-5.2.toml @@ -1,27 +1,18 @@ -name = "GPT-5.2" -description = "GPT model for general reasoning, writing, coding, and tool-assisted tasks" -release_date = "2025-12-11" -last_updated = "2025-12-11" -attachment = true -reasoning = true -reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] -temperature = false -tool_call = true -knowledge = "2025-01-01" -open_weights = false +# Effort: reasoning_effort = none|low|medium|high|xhigh +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "openai/gpt-5.2" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh"] [cost] input = 1.75 -output = 14.00 -cache_read = 0.17 - -[limit] -context = 400_000 -output = 64_000 +output = 14 +cache_read = 0.175 [modalities] input = ["image", "text", "pdf"] -output = ["text"] [provider] npm = "@ai-sdk/openai" diff --git a/providers/zenmux/models/openai/gpt-5.3-chat.toml b/providers/zenmux/models/openai/gpt-5.3-chat.toml deleted file mode 100644 index 7caedff95f8..00000000000 --- a/providers/zenmux/models/openai/gpt-5.3-chat.toml +++ /dev/null @@ -1,26 +0,0 @@ -name = "GPT-5.3 Chat" -description = "Chat-tuned GPT model for conversational assistance, writing, and tool workflows" -release_date = "2026-03-20" -last_updated = "2026-03-20" -attachment = true -reasoning = false -temperature = true -tool_call = true -knowledge = "2025-08-31" -open_weights = false - -[cost] -input = 1.75 -output = 14.00 - -[limit] -context = 128_000 -output = 16_380 - -[modalities] -input = ["text"] -output = ["text"] - -[provider] -npm = "@ai-sdk/openai" -api = "https://zenmux.ai/api/v1" diff --git a/providers/zenmux/models/openai/gpt-5.3-codex.toml b/providers/zenmux/models/openai/gpt-5.3-codex.toml index ff999ce849a..9345a1cf537 100644 --- a/providers/zenmux/models/openai/gpt-5.3-codex.toml +++ b/providers/zenmux/models/openai/gpt-5.3-codex.toml @@ -1,26 +1,16 @@ -name = "GPT-5.3 Codex" -description = "Coding-optimized GPT model for repository edits, reviews, and agentic software work" -release_date = "2026-03-20" -last_updated = "2026-03-20" -attachment = true -reasoning = true -reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] -temperature = true -tool_call = true -knowledge = "2025-08-31" -open_weights = false +# Effort: reasoning_effort = none|low|medium|high|xhigh +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "openai/gpt-5.3-codex" +name = "GPT-5.3-Codex" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh"] [cost] input = 1.75 -output = 14.00 - -[limit] -context = 400_000 -output = 128_000 - -[modalities] -input = ["text"] -output = ["text"] +output = 14 +cache_read = 0.175 [provider] npm = "@ai-sdk/openai" diff --git a/providers/zenmux/models/openai/gpt-5.4-mini.toml b/providers/zenmux/models/openai/gpt-5.4-mini.toml index 7a14a30e636..e3bfc06e5c4 100644 --- a/providers/zenmux/models/openai/gpt-5.4-mini.toml +++ b/providers/zenmux/models/openai/gpt-5.4-mini.toml @@ -1,25 +1,19 @@ +# Effort: reasoning_effort = none|low|medium|high|xhigh +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "openai/gpt-5.4-mini" name = "GPT-5.4 Mini" -description = "Compact GPT model for low-latency assistance and high-volume workloads" -release_date = "2026-03-20" -last_updated = "2026-03-20" -attachment = true -reasoning = false -temperature = true -tool_call = true -knowledge = "2025-08-31" -open_weights = false + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh"] [cost] input = 0.75 -output = 4.50 - -[limit] -context = 400_000 -output = 128_000 +output = 4.5 +cache_read = 0.075 [modalities] -input = ["text"] -output = ["text"] +input = ["text", "image", "pdf"] [provider] npm = "@ai-sdk/openai" diff --git a/providers/zenmux/models/openai/gpt-5.4-nano.toml b/providers/zenmux/models/openai/gpt-5.4-nano.toml index 5e78fa53e0f..9e7361d40c0 100644 --- a/providers/zenmux/models/openai/gpt-5.4-nano.toml +++ b/providers/zenmux/models/openai/gpt-5.4-nano.toml @@ -1,25 +1,19 @@ +# Effort: reasoning_effort = none|low|medium|high|xhigh +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "openai/gpt-5.4-nano" name = "GPT-5.4 Nano" -description = "Compact GPT model for low-latency assistance and high-volume workloads" -release_date = "2026-03-20" -last_updated = "2026-03-20" -attachment = false -reasoning = false -temperature = true -tool_call = true -knowledge = "2025-08-31" -open_weights = false + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh"] [cost] -input = 0.20 +input = 0.2 output = 1.25 - -[limit] -context = 400_000 -output = 128_000 +cache_read = 0.02 [modalities] -input = ["text"] -output = ["text"] +input = ["text", "image", "pdf"] [provider] npm = "@ai-sdk/openai" diff --git a/providers/zenmux/models/openai/gpt-5.4-pro.toml b/providers/zenmux/models/openai/gpt-5.4-pro.toml index 2a4733d120e..f394261d27c 100644 --- a/providers/zenmux/models/openai/gpt-5.4-pro.toml +++ b/providers/zenmux/models/openai/gpt-5.4-pro.toml @@ -1,26 +1,22 @@ -name = "GPT-5.4 Pro" -description = "Frontier GPT model for professional reasoning, coding, and multimodal work" -release_date = "2026-03-20" -last_updated = "2026-03-20" -attachment = true -reasoning = true -reasoning_options = [{ type = "effort", values = ["medium", "high", "xhigh"] }] -temperature = true -tool_call = true -knowledge = "2025-08-31" -open_weights = false +# Effort: reasoning_effort = medium|high|xhigh +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "openai/gpt-5.4-pro" + +[[reasoning_options]] +type = "effort" +values = ["medium", "high", "xhigh"] [cost] -input = 45.00 -output = 225.00 +input = 30 +output = 180 -[limit] -context = 1_050_000 -output = 128_000 +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 60 +output = 270 [modalities] -input = ["text", "image"] -output = ["text"] +input = ["text", "image", "pdf"] [provider] npm = "@ai-sdk/openai" diff --git a/providers/zenmux/models/openai/gpt-5.4.toml b/providers/zenmux/models/openai/gpt-5.4.toml index e728524101d..7d95dfc2d79 100644 --- a/providers/zenmux/models/openai/gpt-5.4.toml +++ b/providers/zenmux/models/openai/gpt-5.4.toml @@ -1,26 +1,21 @@ -name = "GPT-5.4" -description = "Frontier GPT model for professional reasoning, coding, and multimodal work" -release_date = "2026-03-20" -last_updated = "2026-03-20" -attachment = true -reasoning = true -reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] -temperature = true -tool_call = true -knowledge = "2025-08-31" -open_weights = false +# Effort: reasoning_effort = none|low|medium|high|xhigh +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "openai/gpt-5.4" -[cost] -input = 3.75 -output = 18.75 +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh"] -[limit] -context = 1_050_000 -output = 128_000 +[cost] +input = 2.5 +output = 15 +cache_read = 0.25 -[modalities] -input = ["text", "image"] -output = ["text"] +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 5 +output = 22.5 +cache_read = 0.5 [provider] npm = "@ai-sdk/openai" diff --git a/providers/zenmux/models/openai/gpt-5.5-pro.toml b/providers/zenmux/models/openai/gpt-5.5-pro.toml index 523edc05baa..00568d94c9b 100644 --- a/providers/zenmux/models/openai/gpt-5.5-pro.toml +++ b/providers/zenmux/models/openai/gpt-5.5-pro.toml @@ -1,5 +1,10 @@ +# Effort: reasoning_effort = medium|high|xhigh +# https://zenmux.ai/docs/guide/advanced/reasoning.html base_model = "openai/gpt-5.5-pro" -reasoning_options = [{ type = "effort", values = ["medium", "high", "xhigh"] }] + +[[reasoning_options]] +type = "effort" +values = ["medium", "high", "xhigh"] [cost] input = 30 diff --git a/providers/zenmux/models/openai/gpt-5.5.toml b/providers/zenmux/models/openai/gpt-5.5.toml index 0871b3e9a6d..655ffa5169a 100644 --- a/providers/zenmux/models/openai/gpt-5.5.toml +++ b/providers/zenmux/models/openai/gpt-5.5.toml @@ -1,5 +1,10 @@ +# Effort: reasoning_effort = none|low|medium|high|xhigh +# https://zenmux.ai/docs/guide/advanced/reasoning.html base_model = "openai/gpt-5.5" -reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh"] [cost] input = 5 diff --git a/providers/zenmux/models/openai/gpt-5.6-luna.toml b/providers/zenmux/models/openai/gpt-5.6-luna.toml index 89bfa05d30e..d17ca8c99ae 100644 --- a/providers/zenmux/models/openai/gpt-5.6-luna.toml +++ b/providers/zenmux/models/openai/gpt-5.6-luna.toml @@ -1,15 +1,20 @@ +# Effort: reasoning_effort = none|low|medium|high|xhigh|max +# https://zenmux.ai/docs/guide/advanced/reasoning.html base_model = "openai/gpt-5.6-luna" -reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh", "max"] }] + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh", "max"] [cost] -input = 1.00 -output = 6.00 -cache_read = 0.10 -cache_write = 1.25 +input = 0.2 +output = 1.2 +cache_read = 0.02 +cache_write = 0.25 [[cost.tiers]] -tier = { size = 272_000 } -input = 2.00 -output = 9.00 -cache_read = 0.20 -cache_write = 2.50 +tier = { type = "context", size = 272_000 } +input = 0.4 +output = 1.8 +cache_read = 0.04 +cache_write = 0.5 diff --git a/providers/zenmux/models/openai/gpt-5.6-sol.toml b/providers/zenmux/models/openai/gpt-5.6-sol.toml index c2917934cd6..c07bee7b579 100644 --- a/providers/zenmux/models/openai/gpt-5.6-sol.toml +++ b/providers/zenmux/models/openai/gpt-5.6-sol.toml @@ -1,15 +1,20 @@ +# Effort: reasoning_effort = none|low|medium|high|xhigh|max +# https://zenmux.ai/docs/guide/advanced/reasoning.html base_model = "openai/gpt-5.6-sol" -reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh", "max"] }] + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh", "max"] [cost] -input = 5.00 -output = 30.00 -cache_read = 0.50 -cache_write = 6.25 +input = 4 +output = 20 +cache_read = 0.4 +cache_write = 5 [[cost.tiers]] -tier = { size = 272_000 } -input = 10.00 -output = 45.00 -cache_read = 1.00 -cache_write = 12.50 +tier = { type = "context", size = 272_000 } +input = 8 +output = 30 +cache_read = 0.8 +cache_write = 10 diff --git a/providers/zenmux/models/openai/gpt-5.6-terra.toml b/providers/zenmux/models/openai/gpt-5.6-terra.toml index 6e8e33b9d70..c41799a88d9 100644 --- a/providers/zenmux/models/openai/gpt-5.6-terra.toml +++ b/providers/zenmux/models/openai/gpt-5.6-terra.toml @@ -1,15 +1,20 @@ +# Effort: reasoning_effort = none|low|medium|high|xhigh|max +# https://zenmux.ai/docs/guide/advanced/reasoning.html base_model = "openai/gpt-5.6-terra" -reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh", "max"] }] + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh", "max"] [cost] -input = 2.50 -output = 15.00 -cache_read = 0.25 -cache_write = 3.125 +input = 2 +output = 12 +cache_read = 0.2 +cache_write = 2.5 [[cost.tiers]] -tier = { size = 272_000 } -input = 5.00 -output = 22.50 -cache_read = 0.50 -cache_write = 6.25 +tier = { type = "context", size = 272_000 } +input = 4 +output = 18 +cache_read = 0.4 +cache_write = 5 diff --git a/providers/zenmux/models/openai/gpt-5.toml b/providers/zenmux/models/openai/gpt-5.toml index eff8cd1dd0d..52789e8112a 100644 --- a/providers/zenmux/models/openai/gpt-5.toml +++ b/providers/zenmux/models/openai/gpt-5.toml @@ -1,27 +1,18 @@ -name = "GPT-5" -description = "GPT model for general reasoning, writing, coding, and tool-assisted tasks" -release_date = "2025-08-07" -last_updated = "2025-08-07" -attachment = true -reasoning = true -reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false +# Effort: reasoning_effort = minimal|low|medium|high +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "openai/gpt-5" + +[[reasoning_options]] +type = "effort" +values = ["minimal", "low", "medium", "high"] [cost] input = 1.25 -output = 10.00 -cache_read = 0.12 - -[limit] -context = 400_000 -output = 64_000 +output = 10 +cache_read = 0.125 [modalities] input = ["text", "image", "pdf"] -output = ["text"] [provider] npm = "@ai-sdk/openai" diff --git a/providers/zenmux/models/openai/gpt-6-astra.toml b/providers/zenmux/models/openai/gpt-6-astra.toml new file mode 100644 index 00000000000..88b339d94ce --- /dev/null +++ b/providers/zenmux/models/openai/gpt-6-astra.toml @@ -0,0 +1,20 @@ +# Effort: reasoning_effort = low|medium|high|xhigh|max +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "openai/gpt-6-astra" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high", "xhigh", "max"] + +[cost] +input = 10 +output = 50 +cache_read = 1 +cache_write = 12.5 + +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 20 +output = 75 +cache_read = 2 +cache_write = 25 diff --git a/providers/zenmux/models/openai/gpt-image-1.5.toml b/providers/zenmux/models/openai/gpt-image-1.5.toml new file mode 100644 index 00000000000..86f1dc8d549 --- /dev/null +++ b/providers/zenmux/models/openai/gpt-image-1.5.toml @@ -0,0 +1,12 @@ +base_model = "openai/gpt-image-1.5" + +[cost] +input = 5 +output = 32 +cache_read = 1.25 + +[limit] +context = 10_000 + +[modalities] +output = ["image"] diff --git a/providers/zenmux/models/openai/gpt-image-2.5-flare.toml b/providers/zenmux/models/openai/gpt-image-2.5-flare.toml new file mode 100644 index 00000000000..78d13aa01e2 --- /dev/null +++ b/providers/zenmux/models/openai/gpt-image-2.5-flare.toml @@ -0,0 +1,10 @@ +base_model = "openai/gpt-image-2.5-flare" +name = "GPT-Image-2.5-Flare" + +[cost] +input = 5 +output = 30 +cache_read = 1.25 + +[limit] +context = 10_000 diff --git a/providers/zenmux/models/openai/gpt-image-2.5-sunburst.toml b/providers/zenmux/models/openai/gpt-image-2.5-sunburst.toml new file mode 100644 index 00000000000..1370ade1ef9 --- /dev/null +++ b/providers/zenmux/models/openai/gpt-image-2.5-sunburst.toml @@ -0,0 +1,10 @@ +base_model = "openai/gpt-image-2.5-sunburst" +name = "GPT-Image-2.5-Sunburst" + +[cost] +input = 5 +output = 30 +cache_read = 1.25 + +[limit] +context = 10_000 diff --git a/providers/zenmux/models/openai/gpt-image-2.toml b/providers/zenmux/models/openai/gpt-image-2.toml new file mode 100644 index 00000000000..a1a10f4035c --- /dev/null +++ b/providers/zenmux/models/openai/gpt-image-2.toml @@ -0,0 +1,9 @@ +base_model = "openai/gpt-image-2" + +[cost] +input = 5 +output = 30 +cache_read = 1.25 + +[limit] +context = 10_000 diff --git a/providers/zenmux/models/openai/gpt-transcribe.toml b/providers/zenmux/models/openai/gpt-transcribe.toml new file mode 100644 index 00000000000..186937451f3 --- /dev/null +++ b/providers/zenmux/models/openai/gpt-transcribe.toml @@ -0,0 +1 @@ +base_model = "openai/gpt-transcribe" diff --git a/providers/zenmux/models/openai/o4-mini.toml b/providers/zenmux/models/openai/o4-mini.toml new file mode 100644 index 00000000000..5c4b20a0261 --- /dev/null +++ b/providers/zenmux/models/openai/o4-mini.toml @@ -0,0 +1,16 @@ +# Effort: reasoning_effort = low|medium|high +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "openai/o4-mini" +name = "o4 Mini" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high"] + +[cost] +input = 1.1 +output = 4.4 +cache_read = 0.275 + +[modalities] +input = ["image", "text", "pdf"] diff --git a/providers/zenmux/models/openai/text-embedding-3-large.toml b/providers/zenmux/models/openai/text-embedding-3-large.toml new file mode 100644 index 00000000000..f8aa62fae78 --- /dev/null +++ b/providers/zenmux/models/openai/text-embedding-3-large.toml @@ -0,0 +1,10 @@ +base_model = "openai/text-embedding-3-large" +name = "Text Embedding 3 Large" + +[cost] +input = 0.13 +output = 0 + +[limit] +context = 8_192 +output = 8_192 diff --git a/providers/zenmux/models/openai/text-embedding-3-small.toml b/providers/zenmux/models/openai/text-embedding-3-small.toml new file mode 100644 index 00000000000..e500102118a --- /dev/null +++ b/providers/zenmux/models/openai/text-embedding-3-small.toml @@ -0,0 +1,10 @@ +base_model = "openai/text-embedding-3-small" +name = "Text Embedding 3 Small" + +[cost] +input = 0.02 +output = 0 + +[limit] +context = 8_192 +output = 8_192 From 31d47898e06944878e8f7e423f6931cd5eeb6681 Mon Sep 17 00:00:00 2001 From: Alcatraz-Zhang Date: Sat, 19 Sep 2026 06:29:41 +0800 Subject: [PATCH 2/5] fix(zenmux): remove redundant R1 override --- providers/zenmux/models/deepseek/deepseek-r1-0528.toml | 1 - 1 file changed, 1 deletion(-) diff --git a/providers/zenmux/models/deepseek/deepseek-r1-0528.toml b/providers/zenmux/models/deepseek/deepseek-r1-0528.toml index d3bb3ba1179..3dba671e91b 100644 --- a/providers/zenmux/models/deepseek/deepseek-r1-0528.toml +++ b/providers/zenmux/models/deepseek/deepseek-r1-0528.toml @@ -1,5 +1,4 @@ base_model = "deepseek/deepseek-r1-0528" -structured_output = true reasoning_options = [] [cost] From 6921d68216af0b3b376f82a182f3378d08faff4b Mon Sep 17 00:00:00 2001 From: Alcatraz-Zhang Date: Sat, 19 Sep 2026 06:42:59 +0800 Subject: [PATCH 3/5] fix(zenmux): address catalog review findings --- .../zenmux/models/anthropic/claude-sonnet-5.toml | 3 +++ .../zenmux/models/deepseek/deepseek-chat-v3.1.toml | 3 +++ providers/zenmux/models/deepseek/deepseek-v3.2.toml | 2 ++ .../zenmux/models/deepseek/deepseek-v4.1-flash.toml | 12 ++++++++++++ .../zenmux/models/google/gemini-2.5-flash-image.toml | 3 +++ .../zenmux/models/google/gemini-3-pro-image.toml | 3 +++ .../zenmux/models/google/gemma-4-26b-a4b-it.toml | 3 +++ providers/zenmux/models/google/gemma-4-31b-it.toml | 3 +++ .../models/google/veo-3.1-fast-generate-001.toml | 3 +++ .../zenmux/models/google/veo-3.1-generate-001.toml | 3 +++ .../models/google/veo-3.1-lite-generate-001.toml | 3 +++ providers/zenmux/models/openai/chat-latest.toml | 3 +++ providers/zenmux/models/openai/gpt-5-codex.toml | 3 +++ providers/zenmux/models/openai/gpt-transcribe.toml | 3 +++ .../zenmux/models/openai/text-embedding-3-large.toml | 1 - .../zenmux/models/openai/text-embedding-3-small.toml | 1 - 16 files changed, 50 insertions(+), 2 deletions(-) diff --git a/providers/zenmux/models/anthropic/claude-sonnet-5.toml b/providers/zenmux/models/anthropic/claude-sonnet-5.toml index 265d0e0b7ea..2421fc9dcfd 100644 --- a/providers/zenmux/models/anthropic/claude-sonnet-5.toml +++ b/providers/zenmux/models/anthropic/claude-sonnet-5.toml @@ -1,6 +1,9 @@ # Toggle: thinking.type = enabled|disabled # Effort: reasoning_effort = low|medium|high|xhigh|max +# ZenMux's 2026-09-19 OpenAI- and Anthropic-compatible model lists both cap +# this hosted route at 64,000 output tokens. # https://zenmux.ai/docs/guide/advanced/reasoning.html +# https://zenmux.ai/docs/api/anthropic/anthropic-list-models.html base_model = "anthropic/claude-sonnet-5" [[reasoning_options]] diff --git a/providers/zenmux/models/deepseek/deepseek-chat-v3.1.toml b/providers/zenmux/models/deepseek/deepseek-chat-v3.1.toml index b18cf81d588..dd9b2f26e3c 100644 --- a/providers/zenmux/models/deepseek/deepseek-chat-v3.1.toml +++ b/providers/zenmux/models/deepseek/deepseek-chat-v3.1.toml @@ -1,3 +1,6 @@ +# ZenMux's 2026-09-19 model lists expose this as the non-thinking chat route, +# so reasoning=false is an intentional host override. +# https://zenmux.ai/docs/api/openai/openai-list-models.html base_model = "deepseek/deepseek-v3.1" name = "DeepSeek V3.1" reasoning = false diff --git a/providers/zenmux/models/deepseek/deepseek-v3.2.toml b/providers/zenmux/models/deepseek/deepseek-v3.2.toml index e22843debba..ed7bc41bf4d 100644 --- a/providers/zenmux/models/deepseek/deepseek-v3.2.toml +++ b/providers/zenmux/models/deepseek/deepseek-v3.2.toml @@ -1,5 +1,7 @@ # Toggle: thinking.type = enabled|disabled +# ZenMux's 2026-09-19 catalog caps this route at 8,000 output tokens. # https://zenmux.ai/docs/guide/advanced/reasoning.html +# https://zenmux.ai/docs/api/openai/openai-list-models.html base_model = "deepseek/deepseek-v3.2" [[reasoning_options]] diff --git a/providers/zenmux/models/deepseek/deepseek-v4.1-flash.toml b/providers/zenmux/models/deepseek/deepseek-v4.1-flash.toml index 95992ef7deb..72bf637e75f 100644 --- a/providers/zenmux/models/deepseek/deepseek-v4.1-flash.toml +++ b/providers/zenmux/models/deepseek/deepseek-v4.1-flash.toml @@ -1,11 +1,23 @@ # Toggle: thinking.type = enabled|disabled # Effort: reasoning_effort = low|high|max +# ZenMux also publishes time-of-day prices. The catalog records the discounted +# scalar/base rate exposed by the 2026-09-19 model APIs; peak bands are not +# representable by the catalog's context-tier schema. # https://zenmux.ai/docs/guide/advanced/reasoning.html +# https://zenmux.ai/docs/api/openai/openai-list-models.html base_model = "deepseek/deepseek-v4.1-flash" +[interleaved] +field = "reasoning_content" + [[reasoning_options]] type = "toggle" [[reasoning_options]] type = "effort" values = ["low", "high", "max"] + +[cost] +input = 0.075 +output = 0.3 +cache_read = 0.0015 diff --git a/providers/zenmux/models/google/gemini-2.5-flash-image.toml b/providers/zenmux/models/google/gemini-2.5-flash-image.toml index 28e69e2e250..031c7179751 100644 --- a/providers/zenmux/models/google/gemini-2.5-flash-image.toml +++ b/providers/zenmux/models/google/gemini-2.5-flash-image.toml @@ -1,3 +1,6 @@ +# ZenMux's 2026-09-19 catalog and Google-compatible model list both report +# thinking disabled for this image route, so this is an intentional host override. +# https://zenmux.ai/docs/api/vertexai/google-list-models.html base_model = "google/gemini-2.5-flash-image" name = "Gemini 2.5 Flash Image (Nano Banana)" reasoning = false diff --git a/providers/zenmux/models/google/gemini-3-pro-image.toml b/providers/zenmux/models/google/gemini-3-pro-image.toml index 9c317943c12..223dc51f4f6 100644 --- a/providers/zenmux/models/google/gemini-3-pro-image.toml +++ b/providers/zenmux/models/google/gemini-3-pro-image.toml @@ -1,3 +1,6 @@ +# ZenMux's 2026-09-19 catalog and Google-compatible model list both report +# thinking disabled for this route, so this is an intentional host override. +# https://zenmux.ai/docs/api/vertexai/google-list-models.html base_model = "google/gemini-3-pro-image" name = "Nano Banana Pro (Gemini 3 Pro Image)" reasoning = false diff --git a/providers/zenmux/models/google/gemma-4-26b-a4b-it.toml b/providers/zenmux/models/google/gemma-4-26b-a4b-it.toml index 7583789d90d..3859ae82482 100644 --- a/providers/zenmux/models/google/gemma-4-26b-a4b-it.toml +++ b/providers/zenmux/models/google/gemma-4-26b-a4b-it.toml @@ -1,3 +1,6 @@ +# ZenMux's 2026-09-19 protocol model lists report reasoning/thinking disabled +# for this route, so this is an intentional host override. +# https://zenmux.ai/docs/api/openai/openai-list-models.html base_model = "google/gemma-4-26b-a4b-it" reasoning = false diff --git a/providers/zenmux/models/google/gemma-4-31b-it.toml b/providers/zenmux/models/google/gemma-4-31b-it.toml index d8cb297dbed..f361c53760c 100644 --- a/providers/zenmux/models/google/gemma-4-31b-it.toml +++ b/providers/zenmux/models/google/gemma-4-31b-it.toml @@ -1,3 +1,6 @@ +# ZenMux's 2026-09-19 catalog reports reasoning disabled and no public token +# price for this non-free route; cost is intentionally omitted rather than set to zero. +# https://zenmux.ai/docs/api/openai/openai-list-models.html base_model = "google/gemma-4-31b-it" reasoning = false diff --git a/providers/zenmux/models/google/veo-3.1-fast-generate-001.toml b/providers/zenmux/models/google/veo-3.1-fast-generate-001.toml index 9d0f5db6876..8b51bd98b0b 100644 --- a/providers/zenmux/models/google/veo-3.1-fast-generate-001.toml +++ b/providers/zenmux/models/google/veo-3.1-fast-generate-001.toml @@ -1,3 +1,6 @@ +# ZenMux lists a 100,000-token host context and resolution-dependent per-second +# video prices; the token-only cost schema cannot represent those prices. +# https://zenmux.ai/docs/api/vertexai/google-list-models.html base_model = "google/veo-3.1-fast-generate-001" [limit] diff --git a/providers/zenmux/models/google/veo-3.1-generate-001.toml b/providers/zenmux/models/google/veo-3.1-generate-001.toml index bde05604591..8e87d322916 100644 --- a/providers/zenmux/models/google/veo-3.1-generate-001.toml +++ b/providers/zenmux/models/google/veo-3.1-generate-001.toml @@ -1,3 +1,6 @@ +# ZenMux lists a 100,000-token host context and resolution-dependent per-second +# video prices; the token-only cost schema cannot represent those prices. +# https://zenmux.ai/docs/api/vertexai/google-list-models.html base_model = "google/veo-3.1-generate-001" [limit] diff --git a/providers/zenmux/models/google/veo-3.1-lite-generate-001.toml b/providers/zenmux/models/google/veo-3.1-lite-generate-001.toml index 440fc75aabf..2841b7080af 100644 --- a/providers/zenmux/models/google/veo-3.1-lite-generate-001.toml +++ b/providers/zenmux/models/google/veo-3.1-lite-generate-001.toml @@ -1,3 +1,6 @@ +# ZenMux lists a 100,000-token host context and resolution-dependent per-second +# video prices; the token-only cost schema cannot represent those prices. +# https://zenmux.ai/docs/api/vertexai/google-list-models.html base_model = "google/veo-3.1-lite-generate-001" [limit] diff --git a/providers/zenmux/models/openai/chat-latest.toml b/providers/zenmux/models/openai/chat-latest.toml index 14bccd2647d..fe7f147f529 100644 --- a/providers/zenmux/models/openai/chat-latest.toml +++ b/providers/zenmux/models/openai/chat-latest.toml @@ -1,3 +1,6 @@ +# ZenMux's 2026-09-19 OpenAI- and Anthropic-compatible model lists both report +# reasoning disabled for this alias, so this is an intentional host override. +# https://zenmux.ai/docs/api/openai/openai-list-models.html base_model = "openai/gpt-5.5-instant" name = "Chat Latest (GPT-5.5 Instant)" reasoning = false diff --git a/providers/zenmux/models/openai/gpt-5-codex.toml b/providers/zenmux/models/openai/gpt-5-codex.toml index 3f0d2687e68..29d05b941ee 100644 --- a/providers/zenmux/models/openai/gpt-5-codex.toml +++ b/providers/zenmux/models/openai/gpt-5-codex.toml @@ -1,5 +1,8 @@ # Effort: reasoning_effort = low|medium|high +# ZenMux's 2026-09-19 model lists advertise text and image input for this route, +# so attachment=true is an intentional host capability override. # https://zenmux.ai/docs/guide/advanced/reasoning.html +# https://zenmux.ai/docs/api/openai/openai-list-models.html base_model = "openai/gpt-5-codex" name = "GPT-5 Codex" attachment = true diff --git a/providers/zenmux/models/openai/gpt-transcribe.toml b/providers/zenmux/models/openai/gpt-transcribe.toml index 186937451f3..bf645f3a144 100644 --- a/providers/zenmux/models/openai/gpt-transcribe.toml +++ b/providers/zenmux/models/openai/gpt-transcribe.toml @@ -1 +1,4 @@ +# ZenMux bills this audio route at USD 0.000075 per second, not per token; +# the token-only cost schema cannot represent it, so cost is intentionally omitted. +# https://zenmux.ai/docs/api/openai/openai-list-models.html base_model = "openai/gpt-transcribe" diff --git a/providers/zenmux/models/openai/text-embedding-3-large.toml b/providers/zenmux/models/openai/text-embedding-3-large.toml index f8aa62fae78..ada8d70e4a3 100644 --- a/providers/zenmux/models/openai/text-embedding-3-large.toml +++ b/providers/zenmux/models/openai/text-embedding-3-large.toml @@ -7,4 +7,3 @@ output = 0 [limit] context = 8_192 -output = 8_192 diff --git a/providers/zenmux/models/openai/text-embedding-3-small.toml b/providers/zenmux/models/openai/text-embedding-3-small.toml index e500102118a..f0b83d58461 100644 --- a/providers/zenmux/models/openai/text-embedding-3-small.toml +++ b/providers/zenmux/models/openai/text-embedding-3-small.toml @@ -7,4 +7,3 @@ output = 0 [limit] context = 8_192 -output = 8_192 From 0ff06222affcdd4b6ed8b0501d35a1517e4a1c94 Mon Sep 17 00:00:00 2001 From: Alcatraz-Zhang Date: Sat, 19 Sep 2026 06:53:02 +0800 Subject: [PATCH 4/5] fix(zenmux): clarify reviewed host overrides --- providers/zenmux/models/anthropic/claude-opus-4.5.toml | 3 +++ providers/zenmux/models/deepseek/deepseek-v4.1-flash.toml | 7 ++++--- .../zenmux/models/google/gemini-3.1-flash-lite-image.toml | 2 ++ providers/zenmux/models/google/gemini-3.5-flash.toml | 3 +++ providers/zenmux/models/google/gemini-embedding-2.toml | 3 --- providers/zenmux/models/moonshotai/kimi-k2.8-preview.toml | 5 +++++ providers/zenmux/models/openai/gpt-5-pro.toml | 3 +++ providers/zenmux/models/openai/text-embedding-3-large.toml | 3 +++ providers/zenmux/models/openai/text-embedding-3-small.toml | 3 +++ 9 files changed, 26 insertions(+), 6 deletions(-) diff --git a/providers/zenmux/models/anthropic/claude-opus-4.5.toml b/providers/zenmux/models/anthropic/claude-opus-4.5.toml index eac91936470..89ff965010b 100644 --- a/providers/zenmux/models/anthropic/claude-opus-4.5.toml +++ b/providers/zenmux/models/anthropic/claude-opus-4.5.toml @@ -1,6 +1,9 @@ # Effort: reasoning_effort = low|medium|high # Budget: thinking.budget_tokens (integer) +# ZenMux's 2026-09-19 catalog and Anthropic-compatible model list explicitly +# report max_completion_tokens=32000 for this hosted route. # https://zenmux.ai/docs/guide/advanced/reasoning.html +# https://zenmux.ai/docs/api/anthropic/anthropic-list-models.html base_model = "anthropic/claude-opus-4-5" name = "Claude Opus 4.5" structured_output = true diff --git a/providers/zenmux/models/deepseek/deepseek-v4.1-flash.toml b/providers/zenmux/models/deepseek/deepseek-v4.1-flash.toml index 72bf637e75f..5a55042d563 100644 --- a/providers/zenmux/models/deepseek/deepseek-v4.1-flash.toml +++ b/providers/zenmux/models/deepseek/deepseek-v4.1-flash.toml @@ -1,8 +1,9 @@ # Toggle: thinking.type = enabled|disabled # Effort: reasoning_effort = low|high|max -# ZenMux also publishes time-of-day prices. The catalog records the discounted -# scalar/base rate exposed by the 2026-09-19 model APIs; peak bands are not -# representable by the catalog's context-tier schema. +# ZenMux's 2026-09-19 protocol lists publish discounted bands of USD 0.075 input, +# 0.30 output, and 0.0015 cache-read per million tokens. Higher time bands of +# 0.15/0.60/0.003 are intentionally omitted because the schema only models +# context tiers, not time-of-day tiers. # https://zenmux.ai/docs/guide/advanced/reasoning.html # https://zenmux.ai/docs/api/openai/openai-list-models.html base_model = "deepseek/deepseek-v4.1-flash" diff --git a/providers/zenmux/models/google/gemini-3.1-flash-lite-image.toml b/providers/zenmux/models/google/gemini-3.1-flash-lite-image.toml index f36b469aec8..265bf72c522 100644 --- a/providers/zenmux/models/google/gemini-3.1-flash-lite-image.toml +++ b/providers/zenmux/models/google/gemini-3.1-flash-lite-image.toml @@ -1,5 +1,7 @@ # Effort: reasoning_effort = minimal|high +# ZenMux's 2026-09-19 catalog explicitly reports max_completion_tokens=32768. # https://zenmux.ai/docs/guide/advanced/reasoning.html +# https://zenmux.ai/docs/api/vertexai/google-list-models.html base_model = "google/gemini-3.1-flash-lite-image" name = "Nano Banana 2 Lite (Gemini 3.1 Flash-Lite Image)" diff --git a/providers/zenmux/models/google/gemini-3.5-flash.toml b/providers/zenmux/models/google/gemini-3.5-flash.toml index 370405193ea..79e2b3b0505 100644 --- a/providers/zenmux/models/google/gemini-3.5-flash.toml +++ b/providers/zenmux/models/google/gemini-3.5-flash.toml @@ -1,5 +1,8 @@ # Effort: reasoning_effort = minimal|low|medium|high +# ZenMux's 2026-09-19 catalog explicitly lists the 1-hour cache-write token +# band at USD 0 per million tokens; this is a published free price, not a sentinel. # https://zenmux.ai/docs/guide/advanced/reasoning.html +# https://zenmux.ai/docs/api/vertexai/google-list-models.html base_model = "google/gemini-3.5-flash" [[reasoning_options]] diff --git a/providers/zenmux/models/google/gemini-embedding-2.toml b/providers/zenmux/models/google/gemini-embedding-2.toml index da5123f4089..ea461ab7299 100644 --- a/providers/zenmux/models/google/gemini-embedding-2.toml +++ b/providers/zenmux/models/google/gemini-embedding-2.toml @@ -3,6 +3,3 @@ base_model = "google/gemini-embedding-2" [cost] input = 0.2 output = 0 - -[limit] -output = 0 diff --git a/providers/zenmux/models/moonshotai/kimi-k2.8-preview.toml b/providers/zenmux/models/moonshotai/kimi-k2.8-preview.toml index c4e4ca57134..7c8c6d010c5 100644 --- a/providers/zenmux/models/moonshotai/kimi-k2.8-preview.toml +++ b/providers/zenmux/models/moonshotai/kimi-k2.8-preview.toml @@ -1,8 +1,13 @@ # Toggle: thinking.type = enabled|disabled|adaptive # Effort: output_config.effort = low|high|max +# ZenMux's 2026-09-19 catalog explicitly reports max_completion_tokens=1048576. # https://zenmux.ai/docs/guide/advanced/reasoning.html +# https://zenmux.ai/docs/api/openai/openai-list-models.html base_model = "moonshotai/kimi-k2.8-preview" +[interleaved] +field = "reasoning_content" + [[reasoning_options]] type = "toggle" diff --git a/providers/zenmux/models/openai/gpt-5-pro.toml b/providers/zenmux/models/openai/gpt-5-pro.toml index 33dc045f0ea..12c316f426a 100644 --- a/providers/zenmux/models/openai/gpt-5-pro.toml +++ b/providers/zenmux/models/openai/gpt-5-pro.toml @@ -1,5 +1,8 @@ # Effort: reasoning_effort = high +# ZenMux's 2026-09-19 catalog and OpenAI-compatible model list explicitly +# report max_completion_tokens=128000 for this hosted route. # https://zenmux.ai/docs/guide/advanced/reasoning.html +# https://zenmux.ai/docs/api/openai/openai-list-models.html base_model = "openai/gpt-5-pro" [[reasoning_options]] diff --git a/providers/zenmux/models/openai/text-embedding-3-large.toml b/providers/zenmux/models/openai/text-embedding-3-large.toml index ada8d70e4a3..bac23d99196 100644 --- a/providers/zenmux/models/openai/text-embedding-3-large.toml +++ b/providers/zenmux/models/openai/text-embedding-3-large.toml @@ -1,3 +1,6 @@ +# ZenMux's 2026-09-19 catalog reports context_length=8192. The inherited +# limit.output remains the lab embedding dimension rather than that token limit. +# https://zenmux.ai/docs/api/openai/openai-list-models.html base_model = "openai/text-embedding-3-large" name = "Text Embedding 3 Large" diff --git a/providers/zenmux/models/openai/text-embedding-3-small.toml b/providers/zenmux/models/openai/text-embedding-3-small.toml index f0b83d58461..7db9ea0c48a 100644 --- a/providers/zenmux/models/openai/text-embedding-3-small.toml +++ b/providers/zenmux/models/openai/text-embedding-3-small.toml @@ -1,3 +1,6 @@ +# ZenMux's 2026-09-19 catalog reports context_length=8192. The inherited +# limit.output remains the lab embedding dimension rather than that token limit. +# https://zenmux.ai/docs/api/openai/openai-list-models.html base_model = "openai/text-embedding-3-small" name = "Text Embedding 3 Small" From aedca5a87df56c6ed304e8b2a1ada5271dfd72a1 Mon Sep 17 00:00:00 2001 From: Alcatraz-Zhang Date: Sat, 19 Sep 2026 07:14:03 +0800 Subject: [PATCH 5/5] fix(zenmux): document protocol-specific deltas --- providers/zenmux/models/deepseek/deepseek-v4-flash.toml | 3 +++ providers/zenmux/models/deepseek/deepseek-v4-pro.toml | 3 +++ providers/zenmux/models/google/gemini-2.5-flash-image.toml | 5 +++-- providers/zenmux/models/google/gemini-3-pro-image.toml | 5 +++-- providers/zenmux/models/google/gemma-4-26b-a4b-it.toml | 5 +++-- providers/zenmux/models/google/gemma-4-31b-it.toml | 5 +++-- .../zenmux/models/moonshotai/kimi-k2.7-code-highspeed.toml | 5 +++++ providers/zenmux/models/moonshotai/kimi-k2.7-code.toml | 5 +++++ 8 files changed, 28 insertions(+), 8 deletions(-) diff --git a/providers/zenmux/models/deepseek/deepseek-v4-flash.toml b/providers/zenmux/models/deepseek/deepseek-v4-flash.toml index 6c27fb4c758..53a23284856 100644 --- a/providers/zenmux/models/deepseek/deepseek-v4-flash.toml +++ b/providers/zenmux/models/deepseek/deepseek-v4-flash.toml @@ -1,6 +1,9 @@ # Toggle: thinking.type = enabled|disabled # Effort: reasoning_effort = low|high|max +# ZenMux's undated deepseek-v4-flash slug is explicitly displayed as the 0731 +# snapshot in its 2026-09-19 catalog, so the dated base name is intentional. # https://zenmux.ai/docs/guide/advanced/reasoning.html +# https://zenmux.ai/docs/api/openai/openai-list-models.html base_model = "deepseek/deepseek-v4-flash-0731" [interleaved] diff --git a/providers/zenmux/models/deepseek/deepseek-v4-pro.toml b/providers/zenmux/models/deepseek/deepseek-v4-pro.toml index 6026f1c7ac2..83402b4b635 100644 --- a/providers/zenmux/models/deepseek/deepseek-v4-pro.toml +++ b/providers/zenmux/models/deepseek/deepseek-v4-pro.toml @@ -1,6 +1,9 @@ # Toggle: thinking.type = enabled|disabled # Effort: reasoning_effort = high|max +# ZenMux's undated deepseek-v4-pro slug is explicitly displayed as the 0813 +# snapshot in its 2026-09-19 catalog, so the dated base name is intentional. # https://zenmux.ai/docs/guide/advanced/reasoning.html +# https://zenmux.ai/docs/api/openai/openai-list-models.html base_model = "deepseek/deepseek-v4-pro-0813" [interleaved] diff --git a/providers/zenmux/models/google/gemini-2.5-flash-image.toml b/providers/zenmux/models/google/gemini-2.5-flash-image.toml index 031c7179751..fe2240fdee2 100644 --- a/providers/zenmux/models/google/gemini-2.5-flash-image.toml +++ b/providers/zenmux/models/google/gemini-2.5-flash-image.toml @@ -1,5 +1,6 @@ -# ZenMux's 2026-09-19 catalog and Google-compatible model list both report -# thinking disabled for this image route, so this is an intentional host override. +# ZenMux's 2026-09-19 page catalog has supports_reasoning=0 and its live +# Google-compatible model-list response has thinking=false for this exact id. +# The lab's reasoning surface is therefore intentionally disabled on this host. # https://zenmux.ai/docs/api/vertexai/google-list-models.html base_model = "google/gemini-2.5-flash-image" name = "Gemini 2.5 Flash Image (Nano Banana)" diff --git a/providers/zenmux/models/google/gemini-3-pro-image.toml b/providers/zenmux/models/google/gemini-3-pro-image.toml index 223dc51f4f6..56c1d10d15c 100644 --- a/providers/zenmux/models/google/gemini-3-pro-image.toml +++ b/providers/zenmux/models/google/gemini-3-pro-image.toml @@ -1,5 +1,6 @@ -# ZenMux's 2026-09-19 catalog and Google-compatible model list both report -# thinking disabled for this route, so this is an intentional host override. +# ZenMux's 2026-09-19 page catalog has supports_reasoning=0 and its live +# Google-compatible model-list response has thinking=false for this exact id. +# The lab's reasoning surface is therefore intentionally disabled on this host. # https://zenmux.ai/docs/api/vertexai/google-list-models.html base_model = "google/gemini-3-pro-image" name = "Nano Banana Pro (Gemini 3 Pro Image)" diff --git a/providers/zenmux/models/google/gemma-4-26b-a4b-it.toml b/providers/zenmux/models/google/gemma-4-26b-a4b-it.toml index 3859ae82482..a0667665d07 100644 --- a/providers/zenmux/models/google/gemma-4-26b-a4b-it.toml +++ b/providers/zenmux/models/google/gemma-4-26b-a4b-it.toml @@ -1,5 +1,6 @@ -# ZenMux's 2026-09-19 protocol model lists report reasoning/thinking disabled -# for this route, so this is an intentional host override. +# ZenMux's 2026-09-19 page catalog has supports_reasoning=0; its live OpenAI- +# and Anthropic-compatible lists have capabilities.reasoning=false, and its +# Google-compatible list has thinking=false for this exact id. # https://zenmux.ai/docs/api/openai/openai-list-models.html base_model = "google/gemma-4-26b-a4b-it" reasoning = false diff --git a/providers/zenmux/models/google/gemma-4-31b-it.toml b/providers/zenmux/models/google/gemma-4-31b-it.toml index f361c53760c..ffd6edb4574 100644 --- a/providers/zenmux/models/google/gemma-4-31b-it.toml +++ b/providers/zenmux/models/google/gemma-4-31b-it.toml @@ -1,5 +1,6 @@ -# ZenMux's 2026-09-19 catalog reports reasoning disabled and no public token -# price for this non-free route; cost is intentionally omitted rather than set to zero. +# ZenMux's 2026-09-19 page catalog has supports_reasoning=0. Its live OpenAI- +# and Anthropic-compatible entries expose no reasoning capability and no price; +# the route is non-free, so cost is intentionally omitted rather than set to zero. # https://zenmux.ai/docs/api/openai/openai-list-models.html base_model = "google/gemma-4-31b-it" reasoning = false diff --git a/providers/zenmux/models/moonshotai/kimi-k2.7-code-highspeed.toml b/providers/zenmux/models/moonshotai/kimi-k2.7-code-highspeed.toml index 66c82f75088..f115142dc64 100644 --- a/providers/zenmux/models/moonshotai/kimi-k2.7-code-highspeed.toml +++ b/providers/zenmux/models/moonshotai/kimi-k2.7-code-highspeed.toml @@ -1,7 +1,12 @@ +# Response side channel: reasoning_content +# https://zenmux.ai/docs/guide/advanced/reasoning.html base_model = "moonshotai/kimi-k2.7-code-highspeed" name = "Kimi K2.7 Code HighSpeed" reasoning_options = [] +[interleaved] +field = "reasoning_content" + [cost] input = 1.9 output = 8 diff --git a/providers/zenmux/models/moonshotai/kimi-k2.7-code.toml b/providers/zenmux/models/moonshotai/kimi-k2.7-code.toml index d7fe478db1a..684a5746f3a 100644 --- a/providers/zenmux/models/moonshotai/kimi-k2.7-code.toml +++ b/providers/zenmux/models/moonshotai/kimi-k2.7-code.toml @@ -1,6 +1,11 @@ +# Response side channel: reasoning_content +# https://zenmux.ai/docs/guide/advanced/reasoning.html base_model = "moonshotai/kimi-k2.7-code" reasoning_options = [] +[interleaved] +field = "reasoning_content" + [cost] input = 0.95 output = 4