diff --git a/models/deepseek/deepseek-r1-0528.toml b/models/deepseek/deepseek-r1-0528.toml new file mode 100644 index 00000000000..2eb59f40062 --- /dev/null +++ b/models/deepseek/deepseek-r1-0528.toml @@ -0,0 +1,29 @@ +# Sources: +# - https://huggingface.co/deepseek-ai/DeepSeek-R1-0528 +# - https://zenmux.ai/models +# Accessed 2026-09-19. +name = "DeepSeek R1 0528" +description = "May 2025 update to DeepSeek R1 with stronger reasoning, math, coding, and tool use" +family = "deepseek-thinking" +release_date = "2025-05-28" +last_updated = "2025-05-28" +attachment = false +reasoning = true +temperature = true +tool_call = true +structured_output = true +knowledge = "2024-07" +open_weights = true +license = "MIT" + +[limit] +context = 128_000 +output = 32_768 + +[modalities] +input = ["text"] +output = ["text"] + +[[weights]] +label = "Hugging Face" +url = "https://huggingface.co/deepseek-ai/DeepSeek-R1-0528" diff --git a/models/deepseek/deepseek-v3.2-exp.toml b/models/deepseek/deepseek-v3.2-exp.toml new file mode 100644 index 00000000000..756435a6304 --- /dev/null +++ b/models/deepseek/deepseek-v3.2-exp.toml @@ -0,0 +1,30 @@ +# Sources: +# - https://zenmux.ai/models +# - https://zenmux.ai/docs/api/openai/openai-list-models.html +# - https://zenmux.ai/docs/api/anthropic/anthropic-list-models.html +# - https://zenmux.ai/docs/api/vertexai/google-list-models.html +# - https://huggingface.co/deepseek-ai/DeepSeek-V3.2-Exp +# Accessed 2026-09-19. +name = "DeepSeek-V3.2-Exp" +description = "DeepSeek-V3.2-Exp is an experimental model version, serving as an intermediate step toward the next-generation architecture. Built on the foundation of V3.1-Terminus, it introduces the DeepSeek Sparse Attention (DSA) mechanism—a sparse attention mechanism designed to explore and validate the optimization of training and inference efficiency in long-context scenarios. This experimental version represents the team's continuous research on more efficient Transformer architectures, with a specific focus on improving computational efficiency when processing long text sequences. For the first time, DSA enables fine-grained sparse attention, which significantly enhances the efficiency of long-context training and inference while maintaining almost unchanged model output quality." +family = "deepseek" +release_date = "2025-09-29" +last_updated = "2025-09-29" +attachment = false +reasoning = true +temperature = true +tool_call = true +structured_output = true +open_weights = true + +[limit] +context = 163_840 +output = 65_536 + +[modalities] +input = ["text"] +output = ["text"] + +[[weights]] +label = "Hugging Face" +url = "https://huggingface.co/deepseek-ai/DeepSeek-V3.2-Exp" diff --git a/models/google/gemini-omni-1.1-flash-preview.toml b/models/google/gemini-omni-1.1-flash-preview.toml new file mode 100644 index 00000000000..08075b5e5d5 --- /dev/null +++ b/models/google/gemini-omni-1.1-flash-preview.toml @@ -0,0 +1,25 @@ +# Sources: +# - https://zenmux.ai/models +# - https://zenmux.ai/docs/api/openai/openai-list-models.html +# - https://zenmux.ai/docs/api/anthropic/anthropic-list-models.html +# - https://zenmux.ai/docs/api/vertexai/google-list-models.html +# Accessed 2026-09-19. +name = "Gemini Omni 1.1 Flash Preview" +description = "Gemini Omni 1.1 Flash (Preview) is a multimodal model designed for video, image, and text tasks. It is optimized for video generation, offering video output alongside text responses in a single model." +family = "gemini" +release_date = "2026-08-31" +last_updated = "2026-08-31" +attachment = true +reasoning = true +temperature = false +tool_call = false +structured_output = false +open_weights = false + +[limit] +context = 131_072 +output = 57_920 + +[modalities] +input = ["text", "image", "video"] +output = ["text", "video"] diff --git a/models/google/veo-3.1-fast-generate-001.toml b/models/google/veo-3.1-fast-generate-001.toml new file mode 100644 index 00000000000..9811836e922 --- /dev/null +++ b/models/google/veo-3.1-fast-generate-001.toml @@ -0,0 +1,25 @@ +# Sources: +# - https://zenmux.ai/models +# - https://zenmux.ai/docs/api/openai/openai-list-models.html +# - https://zenmux.ai/docs/api/anthropic/anthropic-list-models.html +# - https://zenmux.ai/docs/api/vertexai/google-list-models.html +# Accessed 2026-09-19. +name = "Veo 3.1 Fast" +description = "Veo 3.1 Fast is a speed-optimized variant of Google DeepMind's flagship video generation model. It is designed to generate high-quality video significantly faster and at a lower cost than the standard Veo 3.1 Quality model, making it ideal for rapid prototyping and high-volume content creation." +family = "veo" +release_date = "2025-10-15" +last_updated = "2026-01-01" +attachment = true +reasoning = false +temperature = false +tool_call = false +structured_output = false +open_weights = false + +[limit] +context = 1_024 +output = 0 + +[modalities] +input = ["text", "image", "video"] +output = ["video"] diff --git a/models/google/veo-3.1-generate-001.toml b/models/google/veo-3.1-generate-001.toml new file mode 100644 index 00000000000..0eaa229aa72 --- /dev/null +++ b/models/google/veo-3.1-generate-001.toml @@ -0,0 +1,25 @@ +# Sources: +# - https://zenmux.ai/models +# - https://zenmux.ai/docs/api/openai/openai-list-models.html +# - https://zenmux.ai/docs/api/anthropic/anthropic-list-models.html +# - https://zenmux.ai/docs/api/vertexai/google-list-models.html +# Accessed 2026-09-19. +name = "Veo 3.1" +description = "Veo 3.1 is Google's state-of-the-art model for generating high-fidelity, 8-second 720p, 1080p or 4k videos featuring stunning realism and natively generated audio. You can access this model programmatically using the Gemini API. To learn more about the available Veo model variants, see the Model Versions section." +family = "veo" +release_date = "2025-10-15" +last_updated = "2026-01" +attachment = true +reasoning = false +temperature = false +tool_call = false +structured_output = false +open_weights = false + +[limit] +context = 1_024 +output = 0 + +[modalities] +input = ["text", "image"] +output = ["video"] diff --git a/models/google/veo-3.1-lite-generate-001.toml b/models/google/veo-3.1-lite-generate-001.toml new file mode 100644 index 00000000000..3212a720350 --- /dev/null +++ b/models/google/veo-3.1-lite-generate-001.toml @@ -0,0 +1,25 @@ +# Sources: +# - https://zenmux.ai/models +# - https://zenmux.ai/docs/api/openai/openai-list-models.html +# - https://zenmux.ai/docs/api/anthropic/anthropic-list-models.html +# - https://zenmux.ai/docs/api/vertexai/google-list-models.html +# Accessed 2026-09-19. +name = "Veo 3.1 Lite" +description = "Veo 3.1 Lite is Google DeepMind's most cost-efficient AI video generation model, released on March 31, 2026. It is designed to provide professional-grade video capabilities at a significantly lower price point, making it ideal for developers and content teams who need to scale high-volume video production" +family = "veo" +release_date = "2026-03-31" +last_updated = "2026-03-31" +attachment = true +reasoning = false +temperature = false +tool_call = false +structured_output = false +open_weights = false + +[limit] +context = 1_024 +output = 0 + +[modalities] +input = ["text", "image"] +output = ["video"] diff --git a/models/openai/gpt-transcribe.toml b/models/openai/gpt-transcribe.toml new file mode 100644 index 00000000000..52686fb5a2b --- /dev/null +++ b/models/openai/gpt-transcribe.toml @@ -0,0 +1,25 @@ +# Sources: +# - https://zenmux.ai/models +# - https://zenmux.ai/docs/api/openai/openai-list-models.html +# - https://zenmux.ai/docs/api/anthropic/anthropic-list-models.html +# - https://zenmux.ai/docs/api/vertexai/google-list-models.html +# Accessed 2026-09-19. +name = "GPT Transcribe" +description = "GPT Transcribe is a high-accuracy speech-to-text model from OpenAI. It is suited for recorded audio, streamed file transcription, and committed Realtime turns, with free-form context, keyword hints, and multiple language hints for specialized terms and multilingual speech." +family = "gpt" +release_date = "2026-08-06" +last_updated = "2026-08-06" +attachment = true +reasoning = false +temperature = false +tool_call = false +structured_output = false +open_weights = false + +[limit] +context = 15_000 +output = 15_000 + +[modalities] +input = ["audio"] +output = ["text"] diff --git a/models/openai/text-embedding-3-large.toml b/models/openai/text-embedding-3-large.toml new file mode 100644 index 00000000000..943de42b401 --- /dev/null +++ b/models/openai/text-embedding-3-large.toml @@ -0,0 +1,25 @@ +# Sources: +# - https://zenmux.ai/models +# - https://zenmux.ai/docs/api/openai/openai-list-models.html +# - https://zenmux.ai/docs/api/anthropic/anthropic-list-models.html +# - https://zenmux.ai/docs/api/vertexai/google-list-models.html +# Accessed 2026-09-19. +name = "text-embedding-3-large" +description = "Embedding model for semantic search, retrieval, clustering, and ranking pipelines" +family = "text-embedding" +release_date = "2024-01-25" +last_updated = "2024-01-25" +attachment = false +reasoning = false +temperature = false +tool_call = false +knowledge = "2024-01" +open_weights = false + +[limit] +context = 8_191 +output = 3_072 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/models/openai/text-embedding-3-small.toml b/models/openai/text-embedding-3-small.toml new file mode 100644 index 00000000000..932c390266a --- /dev/null +++ b/models/openai/text-embedding-3-small.toml @@ -0,0 +1,25 @@ +# Sources: +# - https://zenmux.ai/models +# - https://zenmux.ai/docs/api/openai/openai-list-models.html +# - https://zenmux.ai/docs/api/anthropic/anthropic-list-models.html +# - https://zenmux.ai/docs/api/vertexai/google-list-models.html +# Accessed 2026-09-19. +name = "text-embedding-3-small" +description = "Embedding model for semantic search, retrieval, clustering, and ranking pipelines" +family = "text-embedding" +release_date = "2024-01-25" +last_updated = "2024-01-25" +attachment = false +reasoning = false +temperature = false +tool_call = false +knowledge = "2024-01" +open_weights = false + +[limit] +context = 8_191 +output = 1_536 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/zenmux/models/anthropic/claude-3.5-haiku.toml b/providers/zenmux/models/anthropic/claude-3.5-haiku.toml deleted file mode 100644 index c3b0a3607c8..00000000000 --- a/providers/zenmux/models/anthropic/claude-3.5-haiku.toml +++ /dev/null @@ -1,29 +0,0 @@ -name = "Claude 3.5 Haiku" -description = "Fast Claude model for responsive assistance, classification, and lightweight agents" -release_date = "2024-11-04" -last_updated = "2024-11-04" -attachment = true -reasoning = false -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false -status = "deprecated" - -[cost] -input = 0.80 -output = 4.00 -cache_read = 0.08 -cache_write = 1.00 - -[limit] -context = 200_000 -output = 64_000 - -[modalities] -input = ["text", "image"] -output = ["text"] - -[provider] -npm = "@ai-sdk/anthropic" -api = "https://zenmux.ai/api/anthropic/v1" diff --git a/providers/zenmux/models/anthropic/claude-3.7-sonnet.toml b/providers/zenmux/models/anthropic/claude-3.7-sonnet.toml deleted file mode 100644 index e1b9a104ca2..00000000000 --- a/providers/zenmux/models/anthropic/claude-3.7-sonnet.toml +++ /dev/null @@ -1,30 +0,0 @@ -name = "Claude 3.7 Sonnet" -description = "Balanced Claude model for coding, analysis, agent workflows, and cost control" -release_date = "2025-02-24" -last_updated = "2025-02-24" -attachment = true -reasoning = true -reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false -status = "deprecated" - -[cost] -input = 3.00 -output = 15.00 -cache_read = 0.30 -cache_write = 3.75 - -[limit] -context = 200_000 -output = 64_000 - -[modalities] -input = ["text", "image", "pdf"] -output = ["text"] - -[provider] -npm = "@ai-sdk/anthropic" -api = "https://zenmux.ai/api/anthropic/v1" diff --git a/providers/zenmux/models/anthropic/claude-fable-5.1.toml b/providers/zenmux/models/anthropic/claude-fable-5.1.toml new file mode 100644 index 00000000000..0673940f3a3 --- /dev/null +++ b/providers/zenmux/models/anthropic/claude-fable-5.1.toml @@ -0,0 +1,17 @@ +# Effort: reasoning_effort = low|medium|high|xhigh|max +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "anthropic/claude-fable-5-1" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high", "xhigh", "max"] + +[cost] +input = 10 +output = 50 +cache_read = 0.25 +cache_write = 12.5 + +[provider] +npm = "@ai-sdk/anthropic" +api = "https://zenmux.ai/api/anthropic/v1" diff --git a/providers/zenmux/models/anthropic/claude-fable-5.toml b/providers/zenmux/models/anthropic/claude-fable-5.toml index 16e066bd842..5f1b50e3e5a 100644 --- a/providers/zenmux/models/anthropic/claude-fable-5.toml +++ b/providers/zenmux/models/anthropic/claude-fable-5.toml @@ -1,11 +1,16 @@ +# Effort: reasoning_effort = low|medium|high|xhigh|max +# https://zenmux.ai/docs/guide/advanced/reasoning.html base_model = "anthropic/claude-fable-5" -reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high", "xhigh", "max"] [cost] -input = 10.00 -output = 50.00 -cache_read = 1.00 -cache_write = 12.50 +input = 10 +output = 50 +cache_read = 1 +cache_write = 12.5 [provider] npm = "@ai-sdk/anthropic" diff --git a/providers/zenmux/models/anthropic/claude-haiku-4.5.toml b/providers/zenmux/models/anthropic/claude-haiku-4.5.toml index d3b93b2f05f..d4578029bb8 100644 --- a/providers/zenmux/models/anthropic/claude-haiku-4.5.toml +++ b/providers/zenmux/models/anthropic/claude-haiku-4.5.toml @@ -1,28 +1,19 @@ +# Budget: thinking.budget_tokens (integer) +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "anthropic/claude-haiku-4-5" name = "Claude Haiku 4.5" -description = "Fast Claude model for responsive assistance, classification, and lightweight agents" -release_date = "2025-10-15" -last_updated = "2025-10-15" -attachment = true -reasoning = false -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false +structured_output = true + +[[reasoning_options]] +type = "budget_tokens" +min = 1_024 [cost] -input = 1.00 -output = 5.00 -cache_read = 0.10 +input = 1 +output = 5 +cache_read = 0.1 cache_write = 1.25 -[limit] -context = 200_000 -output = 64_000 - -[modalities] -input = ["image", "text"] -output = ["text"] - [provider] npm = "@ai-sdk/anthropic" api = "https://zenmux.ai/api/anthropic/v1" diff --git a/providers/zenmux/models/anthropic/claude-opus-4.1.toml b/providers/zenmux/models/anthropic/claude-opus-4.1.toml index 1852583315e..024c7ebddf8 100644 --- a/providers/zenmux/models/anthropic/claude-opus-4.1.toml +++ b/providers/zenmux/models/anthropic/claude-opus-4.1.toml @@ -1,29 +1,19 @@ +# Budget: thinking.budget_tokens (integer) +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "anthropic/claude-opus-4-1" name = "Claude Opus 4.1" -description = "Flagship Claude model for deep reasoning, coding, and long-horizon agents" -release_date = "2025-08-05" -last_updated = "2025-08-05" -attachment = true -reasoning = true -reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false +structured_output = true + +[[reasoning_options]] +type = "budget_tokens" +min = 1_024 [cost] -input = 15.00 -output = 75.00 -cache_read = 1.50 +input = 15 +output = 75 +cache_read = 1.5 cache_write = 18.75 -[limit] -context = 200_000 -output = 64_000 - -[modalities] -input = ["image", "text", "pdf"] -output = ["text"] - [provider] npm = "@ai-sdk/anthropic" api = "https://zenmux.ai/api/anthropic/v1" diff --git a/providers/zenmux/models/anthropic/claude-opus-4.5.toml b/providers/zenmux/models/anthropic/claude-opus-4.5.toml index 5f09cfcad6a..89ff965010b 100644 --- a/providers/zenmux/models/anthropic/claude-opus-4.5.toml +++ b/providers/zenmux/models/anthropic/claude-opus-4.5.toml @@ -1,28 +1,29 @@ +# Effort: reasoning_effort = low|medium|high +# Budget: thinking.budget_tokens (integer) +# ZenMux's 2026-09-19 catalog and Anthropic-compatible model list explicitly +# report max_completion_tokens=32000 for this hosted route. +# https://zenmux.ai/docs/guide/advanced/reasoning.html +# https://zenmux.ai/docs/api/anthropic/anthropic-list-models.html +base_model = "anthropic/claude-opus-4-5" name = "Claude Opus 4.5" -description = "Flagship Claude model for deep reasoning, coding, and long-horizon agents" -release_date = "2025-11-24" -last_updated = "2025-11-24" -attachment = true -reasoning = true -reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false +structured_output = true + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high"] + +[[reasoning_options]] +type = "budget_tokens" +min = 1_024 [cost] -input = 5.00 -output = 25.00 -cache_read = 0.50 +input = 5 +output = 25 +cache_read = 0.5 cache_write = 6.25 [limit] -context = 200_000 -output = 64_000 - -[modalities] -input = ["pdf", "image", "text"] -output = ["text"] +output = 32_000 [provider] npm = "@ai-sdk/anthropic" diff --git a/providers/zenmux/models/anthropic/claude-opus-4.6.toml b/providers/zenmux/models/anthropic/claude-opus-4.6.toml index 19c0e670677..2452e86cd51 100644 --- a/providers/zenmux/models/anthropic/claude-opus-4.6.toml +++ b/providers/zenmux/models/anthropic/claude-opus-4.6.toml @@ -1,28 +1,21 @@ -name = "Claude Opus 4.6" -description = "Flagship Claude model for deep reasoning, coding, and long-horizon agents" -release_date = "2026-02-06" -last_updated = "2026-02-06" -attachment = true -reasoning = true -reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] -temperature = true -tool_call = true -knowledge = "2025-05-31" -open_weights = false +# Effort: reasoning_effort = low|medium|high|max +# Budget: thinking.budget_tokens (integer) +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "anthropic/claude-opus-4-6" -[cost] -input = 5.00 -output = 25.00 -cache_read = 0.50 -cache_write = 6.25 +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high", "max"] -[limit] -context = 1_000_000 -output = 128_000 +[[reasoning_options]] +type = "budget_tokens" +min = 1_024 -[modalities] -input = ["image", "text"] -output = ["text"] +[cost] +input = 5 +output = 25 +cache_read = 0.5 +cache_write = 6.25 [provider] npm = "@ai-sdk/anthropic" diff --git a/providers/zenmux/models/anthropic/claude-opus-4.7.toml b/providers/zenmux/models/anthropic/claude-opus-4.7.toml index 59aff8efcf6..c75963a946b 100644 --- a/providers/zenmux/models/anthropic/claude-opus-4.7.toml +++ b/providers/zenmux/models/anthropic/claude-opus-4.7.toml @@ -1,30 +1,17 @@ -name = "Claude Opus 4.7" -description = "Flagship Claude model for deep reasoning, coding, and long-horizon agents" -release_date = "2026-04-16" -last_updated = "2026-04-16" -attachment = true -reasoning = true -reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] -temperature = false -tool_call = true -knowledge = "2026-01-31" -open_weights = false +# Effort: reasoning_effort = low|medium|high|xhigh|max +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "anthropic/claude-opus-4-7" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high", "xhigh", "max"] [cost] -input = 5.00 -output = 25.00 -cache_read = 0.50 +input = 5 +output = 25 +cache_read = 0.5 cache_write = 6.25 -[limit] -context = 1_000_000 -output = 128_000 - -[modalities] -input = ["text", "image", "pdf"] -output = ["text"] - - [provider] npm = "@ai-sdk/anthropic" api = "https://zenmux.ai/api/anthropic/v1" diff --git a/providers/zenmux/models/anthropic/claude-opus-4.8.toml b/providers/zenmux/models/anthropic/claude-opus-4.8.toml index 6fbaf57cd45..9cc519f01d3 100644 --- a/providers/zenmux/models/anthropic/claude-opus-4.8.toml +++ b/providers/zenmux/models/anthropic/claude-opus-4.8.toml @@ -1,10 +1,15 @@ +# Effort: reasoning_effort = low|medium|high|xhigh|max +# https://zenmux.ai/docs/guide/advanced/reasoning.html base_model = "anthropic/claude-opus-4-8" -reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high", "xhigh", "max"] [cost] -input = 5.00 -output = 25.00 -cache_read = 0.50 +input = 5 +output = 25 +cache_read = 0.5 cache_write = 6.25 [provider] diff --git a/providers/zenmux/models/anthropic/claude-opus-4.toml b/providers/zenmux/models/anthropic/claude-opus-4.toml deleted file mode 100644 index 48c39f7c072..00000000000 --- a/providers/zenmux/models/anthropic/claude-opus-4.toml +++ /dev/null @@ -1,29 +0,0 @@ -name = "Claude Opus 4" -description = "Flagship Claude model for deep reasoning, coding, and long-horizon agents" -release_date = "2025-05-22" -last_updated = "2025-05-22" -attachment = true -reasoning = true -reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false - -[cost] -input = 15.00 -output = 75.00 -cache_read = 1.50 -cache_write = 18.75 - -[limit] -context = 200_000 -output = 32_000 - -[modalities] -input = ["image", "text", "pdf"] -output = ["text"] - -[provider] -npm = "@ai-sdk/anthropic" -api = "https://zenmux.ai/api/anthropic/v1" diff --git a/providers/zenmux/models/anthropic/claude-opus-5.toml b/providers/zenmux/models/anthropic/claude-opus-5.toml new file mode 100644 index 00000000000..2be7ab604c2 --- /dev/null +++ b/providers/zenmux/models/anthropic/claude-opus-5.toml @@ -0,0 +1,17 @@ +# Effort: reasoning_effort = low|medium|high|xhigh|max +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "anthropic/claude-opus-5" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high", "xhigh", "max"] + +[cost] +input = 5 +output = 25 +cache_read = 0.5 +cache_write = 6.25 + +[provider] +npm = "@ai-sdk/anthropic" +api = "https://zenmux.ai/api/anthropic/v1" diff --git a/providers/zenmux/models/anthropic/claude-sonnet-4.5.toml b/providers/zenmux/models/anthropic/claude-sonnet-4.5.toml index bca5d46dbc5..adb2642adad 100644 --- a/providers/zenmux/models/anthropic/claude-sonnet-4.5.toml +++ b/providers/zenmux/models/anthropic/claude-sonnet-4.5.toml @@ -1,28 +1,28 @@ +# Budget: thinking.budget_tokens (integer) +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "anthropic/claude-sonnet-4-5" name = "Claude Sonnet 4.5" -description = "Balanced Claude model for coding, analysis, agent workflows, and cost control" -release_date = "2025-09-29" -last_updated = "2025-09-29" -attachment = true -reasoning = true -reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false +structured_output = true + +[[reasoning_options]] +type = "budget_tokens" +min = 1_024 [cost] -input = 3.00 -output = 15.00 -cache_read = 0.30 +input = 3 +output = 15 +cache_read = 0.3 cache_write = 3.75 -[limit] -context = 1000_000 -output = 64_000 +[[cost.tiers]] +tier = { type = "context", size = 200_000 } +input = 6 +output = 22.5 +cache_read = 0.6 +cache_write = 7.5 -[modalities] -input = ["text", "image", "pdf"] -output = ["text"] +[limit] +context = 1_000_000 [provider] npm = "@ai-sdk/anthropic" diff --git a/providers/zenmux/models/anthropic/claude-sonnet-4.6.toml b/providers/zenmux/models/anthropic/claude-sonnet-4.6.toml index c69459b62d6..e31dfca3f91 100644 --- a/providers/zenmux/models/anthropic/claude-sonnet-4.6.toml +++ b/providers/zenmux/models/anthropic/claude-sonnet-4.6.toml @@ -1,28 +1,22 @@ -name = "Claude Sonnet 4.6" -description = "Balanced Claude model for coding, analysis, agent workflows, and cost control" -release_date = "2026-02-18" -last_updated = "2026-02-18" -attachment = true -reasoning = true -reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] -temperature = true -tool_call = true -knowledge = "2025-08-31" -open_weights = false +# Effort: reasoning_effort = low|medium|high|max +# Budget: thinking.budget_tokens (integer) +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "anthropic/claude-sonnet-4-6" +structured_output = true -[cost] -input = 3.00 -output = 15.00 -cache_read = 0.30 -cache_write = 3.75 +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high", "max"] -[limit] -context = 1_000_000 -output = 64_000 +[[reasoning_options]] +type = "budget_tokens" +min = 1_024 -[modalities] -input = ["text", "image"] -output = ["text"] +[cost] +input = 3 +output = 15 +cache_read = 0.3 +cache_write = 3.75 [provider] npm = "@ai-sdk/anthropic" diff --git a/providers/zenmux/models/anthropic/claude-sonnet-4.toml b/providers/zenmux/models/anthropic/claude-sonnet-4.toml deleted file mode 100644 index 3bd6222f82d..00000000000 --- a/providers/zenmux/models/anthropic/claude-sonnet-4.toml +++ /dev/null @@ -1,29 +0,0 @@ -name = "Claude Sonnet 4" -description = "Balanced Claude model for coding, analysis, agent workflows, and cost control" -release_date = "2025-05-22" -last_updated = "2025-05-22" -attachment = true -reasoning = true -reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false - -[cost] -input = 3.00 -output = 15.00 -cache_read = 0.30 -cache_write = 3.75 - -[limit] -context = 1000_000 -output = 64_000 - -[modalities] -input = ["image", "text", "pdf"] -output = ["text"] - -[provider] -npm = "@ai-sdk/anthropic" -api = "https://zenmux.ai/api/anthropic/v1" diff --git a/providers/zenmux/models/anthropic/claude-sonnet-5-free.toml b/providers/zenmux/models/anthropic/claude-sonnet-5-free.toml deleted file mode 100644 index 97f3f99e91b..00000000000 --- a/providers/zenmux/models/anthropic/claude-sonnet-5-free.toml +++ /dev/null @@ -1,13 +0,0 @@ -base_model = "anthropic/claude-sonnet-5" -name = "Claude Sonnet 5 (Free)" -reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] - -[cost] -input = 0.00 -output = 0.00 -cache_read = 0.00 -cache_write = 0.00 - -[provider] -npm = "@ai-sdk/anthropic" -api = "https://zenmux.ai/api/anthropic/v1" diff --git a/providers/zenmux/models/anthropic/claude-sonnet-5.toml b/providers/zenmux/models/anthropic/claude-sonnet-5.toml index 693121528d8..2421fc9dcfd 100644 --- a/providers/zenmux/models/anthropic/claude-sonnet-5.toml +++ b/providers/zenmux/models/anthropic/claude-sonnet-5.toml @@ -1,11 +1,26 @@ +# Toggle: thinking.type = enabled|disabled +# Effort: reasoning_effort = low|medium|high|xhigh|max +# ZenMux's 2026-09-19 OpenAI- and Anthropic-compatible model lists both cap +# this hosted route at 64,000 output tokens. +# https://zenmux.ai/docs/guide/advanced/reasoning.html +# https://zenmux.ai/docs/api/anthropic/anthropic-list-models.html base_model = "anthropic/claude-sonnet-5" -reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high", "xhigh", "max"] [cost] -input = 2.00 -output = 10.00 -cache_read = 0.20 -cache_write = 4.00 +input = 2 +output = 10 +cache_read = 0.2 +cache_write = 2.5 + +[limit] +output = 64_000 [provider] npm = "@ai-sdk/anthropic" diff --git a/providers/zenmux/models/deepseek/deepseek-chat-v3.1.toml b/providers/zenmux/models/deepseek/deepseek-chat-v3.1.toml new file mode 100644 index 00000000000..dd9b2f26e3c --- /dev/null +++ b/providers/zenmux/models/deepseek/deepseek-chat-v3.1.toml @@ -0,0 +1,16 @@ +# ZenMux's 2026-09-19 model lists expose this as the non-thinking chat route, +# so reasoning=false is an intentional host override. +# https://zenmux.ai/docs/api/openai/openai-list-models.html +base_model = "deepseek/deepseek-v3.1" +name = "DeepSeek V3.1" +reasoning = false +structured_output = true + +[cost] +input = 0.28 +output = 1.11 +cache_read = 0.056 + +[limit] +context = 128_000 +output = 65_536 diff --git a/providers/zenmux/models/deepseek/deepseek-chat.toml b/providers/zenmux/models/deepseek/deepseek-chat.toml deleted file mode 100644 index b437a255b41..00000000000 --- a/providers/zenmux/models/deepseek/deepseek-chat.toml +++ /dev/null @@ -1,24 +0,0 @@ -name = "DeepSeek-V3.2 (Non-thinking Mode)" -description = "DeepSeek chat model for instruction following, coding, and analysis" -release_date = "2025-12-01" -last_updated = "2025-12-01" -attachment = false -reasoning = false -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false -status = "deprecated" - -[cost] -input = 0.28 -output = 0.42 -cache_read = 0.03 - -[limit] -context = 128_000 -output = 64_000 - -[modalities] -input = ["text"] -output = ["text"] diff --git a/providers/zenmux/models/deepseek/deepseek-r1-0528.toml b/providers/zenmux/models/deepseek/deepseek-r1-0528.toml new file mode 100644 index 00000000000..3dba671e91b --- /dev/null +++ b/providers/zenmux/models/deepseek/deepseek-r1-0528.toml @@ -0,0 +1,11 @@ +base_model = "deepseek/deepseek-r1-0528" +reasoning_options = [] + +[cost] +input = 0.56 +output = 2.23 +cache_read = 0.112 + +[limit] +context = 64_000 +output = 64_000 diff --git a/providers/zenmux/models/deepseek/deepseek-v3.2-exp.toml b/providers/zenmux/models/deepseek/deepseek-v3.2-exp.toml index 2580aef1fb7..d922c63e17a 100644 --- a/providers/zenmux/models/deepseek/deepseek-v3.2-exp.toml +++ b/providers/zenmux/models/deepseek/deepseek-v3.2-exp.toml @@ -1,23 +1,10 @@ -name = "DeepSeek-V3.2-Exp" -description = "DeepSeek chat model for instruction following, coding, and analysis" -release_date = "2025-09-29" -last_updated = "2025-09-29" -attachment = false -reasoning = true -reasoning_options = [{ type = "toggle" }] -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false +# Toggle: thinking.type = enabled|disabled +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "deepseek/deepseek-v3.2-exp" -[cost] -input = 0.22 -output = 0.33 - -[limit] -context = 163_000 -output = 64_000 +[[reasoning_options]] +type = "toggle" -[modalities] -input = ["text"] -output = ["text"] +[cost] +input = 0.216 +output = 0.328 diff --git a/providers/zenmux/models/deepseek/deepseek-v3.2.toml b/providers/zenmux/models/deepseek/deepseek-v3.2.toml index 63edf977fda..ed7bc41bf4d 100644 --- a/providers/zenmux/models/deepseek/deepseek-v3.2.toml +++ b/providers/zenmux/models/deepseek/deepseek-v3.2.toml @@ -1,23 +1,16 @@ -name = "DeepSeek V3.2" -description = "DeepSeek chat model for instruction following, coding, and analysis" -release_date = "2025-12-05" -last_updated = "2025-12-05" -attachment = false -reasoning = true -reasoning_options = [{ type = "toggle" }] -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false +# Toggle: thinking.type = enabled|disabled +# ZenMux's 2026-09-19 catalog caps this route at 8,000 output tokens. +# https://zenmux.ai/docs/guide/advanced/reasoning.html +# https://zenmux.ai/docs/api/openai/openai-list-models.html +base_model = "deepseek/deepseek-v3.2" + +[[reasoning_options]] +type = "toggle" [cost] -input = 0.28 -output = 0.43 +input = 0.293 +output = 0.4395 +cache_read = 0.0293 [limit] -context = 128_000 -output = 64_000 - -[modalities] -input = ["text"] -output = ["text"] +output = 8_000 diff --git a/providers/zenmux/models/deepseek/deepseek-v4-flash.toml b/providers/zenmux/models/deepseek/deepseek-v4-flash.toml index 45ee40683c5..53a23284856 100644 --- a/providers/zenmux/models/deepseek/deepseek-v4-flash.toml +++ b/providers/zenmux/models/deepseek/deepseek-v4-flash.toml @@ -1,9 +1,21 @@ -base_model = "deepseek/deepseek-v4-flash" -reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high"] }] +# Toggle: thinking.type = enabled|disabled +# Effort: reasoning_effort = low|high|max +# ZenMux's undated deepseek-v4-flash slug is explicitly displayed as the 0731 +# snapshot in its 2026-09-19 catalog, so the dated base name is intentional. +# https://zenmux.ai/docs/guide/advanced/reasoning.html +# https://zenmux.ai/docs/api/openai/openai-list-models.html +base_model = "deepseek/deepseek-v4-flash-0731" [interleaved] field = "reasoning_content" +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["low", "high", "max"] + [cost] input = 0.14 output = 0.28 diff --git a/providers/zenmux/models/deepseek/deepseek-v4-pro.toml b/providers/zenmux/models/deepseek/deepseek-v4-pro.toml index 5e110965760..83402b4b635 100644 --- a/providers/zenmux/models/deepseek/deepseek-v4-pro.toml +++ b/providers/zenmux/models/deepseek/deepseek-v4-pro.toml @@ -1,9 +1,21 @@ -base_model = "deepseek/deepseek-v4-pro" -reasoning_options = [{ type = "toggle" }, { type = "effort", values = ["low", "medium", "high"] }] +# Toggle: thinking.type = enabled|disabled +# Effort: reasoning_effort = high|max +# ZenMux's undated deepseek-v4-pro slug is explicitly displayed as the 0813 +# snapshot in its 2026-09-19 catalog, so the dated base name is intentional. +# https://zenmux.ai/docs/guide/advanced/reasoning.html +# https://zenmux.ai/docs/api/openai/openai-list-models.html +base_model = "deepseek/deepseek-v4-pro-0813" [interleaved] field = "reasoning_content" +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["high", "max"] + [cost] input = 0.435 output = 0.87 diff --git a/providers/zenmux/models/deepseek/deepseek-v4.1-flash.toml b/providers/zenmux/models/deepseek/deepseek-v4.1-flash.toml new file mode 100644 index 00000000000..5a55042d563 --- /dev/null +++ b/providers/zenmux/models/deepseek/deepseek-v4.1-flash.toml @@ -0,0 +1,24 @@ +# Toggle: thinking.type = enabled|disabled +# Effort: reasoning_effort = low|high|max +# ZenMux's 2026-09-19 protocol lists publish discounted bands of USD 0.075 input, +# 0.30 output, and 0.0015 cache-read per million tokens. Higher time bands of +# 0.15/0.60/0.003 are intentionally omitted because the schema only models +# context tiers, not time-of-day tiers. +# https://zenmux.ai/docs/guide/advanced/reasoning.html +# https://zenmux.ai/docs/api/openai/openai-list-models.html +base_model = "deepseek/deepseek-v4.1-flash" + +[interleaved] +field = "reasoning_content" + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["low", "high", "max"] + +[cost] +input = 0.075 +output = 0.3 +cache_read = 0.0015 diff --git a/providers/zenmux/models/google/gemini-2.5-flash-image.toml b/providers/zenmux/models/google/gemini-2.5-flash-image.toml new file mode 100644 index 00000000000..fe2240fdee2 --- /dev/null +++ b/providers/zenmux/models/google/gemini-2.5-flash-image.toml @@ -0,0 +1,16 @@ +# ZenMux's 2026-09-19 page catalog has supports_reasoning=0 and its live +# Google-compatible model-list response has thinking=false for this exact id. +# The lab's reasoning surface is therefore intentionally disabled on this host. +# https://zenmux.ai/docs/api/vertexai/google-list-models.html +base_model = "google/gemini-2.5-flash-image" +name = "Gemini 2.5 Flash Image (Nano Banana)" +reasoning = false +structured_output = true + +[cost] +input = 0.3 +output = 30 +cache_read = 0.03 + +[limit] +output = 8_192 diff --git a/providers/zenmux/models/google/gemini-2.5-flash-lite.toml b/providers/zenmux/models/google/gemini-2.5-flash-lite.toml index f83ea6a084c..395ce356f55 100644 --- a/providers/zenmux/models/google/gemini-2.5-flash-lite.toml +++ b/providers/zenmux/models/google/gemini-2.5-flash-lite.toml @@ -1,26 +1,22 @@ -# Vertex `thinkingBudget`: -1 (dynamic) or 512..24576; no disable sentinel is listed. -# https://docs.zenmux.ai/api/vertexai/generate-content.html (accessed 2026-06-25) +# Toggle: reasoning.enabled = true|false +# Budget: reasoning.max_tokens (integer) +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "google/gemini-2.5-flash-lite" name = "Gemini 2.5 Flash Lite" -description = "Low-latency Gemini model for high-volume multimodal and agent workloads" -release_date = "2025-07-22" -last_updated = "2025-07-22" -attachment = true -reasoning = false -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "budget_tokens" +min = 512 +max = 24_576 [cost] -input = 0.10 -output = 0.40 -cache_read = 0.03 -cache_write = 1.00 +input = 0.1 +output = 0.4 +cache_read = 0.01 +cache_write = 1 [limit] -context = 1048_000 -output = 64_000 - -[modalities] -input = ["pdf", "image", "text", "audio"] -output = ["text"] +output = 65_535 diff --git a/providers/zenmux/models/google/gemini-2.5-flash.toml b/providers/zenmux/models/google/gemini-2.5-flash.toml index 41e23a328c3..ecea4b32b36 100644 --- a/providers/zenmux/models/google/gemini-2.5-flash.toml +++ b/providers/zenmux/models/google/gemini-2.5-flash.toml @@ -1,27 +1,21 @@ -# Vertex `thinkingBudget`: -1 (dynamic) or 0..24576; 0 disables thinking. -# https://docs.zenmux.ai/api/vertexai/generate-content.html (accessed 2026-06-25) -name = "Gemini 2.5 Flash" -description = "Fast Gemini model balancing multimodal reasoning, tool use, and cost" -release_date = "2025-06-17" -last_updated = "2025-06-17" -attachment = true -reasoning = true -reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false +# Toggle: reasoning.enabled = true|false +# Budget: reasoning.max_tokens (integer) +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "google/gemini-2.5-flash" + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "budget_tokens" +min = 0 +max = 24_576 [cost] -input = 0.30 -output = 2.50 -cache_read = 0.07 -cache_write = 1.00 +input = 0.3 +output = 2.5 +cache_read = 0.03 +cache_write = 1 [limit] -context = 1048_000 -output = 64_000 - -[modalities] -input = ["pdf", "image", "text", "audio"] -output = ["text"] +output = 65_535 diff --git a/providers/zenmux/models/google/gemini-2.5-pro.toml b/providers/zenmux/models/google/gemini-2.5-pro.toml index 44216397c1e..f3f65b21018 100644 --- a/providers/zenmux/models/google/gemini-2.5-pro.toml +++ b/providers/zenmux/models/google/gemini-2.5-pro.toml @@ -1,27 +1,24 @@ -# Vertex `thinkingBudget`: -1 (dynamic) or 128..32768; thinking cannot be disabled. -# https://docs.zenmux.ai/api/vertexai/generate-content.html (accessed 2026-06-25) -name = "Gemini 2.5 Pro" -description = "Advanced Gemini model for complex reasoning, coding, and multimodal analysis" -release_date = "2025-06-17" -last_updated = "2025-06-17" -attachment = true -reasoning = true -reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false +# Budget: reasoning.max_tokens (integer) +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "google/gemini-2.5-pro" + +[[reasoning_options]] +type = "budget_tokens" +min = 128 +max = 32_768 [cost] input = 1.25 -output = 10.00 -cache_read = 0.31 -cache_write = 4.50 +output = 10 +cache_read = 0.13 +cache_write = 4.5 -[limit] -context = 1048_000 -output = 64_000 +[[cost.tiers]] +tier = { type = "context", size = 200_000 } +input = 2.5 +output = 15 +cache_read = 0.25 +cache_write = 4.5 -[modalities] -input = ["pdf", "image", "text", "audio", "video"] -output = ["text"] +[limit] +output = 65_535 diff --git a/providers/zenmux/models/google/gemini-3-flash-preview.toml b/providers/zenmux/models/google/gemini-3-flash-preview.toml index 8dd6f77cc68..dd8846f7703 100644 --- a/providers/zenmux/models/google/gemini-3-flash-preview.toml +++ b/providers/zenmux/models/google/gemini-3-flash-preview.toml @@ -1,25 +1,15 @@ -name = "Gemini 3 Flash Preview" -description = "Fast Gemini model balancing multimodal reasoning, tool use, and cost" -release_date = "2025-12-17" -last_updated = "2025-12-17" -attachment = true -reasoning = true -reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false +# Effort: reasoning_effort = minimal|low|medium|high +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "google/gemini-3-flash-preview" + +[[reasoning_options]] +type = "effort" +values = ["minimal", "low", "medium", "high"] [cost] -input = 0.50 -output = 3.00 +input = 0.5 +output = 3 cache_read = 0.05 -cache_write = 1.00 [limit] -context = 1048_000 -output = 64_000 - -[modalities] -input = ["text", "image", "pdf", "audio"] -output = ["text"] +output = 65_535 diff --git a/providers/zenmux/models/google/gemini-3-pro-image.toml b/providers/zenmux/models/google/gemini-3-pro-image.toml new file mode 100644 index 00000000000..56c1d10d15c --- /dev/null +++ b/providers/zenmux/models/google/gemini-3-pro-image.toml @@ -0,0 +1,14 @@ +# ZenMux's 2026-09-19 page catalog has supports_reasoning=0 and its live +# Google-compatible model-list response has thinking=false for this exact id. +# The lab's reasoning surface is therefore intentionally disabled on this host. +# https://zenmux.ai/docs/api/vertexai/google-list-models.html +base_model = "google/gemini-3-pro-image" +name = "Nano Banana Pro (Gemini 3 Pro Image)" +reasoning = false + +[cost] +input = 2 +output = 120 + +[limit] +output = 8_192 diff --git a/providers/zenmux/models/google/gemini-3.1-flash-image.toml b/providers/zenmux/models/google/gemini-3.1-flash-image.toml new file mode 100644 index 00000000000..5052f88acef --- /dev/null +++ b/providers/zenmux/models/google/gemini-3.1-flash-image.toml @@ -0,0 +1,18 @@ +# Effort: reasoning_effort = minimal|high +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "google/gemini-3.1-flash-image" +name = "Nano Banana 2 (Gemini 3.1 Flash Image)" + +[[reasoning_options]] +type = "effort" +values = ["minimal", "high"] + +[cost] +input = 0.5 +output = 60 + +[limit] +context = 65_536 + +[modalities] +input = ["image", "text", "video"] diff --git a/providers/zenmux/models/google/gemini-3.1-flash-lite-image.toml b/providers/zenmux/models/google/gemini-3.1-flash-lite-image.toml new file mode 100644 index 00000000000..265bf72c522 --- /dev/null +++ b/providers/zenmux/models/google/gemini-3.1-flash-lite-image.toml @@ -0,0 +1,17 @@ +# Effort: reasoning_effort = minimal|high +# ZenMux's 2026-09-19 catalog explicitly reports max_completion_tokens=32768. +# https://zenmux.ai/docs/guide/advanced/reasoning.html +# https://zenmux.ai/docs/api/vertexai/google-list-models.html +base_model = "google/gemini-3.1-flash-lite-image" +name = "Nano Banana 2 Lite (Gemini 3.1 Flash-Lite Image)" + +[[reasoning_options]] +type = "effort" +values = ["minimal", "high"] + +[cost] +input = 0.25 +output = 30 + +[limit] +output = 32_768 diff --git a/providers/zenmux/models/google/gemini-3.1-flash-lite-preview.toml b/providers/zenmux/models/google/gemini-3.1-flash-lite-preview.toml deleted file mode 100644 index f6c3fa28eaf..00000000000 --- a/providers/zenmux/models/google/gemini-3.1-flash-lite-preview.toml +++ /dev/null @@ -1,21 +0,0 @@ -name = "Gemini 3.1 Flash Lite Preview" -description = "Low-latency Gemini model for high-volume multimodal and agent workloads" -release_date = "2025-03-20" -last_updated = "2025-03-20" -attachment = true -reasoning = false -temperature = true -tool_call = true -open_weights = false - -[cost] -input = 0.25 -output = 1.50 - -[limit] -context = 1050000 -output = 65530 - -[modalities] -input = ["text", "image", "audio", "video"] -output = ["text"] diff --git a/providers/zenmux/models/google/gemini-3.1-flash-lite.toml b/providers/zenmux/models/google/gemini-3.1-flash-lite.toml index 8e69ceae67f..d8eacb4c721 100644 --- a/providers/zenmux/models/google/gemini-3.1-flash-lite.toml +++ b/providers/zenmux/models/google/gemini-3.1-flash-lite.toml @@ -1,7 +1,15 @@ +# Effort: reasoning_effort = minimal|low|medium|high +# https://zenmux.ai/docs/guide/advanced/reasoning.html base_model = "google/gemini-3.1-flash-lite" -reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] + +[[reasoning_options]] +type = "effort" +values = ["minimal", "low", "medium", "high"] [cost] input = 0.25 -output = 1.50 +output = 1.5 cache_read = 0.025 + +[limit] +output = 65_535 diff --git a/providers/zenmux/models/google/gemini-3.1-flash-tts-preview.toml b/providers/zenmux/models/google/gemini-3.1-flash-tts-preview.toml new file mode 100644 index 00000000000..3e9d7912df9 --- /dev/null +++ b/providers/zenmux/models/google/gemini-3.1-flash-tts-preview.toml @@ -0,0 +1,8 @@ +base_model = "google/gemini-3.1-flash-tts-preview" + +[cost] +input = 1 +output = 20 + +[limit] +output = 8_192 diff --git a/providers/zenmux/models/google/gemini-3.1-pro-preview.toml b/providers/zenmux/models/google/gemini-3.1-pro-preview.toml index d614e3d8f8b..573cf6a8cfd 100644 --- a/providers/zenmux/models/google/gemini-3.1-pro-preview.toml +++ b/providers/zenmux/models/google/gemini-3.1-pro-preview.toml @@ -1,25 +1,23 @@ -name = "Gemini 3.1 Pro Preview" -description = "Advanced Gemini model for complex reasoning, coding, and multimodal analysis" -release_date = "2026-02-19" -last_updated = "2026-02-19" -attachment = true -reasoning = true -reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] -temperature = true -tool_call = true -knowledge = "2026-02-19" -open_weights = false +# Effort: reasoning_effort = low|medium|high +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "google/gemini-3.1-pro-preview" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high"] [cost] -input = 2.00 -output = 12.00 -cache_read = 0.20 -cache_write = 4.50 +input = 2 +output = 12 +cache_read = 0.2 +cache_write = 4.5 -[limit] -context = 1048_000 -output = 64_000 +[[cost.tiers]] +tier = { type = "context", size = 200_000 } +input = 4 +output = 18 +cache_read = 0.4 +cache_write = 4.5 -[modalities] -input = ["text", "image", "pdf", "audio", "video"] -output = ["text"] +[limit] +output = 65_535 diff --git a/providers/zenmux/models/google/gemini-3.5-flash-lite.toml b/providers/zenmux/models/google/gemini-3.5-flash-lite.toml new file mode 100644 index 00000000000..4f2913c2fe5 --- /dev/null +++ b/providers/zenmux/models/google/gemini-3.5-flash-lite.toml @@ -0,0 +1,15 @@ +# Effort: reasoning_effort = minimal|low|medium|high +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "google/gemini-3.5-flash-lite" + +[[reasoning_options]] +type = "effort" +values = ["minimal", "low", "medium", "high"] + +[cost] +input = 0.3 +output = 2.5 +cache_read = 0.03 + +[limit] +output = 65_535 diff --git a/providers/zenmux/models/google/gemini-3.5-flash.toml b/providers/zenmux/models/google/gemini-3.5-flash.toml index 53b64265d5b..79e2b3b0505 100644 --- a/providers/zenmux/models/google/gemini-3.5-flash.toml +++ b/providers/zenmux/models/google/gemini-3.5-flash.toml @@ -1,7 +1,19 @@ +# Effort: reasoning_effort = minimal|low|medium|high +# ZenMux's 2026-09-19 catalog explicitly lists the 1-hour cache-write token +# band at USD 0 per million tokens; this is a published free price, not a sentinel. +# https://zenmux.ai/docs/guide/advanced/reasoning.html +# https://zenmux.ai/docs/api/vertexai/google-list-models.html base_model = "google/gemini-3.5-flash" -reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] + +[[reasoning_options]] +type = "effort" +values = ["minimal", "low", "medium", "high"] [cost] -input = 1.50 -output = 9.00 +input = 1.5 +output = 9 cache_read = 0.15 +cache_write = 0 + +[limit] +output = 65_535 diff --git a/providers/zenmux/models/google/gemini-3.6-flash.toml b/providers/zenmux/models/google/gemini-3.6-flash.toml new file mode 100644 index 00000000000..824853d6b60 --- /dev/null +++ b/providers/zenmux/models/google/gemini-3.6-flash.toml @@ -0,0 +1,15 @@ +# Effort: reasoning_effort = minimal|low|medium|high +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "google/gemini-3.6-flash" + +[[reasoning_options]] +type = "effort" +values = ["minimal", "low", "medium", "high"] + +[cost] +input = 1.5 +output = 7.5 +cache_read = 0.15 + +[limit] +output = 65_535 diff --git a/providers/zenmux/models/google/gemini-3.7-flash.toml b/providers/zenmux/models/google/gemini-3.7-flash.toml new file mode 100644 index 00000000000..728500331f7 --- /dev/null +++ b/providers/zenmux/models/google/gemini-3.7-flash.toml @@ -0,0 +1,15 @@ +# Effort: reasoning_effort = low|medium|high +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "google/gemini-3.7-flash" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high"] + +[cost] +input = 1.5 +output = 7.5 +cache_read = 0.15 + +[limit] +output = 65_535 diff --git a/providers/zenmux/models/google/gemini-3.8-flash.toml b/providers/zenmux/models/google/gemini-3.8-flash.toml new file mode 100644 index 00000000000..54f4e11f662 --- /dev/null +++ b/providers/zenmux/models/google/gemini-3.8-flash.toml @@ -0,0 +1,12 @@ +# Effort: reasoning_effort = low|medium|high +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "google/gemini-3.8-flash" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high"] + +[cost] +input = 1.5 +output = 7.5 +cache_read = 0.15 diff --git a/providers/zenmux/models/google/gemini-embedding-2.toml b/providers/zenmux/models/google/gemini-embedding-2.toml new file mode 100644 index 00000000000..ea461ab7299 --- /dev/null +++ b/providers/zenmux/models/google/gemini-embedding-2.toml @@ -0,0 +1,5 @@ +base_model = "google/gemini-embedding-2" + +[cost] +input = 0.2 +output = 0 diff --git a/providers/zenmux/models/google/gemini-omni-1.1-flash-preview.toml b/providers/zenmux/models/google/gemini-omni-1.1-flash-preview.toml new file mode 100644 index 00000000000..37b2a2874dc --- /dev/null +++ b/providers/zenmux/models/google/gemini-omni-1.1-flash-preview.toml @@ -0,0 +1,6 @@ +base_model = "google/gemini-omni-1.1-flash-preview" +reasoning_options = [] + +[cost] +input = 1.5 +output = 17.5 diff --git a/providers/zenmux/models/google/gemini-omni-flash-preview.toml b/providers/zenmux/models/google/gemini-omni-flash-preview.toml new file mode 100644 index 00000000000..cb5858ba1a0 --- /dev/null +++ b/providers/zenmux/models/google/gemini-omni-flash-preview.toml @@ -0,0 +1,12 @@ +base_model = "google/gemini-omni-flash-preview" +reasoning_options = [] + +[cost] +input = 1.5 +output = 17.5 + +[limit] +context = 131_072 + +[modalities] +output = ["text", "video"] diff --git a/providers/zenmux/models/google/gemma-4-26b-a4b-it.toml b/providers/zenmux/models/google/gemma-4-26b-a4b-it.toml new file mode 100644 index 00000000000..a0667665d07 --- /dev/null +++ b/providers/zenmux/models/google/gemma-4-26b-a4b-it.toml @@ -0,0 +1,17 @@ +# ZenMux's 2026-09-19 page catalog has supports_reasoning=0; its live OpenAI- +# and Anthropic-compatible lists have capabilities.reasoning=false, and its +# Google-compatible list has thinking=false for this exact id. +# https://zenmux.ai/docs/api/openai/openai-list-models.html +base_model = "google/gemma-4-26b-a4b-it" +reasoning = false + +[cost] +input = 0.15 +output = 0.6 +cache_read = 0.015 + +[limit] +output = 65_535 + +[modalities] +input = ["image", "video", "text"] diff --git a/providers/zenmux/models/google/gemma-4-31b-it.toml b/providers/zenmux/models/google/gemma-4-31b-it.toml new file mode 100644 index 00000000000..ffd6edb4574 --- /dev/null +++ b/providers/zenmux/models/google/gemma-4-31b-it.toml @@ -0,0 +1,12 @@ +# ZenMux's 2026-09-19 page catalog has supports_reasoning=0. Its live OpenAI- +# and Anthropic-compatible entries expose no reasoning capability and no price; +# the route is non-free, so cost is intentionally omitted rather than set to zero. +# https://zenmux.ai/docs/api/openai/openai-list-models.html +base_model = "google/gemma-4-31b-it" +reasoning = false + +[limit] +output = 65_535 + +[modalities] +input = ["image", "text", "video"] diff --git a/providers/zenmux/models/google/veo-3.1-fast-generate-001.toml b/providers/zenmux/models/google/veo-3.1-fast-generate-001.toml new file mode 100644 index 00000000000..8b51bd98b0b --- /dev/null +++ b/providers/zenmux/models/google/veo-3.1-fast-generate-001.toml @@ -0,0 +1,7 @@ +# ZenMux lists a 100,000-token host context and resolution-dependent per-second +# video prices; the token-only cost schema cannot represent those prices. +# https://zenmux.ai/docs/api/vertexai/google-list-models.html +base_model = "google/veo-3.1-fast-generate-001" + +[limit] +context = 100_000 diff --git a/providers/zenmux/models/google/veo-3.1-generate-001.toml b/providers/zenmux/models/google/veo-3.1-generate-001.toml new file mode 100644 index 00000000000..8e87d322916 --- /dev/null +++ b/providers/zenmux/models/google/veo-3.1-generate-001.toml @@ -0,0 +1,10 @@ +# ZenMux lists a 100,000-token host context and resolution-dependent per-second +# video prices; the token-only cost schema cannot represent those prices. +# https://zenmux.ai/docs/api/vertexai/google-list-models.html +base_model = "google/veo-3.1-generate-001" + +[limit] +context = 100_000 + +[modalities] +input = ["text", "image", "video"] diff --git a/providers/zenmux/models/google/veo-3.1-lite-generate-001.toml b/providers/zenmux/models/google/veo-3.1-lite-generate-001.toml new file mode 100644 index 00000000000..2841b7080af --- /dev/null +++ b/providers/zenmux/models/google/veo-3.1-lite-generate-001.toml @@ -0,0 +1,7 @@ +# ZenMux lists a 100,000-token host context and resolution-dependent per-second +# video prices; the token-only cost schema cannot represent those prices. +# https://zenmux.ai/docs/api/vertexai/google-list-models.html +base_model = "google/veo-3.1-lite-generate-001" + +[limit] +context = 100_000 diff --git a/providers/zenmux/models/moonshotai/kimi-k2-0905.toml b/providers/zenmux/models/moonshotai/kimi-k2-0905.toml deleted file mode 100644 index ea07cdb73de..00000000000 --- a/providers/zenmux/models/moonshotai/kimi-k2-0905.toml +++ /dev/null @@ -1,24 +0,0 @@ -name = "Kimi K2 0905" -description = "Kimi model for long-context chat, coding, and agentic reasoning" -release_date = "2025-09-04" -last_updated = "2025-09-04" -attachment = false -reasoning = false -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false -status = "deprecated" - -[cost] -input = 0.60 -output = 2.50 -cache_read = 0.15 - -[limit] -context = 262_000 -output = 64_000 - -[modalities] -input = ["text"] -output = ["text"] diff --git a/providers/zenmux/models/moonshotai/kimi-k2-thinking-turbo.toml b/providers/zenmux/models/moonshotai/kimi-k2-thinking-turbo.toml deleted file mode 100644 index 8cd5c98b080..00000000000 --- a/providers/zenmux/models/moonshotai/kimi-k2-thinking-turbo.toml +++ /dev/null @@ -1,25 +0,0 @@ -name = "Kimi K2 Thinking Turbo" -description = "Kimi reasoning model for long-horizon research, planning, and tool use" -release_date = "2025-11-06" -last_updated = "2025-11-06" -attachment = false -reasoning = true -reasoning_options = [] -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false -status = "deprecated" - -[cost] -input = 1.15 -output = 8.00 -cache_read = 0.15 - -[limit] -context = 262_000 -output = 64_000 - -[modalities] -input = ["text"] -output = ["text"] diff --git a/providers/zenmux/models/moonshotai/kimi-k2-thinking.toml b/providers/zenmux/models/moonshotai/kimi-k2-thinking.toml deleted file mode 100644 index 7760c8ebb87..00000000000 --- a/providers/zenmux/models/moonshotai/kimi-k2-thinking.toml +++ /dev/null @@ -1,25 +0,0 @@ -name = "Kimi K2 Thinking" -description = "Kimi reasoning model for long-horizon research, planning, and tool use" -release_date = "2025-11-06" -last_updated = "2025-11-06" -attachment = false -reasoning = true -reasoning_options = [] -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false -status = "deprecated" - -[cost] -input = 0.60 -output = 2.50 -cache_read = 0.15 - -[limit] -context = 262_000 -output = 64_000 - -[modalities] -input = ["text"] -output = ["text"] diff --git a/providers/zenmux/models/moonshotai/kimi-k2.5.toml b/providers/zenmux/models/moonshotai/kimi-k2.5.toml deleted file mode 100644 index 417b6fef769..00000000000 --- a/providers/zenmux/models/moonshotai/kimi-k2.5.toml +++ /dev/null @@ -1,27 +0,0 @@ -name = "Kimi K2.5" -description = "Kimi multimodal agent model for visual understanding, coding, and planning" -release_date = "2026-01-27" -last_updated = "2026-01-27" -attachment = true -reasoning = true -reasoning_options = [{ type = "toggle" }] -temperature = false -tool_call = true -knowledge = "2025-01-01" -open_weights = false - -[cost] -input = 0.58 -output = 3.02 -cache_read = 0.10 - -[limit] -context = 262_000 -output = 64_000 - -[interleaved] -field = "reasoning_content" - -[modalities] -input = ["text", "image", "video"] -output = ["text"] diff --git a/providers/zenmux/models/moonshotai/kimi-k2.6.toml b/providers/zenmux/models/moonshotai/kimi-k2.6.toml index c0feb549167..e12b9307b15 100644 --- a/providers/zenmux/models/moonshotai/kimi-k2.6.toml +++ b/providers/zenmux/models/moonshotai/kimi-k2.6.toml @@ -1,27 +1,14 @@ -name = "Kimi K2.6" -description = "Kimi multimodal agent model for visual understanding, coding, and planning" -release_date = "2026-04-20" -last_updated = "2026-04-20" -attachment = true -reasoning = true -reasoning_options = [{ type = "toggle" }] -temperature = false -tool_call = true -knowledge = "2025-01-01" -open_weights = true +# Toggle: thinking.type = enabled|disabled|adaptive +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "moonshotai/kimi-k2.6" + +[interleaved] +field = "reasoning_content" + +[[reasoning_options]] +type = "toggle" [cost] input = 0.95 output = 4 cache_read = 0.16 - -[limit] -context = 262_140 -output = 262_140 - -[interleaved] -field = "reasoning_content" - -[modalities] -input = ["text", "image", "video"] -output = ["text"] diff --git a/providers/zenmux/models/moonshotai/kimi-k2.7-code-free.toml b/providers/zenmux/models/moonshotai/kimi-k2.7-code-free.toml deleted file mode 100644 index b8198b1b767..00000000000 --- a/providers/zenmux/models/moonshotai/kimi-k2.7-code-free.toml +++ /dev/null @@ -1,8 +0,0 @@ -base_model="moonshotai/kimi-k2.7-code" -reasoning_options = [] -name = "Kimi K2.7 Code (Free)" - -[cost] -input = 0 -output = 0 -cache_read = 0 diff --git a/providers/zenmux/models/moonshotai/kimi-k2.7-code-highspeed.toml b/providers/zenmux/models/moonshotai/kimi-k2.7-code-highspeed.toml new file mode 100644 index 00000000000..f115142dc64 --- /dev/null +++ b/providers/zenmux/models/moonshotai/kimi-k2.7-code-highspeed.toml @@ -0,0 +1,13 @@ +# Response side channel: reasoning_content +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "moonshotai/kimi-k2.7-code-highspeed" +name = "Kimi K2.7 Code HighSpeed" +reasoning_options = [] + +[interleaved] +field = "reasoning_content" + +[cost] +input = 1.9 +output = 8 +cache_read = 0.38 diff --git a/providers/zenmux/models/moonshotai/kimi-k2.7-code.toml b/providers/zenmux/models/moonshotai/kimi-k2.7-code.toml index 7e744c06b46..684a5746f3a 100644 --- a/providers/zenmux/models/moonshotai/kimi-k2.7-code.toml +++ b/providers/zenmux/models/moonshotai/kimi-k2.7-code.toml @@ -1,7 +1,12 @@ -base_model="moonshotai/kimi-k2.7-code" +# Response side channel: reasoning_content +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "moonshotai/kimi-k2.7-code" reasoning_options = [] +[interleaved] +field = "reasoning_content" + [cost] input = 0.95 output = 4 -cache_read = 0.16 +cache_read = 0.19 diff --git a/providers/zenmux/models/moonshotai/kimi-k2.8-preview.toml b/providers/zenmux/models/moonshotai/kimi-k2.8-preview.toml new file mode 100644 index 00000000000..7c8c6d010c5 --- /dev/null +++ b/providers/zenmux/models/moonshotai/kimi-k2.8-preview.toml @@ -0,0 +1,24 @@ +# Toggle: thinking.type = enabled|disabled|adaptive +# Effort: output_config.effort = low|high|max +# ZenMux's 2026-09-19 catalog explicitly reports max_completion_tokens=1048576. +# https://zenmux.ai/docs/guide/advanced/reasoning.html +# https://zenmux.ai/docs/api/openai/openai-list-models.html +base_model = "moonshotai/kimi-k2.8-preview" + +[interleaved] +field = "reasoning_content" + +[[reasoning_options]] +type = "toggle" + +[[reasoning_options]] +type = "effort" +values = ["low", "high", "max"] + +[cost] +input = 1 +output = 4 +cache_read = 0.25 + +[limit] +output = 1_048_576 diff --git a/providers/zenmux/models/moonshotai/kimi-k3-free.toml b/providers/zenmux/models/moonshotai/kimi-k3-free.toml deleted file mode 100644 index 41c692633b2..00000000000 --- a/providers/zenmux/models/moonshotai/kimi-k3-free.toml +++ /dev/null @@ -1,21 +0,0 @@ -# API: thinking.type = "enabled" | "disabled" | "adaptive"; in adaptive mode the -# reasoning depth is set by output_config.effort, which only accepts "max". -# Pricing: free tier via ZenMux (accessed 2026-07-20). -# https://docs.zenmux.ai -base_model = "moonshotai/kimi-k3" -name = "Kimi K3 (Free)" - -[[reasoning_options]] -type = "toggle" - -[[reasoning_options]] -type = "effort" -values = ["max"] - -[interleaved] -field = "reasoning_content" - -[cost] -input = 0 -output = 0 -cache_read = 0 diff --git a/providers/zenmux/models/moonshotai/kimi-k3.toml b/providers/zenmux/models/moonshotai/kimi-k3.toml index 942632af006..b24ebb3223d 100644 --- a/providers/zenmux/models/moonshotai/kimi-k3.toml +++ b/providers/zenmux/models/moonshotai/kimi-k3.toml @@ -1,19 +1,22 @@ -# API: thinking.type = "enabled" | "disabled" | "adaptive"; in adaptive mode the -# reasoning depth is set by output_config.effort, which only accepts "max". -# Pricing: https://platform.kimi.ai/docs/pricing/chat-k3 (accessed 2026-07-17). +# Toggle: thinking.type = enabled|disabled|adaptive +# Effort: output_config.effort = low|high|max +# https://zenmux.ai/docs/guide/advanced/reasoning.html base_model = "moonshotai/kimi-k3" +[interleaved] +field = "reasoning_content" + [[reasoning_options]] type = "toggle" [[reasoning_options]] type = "effort" -values = ["max"] - -[interleaved] -field = "reasoning_content" +values = ["low", "high", "max"] [cost] -input = 3.0 -output = 15.0 +input = 3 +output = 15 cache_read = 0.3 + +[limit] +output = 262_144 diff --git a/providers/zenmux/models/openai/chat-latest.toml b/providers/zenmux/models/openai/chat-latest.toml new file mode 100644 index 00000000000..fe7f147f529 --- /dev/null +++ b/providers/zenmux/models/openai/chat-latest.toml @@ -0,0 +1,11 @@ +# ZenMux's 2026-09-19 OpenAI- and Anthropic-compatible model lists both report +# reasoning disabled for this alias, so this is an intentional host override. +# https://zenmux.ai/docs/api/openai/openai-list-models.html +base_model = "openai/gpt-5.5-instant" +name = "Chat Latest (GPT-5.5 Instant)" +reasoning = false + +[cost] +input = 5 +output = 30 +cache_read = 0.5 diff --git a/providers/zenmux/models/openai/gpt-4.1-mini.toml b/providers/zenmux/models/openai/gpt-4.1-mini.toml new file mode 100644 index 00000000000..a6b4371ba16 --- /dev/null +++ b/providers/zenmux/models/openai/gpt-4.1-mini.toml @@ -0,0 +1,7 @@ +base_model = "openai/gpt-4.1-mini" +name = "GPT-4.1 Mini" + +[cost] +input = 0.4 +output = 1.6 +cache_read = 0.1 diff --git a/providers/zenmux/models/openai/gpt-4.1-nano.toml b/providers/zenmux/models/openai/gpt-4.1-nano.toml new file mode 100644 index 00000000000..959cb906795 --- /dev/null +++ b/providers/zenmux/models/openai/gpt-4.1-nano.toml @@ -0,0 +1,10 @@ +base_model = "openai/gpt-4.1-nano" +name = "GPT-4.1 Nano" + +[cost] +input = 0.1 +output = 0.4 +cache_read = 0.025 + +[modalities] +input = ["image", "text", "pdf"] diff --git a/providers/zenmux/models/openai/gpt-4.1.toml b/providers/zenmux/models/openai/gpt-4.1.toml new file mode 100644 index 00000000000..1fdb518f628 --- /dev/null +++ b/providers/zenmux/models/openai/gpt-4.1.toml @@ -0,0 +1,6 @@ +base_model = "openai/gpt-4.1" + +[cost] +input = 2 +output = 8 +cache_read = 0.5 diff --git a/providers/zenmux/models/openai/gpt-4o-mini.toml b/providers/zenmux/models/openai/gpt-4o-mini.toml new file mode 100644 index 00000000000..800564f45fd --- /dev/null +++ b/providers/zenmux/models/openai/gpt-4o-mini.toml @@ -0,0 +1,7 @@ +base_model = "openai/gpt-4o-mini" +name = "GPT-4o-mini" + +[cost] +input = 0.15 +output = 0.6 +cache_read = 0.075 diff --git a/providers/zenmux/models/openai/gpt-4o.toml b/providers/zenmux/models/openai/gpt-4o.toml new file mode 100644 index 00000000000..1b57e520127 --- /dev/null +++ b/providers/zenmux/models/openai/gpt-4o.toml @@ -0,0 +1,6 @@ +base_model = "openai/gpt-4o" + +[cost] +input = 2.5 +output = 10 +cache_read = 1.25 diff --git a/providers/zenmux/models/openai/gpt-5-codex.toml b/providers/zenmux/models/openai/gpt-5-codex.toml index 4a42dd4975e..29d05b941ee 100644 --- a/providers/zenmux/models/openai/gpt-5-codex.toml +++ b/providers/zenmux/models/openai/gpt-5-codex.toml @@ -1,27 +1,20 @@ +# Effort: reasoning_effort = low|medium|high +# ZenMux's 2026-09-19 model lists advertise text and image input for this route, +# so attachment=true is an intentional host capability override. +# https://zenmux.ai/docs/guide/advanced/reasoning.html +# https://zenmux.ai/docs/api/openai/openai-list-models.html +base_model = "openai/gpt-5-codex" name = "GPT-5 Codex" -description = "Coding-optimized GPT model for repository edits, reviews, and agentic software work" -release_date = "2025-09-23" -last_updated = "2025-09-23" attachment = true -reasoning = true -reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high"] [cost] input = 1.25 -output = 10.00 -cache_read = 0.12 - -[limit] -context = 400_000 -output = 64_000 - -[modalities] -input = ["text", "image"] -output = ["text"] +output = 10 +cache_read = 0.125 [provider] npm = "@ai-sdk/openai" diff --git a/providers/zenmux/models/openai/gpt-5-mini.toml b/providers/zenmux/models/openai/gpt-5-mini.toml new file mode 100644 index 00000000000..882ad0ea7de --- /dev/null +++ b/providers/zenmux/models/openai/gpt-5-mini.toml @@ -0,0 +1,15 @@ +# Effort: reasoning_effort = minimal|low|medium|high +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "openai/gpt-5-mini" + +[[reasoning_options]] +type = "effort" +values = ["minimal", "low", "medium", "high"] + +[cost] +input = 0.25 +output = 2 +cache_read = 0.025 + +[modalities] +input = ["text", "image", "pdf"] diff --git a/providers/zenmux/models/openai/gpt-5-nano.toml b/providers/zenmux/models/openai/gpt-5-nano.toml new file mode 100644 index 00000000000..83e79ef5eb0 --- /dev/null +++ b/providers/zenmux/models/openai/gpt-5-nano.toml @@ -0,0 +1,15 @@ +# Effort: reasoning_effort = minimal|low|medium|high +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "openai/gpt-5-nano" + +[[reasoning_options]] +type = "effort" +values = ["minimal", "low", "medium", "high"] + +[cost] +input = 0.05 +output = 0.4 +cache_read = 0.005 + +[modalities] +input = ["text", "image", "pdf"] diff --git a/providers/zenmux/models/openai/gpt-5-pro.toml b/providers/zenmux/models/openai/gpt-5-pro.toml new file mode 100644 index 00000000000..12c316f426a --- /dev/null +++ b/providers/zenmux/models/openai/gpt-5-pro.toml @@ -0,0 +1,20 @@ +# Effort: reasoning_effort = high +# ZenMux's 2026-09-19 catalog and OpenAI-compatible model list explicitly +# report max_completion_tokens=128000 for this hosted route. +# https://zenmux.ai/docs/guide/advanced/reasoning.html +# https://zenmux.ai/docs/api/openai/openai-list-models.html +base_model = "openai/gpt-5-pro" + +[[reasoning_options]] +type = "effort" +values = ["high"] + +[cost] +input = 15 +output = 120 + +[limit] +output = 128_000 + +[modalities] +input = ["image", "text", "pdf"] diff --git a/providers/zenmux/models/openai/gpt-5.1-chat.toml b/providers/zenmux/models/openai/gpt-5.1-chat.toml deleted file mode 100644 index c35f6c1085e..00000000000 --- a/providers/zenmux/models/openai/gpt-5.1-chat.toml +++ /dev/null @@ -1,29 +0,0 @@ -name = "GPT-5.1 Chat" -description = "Chat-tuned GPT model for conversational assistance, writing, and tool workflows" -release_date = "2025-11-13" -last_updated = "2025-11-13" -attachment = true -reasoning = false -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false - -status = "deprecated" - -[cost] -input = 1.25 -output = 10.00 -cache_read = 0.12 - -[limit] -context = 128_000 -output = 64_000 - -[modalities] -input = ["pdf", "image", "text"] -output = ["text"] - -[provider] -npm = "@ai-sdk/openai" -api = "https://zenmux.ai/api/v1" diff --git a/providers/zenmux/models/openai/gpt-5.1-codex-mini.toml b/providers/zenmux/models/openai/gpt-5.1-codex-mini.toml index 323be60fb0d..b5b6086fd90 100644 --- a/providers/zenmux/models/openai/gpt-5.1-codex-mini.toml +++ b/providers/zenmux/models/openai/gpt-5.1-codex-mini.toml @@ -1,27 +1,19 @@ +# Effort: reasoning_effort = low|medium|high +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "openai/gpt-5.1-codex-mini" name = "GPT-5.1-Codex-Mini" -description = "Coding-optimized GPT model for repository edits, reviews, and agentic software work" -release_date = "2025-11-13" -last_updated = "2025-11-13" -attachment = true -reasoning = true -reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high"] [cost] input = 0.25 -output = 2.00 -cache_read = 0.03 +output = 2 +cache_read = 0.025 [limit] -context = 400_000 -output = 64_000 - -[modalities] -input = ["image", "text"] -output = ["text"] +output = 100_000 [provider] npm = "@ai-sdk/openai" diff --git a/providers/zenmux/models/openai/gpt-5.1-codex.toml b/providers/zenmux/models/openai/gpt-5.1-codex.toml index 716e439736d..523eb508304 100644 --- a/providers/zenmux/models/openai/gpt-5.1-codex.toml +++ b/providers/zenmux/models/openai/gpt-5.1-codex.toml @@ -1,27 +1,16 @@ +# Effort: reasoning_effort = low|medium|high +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "openai/gpt-5.1-codex" name = "GPT-5.1-Codex" -description = "Coding-optimized GPT model for repository edits, reviews, and agentic software work" -release_date = "2025-11-13" -last_updated = "2025-11-13" -attachment = true -reasoning = true -reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high"] [cost] input = 1.25 -output = 10.00 -cache_read = 0.12 - -[limit] -context = 400_000 -output = 64_000 - -[modalities] -input = ["text", "image"] -output = ["text"] +output = 10 +cache_read = 0.125 [provider] npm = "@ai-sdk/openai" diff --git a/providers/zenmux/models/openai/gpt-5.1.toml b/providers/zenmux/models/openai/gpt-5.1.toml index f4eee5a2053..90ba7e5bdd2 100644 --- a/providers/zenmux/models/openai/gpt-5.1.toml +++ b/providers/zenmux/models/openai/gpt-5.1.toml @@ -1,27 +1,18 @@ -name = "GPT-5.1" -description = "GPT model for general reasoning, writing, coding, and tool-assisted tasks" -release_date = "2025-11-13" -last_updated = "2025-11-13" -attachment = true -reasoning = true -reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false +# Effort: reasoning_effort = none|low|medium|high +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "openai/gpt-5.1" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high"] [cost] input = 1.25 -output = 10.00 -cache_read = 0.12 - -[limit] -context = 400_000 -output = 64_000 +output = 10 +cache_read = 0.125 [modalities] input = ["image", "text", "pdf"] -output = ["text"] [provider] npm = "@ai-sdk/openai" diff --git a/providers/zenmux/models/openai/gpt-5.2-codex.toml b/providers/zenmux/models/openai/gpt-5.2-codex.toml index f7ce8937dbd..3d30d33c8d0 100644 --- a/providers/zenmux/models/openai/gpt-5.2-codex.toml +++ b/providers/zenmux/models/openai/gpt-5.2-codex.toml @@ -1,27 +1,16 @@ +# Effort: reasoning_effort = low|medium|high +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "openai/gpt-5.2-codex" name = "GPT-5.2-Codex" -description = "Coding-optimized GPT model for repository edits, reviews, and agentic software work" -release_date = "2026-01-15" -last_updated = "2026-01-15" -attachment = true -reasoning = true -reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] -temperature = false -tool_call = true -knowledge = "2025-01-01" -open_weights = false + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high"] [cost] input = 1.75 -output = 14.00 -cache_read = 0.17 - -[limit] -context = 400_000 -output = 64_000 - -[modalities] -input = ["text", "image", "pdf"] -output = ["text"] +output = 14 +cache_read = 0.175 [provider] npm = "@ai-sdk/openai" diff --git a/providers/zenmux/models/openai/gpt-5.2-pro.toml b/providers/zenmux/models/openai/gpt-5.2-pro.toml index f535798c000..30cf6fc1f61 100644 --- a/providers/zenmux/models/openai/gpt-5.2-pro.toml +++ b/providers/zenmux/models/openai/gpt-5.2-pro.toml @@ -1,26 +1,17 @@ -name = "GPT-5.2-Pro" -description = "Frontier GPT model for professional reasoning, coding, and multimodal work" -release_date = "2025-12-11" -last_updated = "2025-12-11" -attachment = true -reasoning = true -reasoning_options = [{ type = "effort", values = ["medium", "high", "xhigh"] }] -temperature = false -tool_call = true -knowledge = "2025-08-31" -open_weights = false +# Effort: reasoning_effort = medium|high|xhigh +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "openai/gpt-5.2-pro" -[cost] -input = 21.00 -output = 168.00 +[[reasoning_options]] +type = "effort" +values = ["medium", "high", "xhigh"] -[limit] -context = 400_000 -output = 128_000 +[cost] +input = 21 +output = 168 [modalities] -input = ["text", "image", "pdf"] -output = ["text"] +input = ["image", "text", "pdf"] [provider] npm = "@ai-sdk/openai" diff --git a/providers/zenmux/models/openai/gpt-5.2.toml b/providers/zenmux/models/openai/gpt-5.2.toml index c0b1515bc5c..1f83b79beff 100644 --- a/providers/zenmux/models/openai/gpt-5.2.toml +++ b/providers/zenmux/models/openai/gpt-5.2.toml @@ -1,27 +1,18 @@ -name = "GPT-5.2" -description = "GPT model for general reasoning, writing, coding, and tool-assisted tasks" -release_date = "2025-12-11" -last_updated = "2025-12-11" -attachment = true -reasoning = true -reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] -temperature = false -tool_call = true -knowledge = "2025-01-01" -open_weights = false +# Effort: reasoning_effort = none|low|medium|high|xhigh +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "openai/gpt-5.2" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh"] [cost] input = 1.75 -output = 14.00 -cache_read = 0.17 - -[limit] -context = 400_000 -output = 64_000 +output = 14 +cache_read = 0.175 [modalities] input = ["image", "text", "pdf"] -output = ["text"] [provider] npm = "@ai-sdk/openai" diff --git a/providers/zenmux/models/openai/gpt-5.3-chat.toml b/providers/zenmux/models/openai/gpt-5.3-chat.toml deleted file mode 100644 index 7caedff95f8..00000000000 --- a/providers/zenmux/models/openai/gpt-5.3-chat.toml +++ /dev/null @@ -1,26 +0,0 @@ -name = "GPT-5.3 Chat" -description = "Chat-tuned GPT model for conversational assistance, writing, and tool workflows" -release_date = "2026-03-20" -last_updated = "2026-03-20" -attachment = true -reasoning = false -temperature = true -tool_call = true -knowledge = "2025-08-31" -open_weights = false - -[cost] -input = 1.75 -output = 14.00 - -[limit] -context = 128_000 -output = 16_380 - -[modalities] -input = ["text"] -output = ["text"] - -[provider] -npm = "@ai-sdk/openai" -api = "https://zenmux.ai/api/v1" diff --git a/providers/zenmux/models/openai/gpt-5.3-codex.toml b/providers/zenmux/models/openai/gpt-5.3-codex.toml index ff999ce849a..9345a1cf537 100644 --- a/providers/zenmux/models/openai/gpt-5.3-codex.toml +++ b/providers/zenmux/models/openai/gpt-5.3-codex.toml @@ -1,26 +1,16 @@ -name = "GPT-5.3 Codex" -description = "Coding-optimized GPT model for repository edits, reviews, and agentic software work" -release_date = "2026-03-20" -last_updated = "2026-03-20" -attachment = true -reasoning = true -reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] -temperature = true -tool_call = true -knowledge = "2025-08-31" -open_weights = false +# Effort: reasoning_effort = none|low|medium|high|xhigh +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "openai/gpt-5.3-codex" +name = "GPT-5.3-Codex" + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh"] [cost] input = 1.75 -output = 14.00 - -[limit] -context = 400_000 -output = 128_000 - -[modalities] -input = ["text"] -output = ["text"] +output = 14 +cache_read = 0.175 [provider] npm = "@ai-sdk/openai" diff --git a/providers/zenmux/models/openai/gpt-5.4-mini.toml b/providers/zenmux/models/openai/gpt-5.4-mini.toml index 7a14a30e636..e3bfc06e5c4 100644 --- a/providers/zenmux/models/openai/gpt-5.4-mini.toml +++ b/providers/zenmux/models/openai/gpt-5.4-mini.toml @@ -1,25 +1,19 @@ +# Effort: reasoning_effort = none|low|medium|high|xhigh +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "openai/gpt-5.4-mini" name = "GPT-5.4 Mini" -description = "Compact GPT model for low-latency assistance and high-volume workloads" -release_date = "2026-03-20" -last_updated = "2026-03-20" -attachment = true -reasoning = false -temperature = true -tool_call = true -knowledge = "2025-08-31" -open_weights = false + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh"] [cost] input = 0.75 -output = 4.50 - -[limit] -context = 400_000 -output = 128_000 +output = 4.5 +cache_read = 0.075 [modalities] -input = ["text"] -output = ["text"] +input = ["text", "image", "pdf"] [provider] npm = "@ai-sdk/openai" diff --git a/providers/zenmux/models/openai/gpt-5.4-nano.toml b/providers/zenmux/models/openai/gpt-5.4-nano.toml index 5e78fa53e0f..9e7361d40c0 100644 --- a/providers/zenmux/models/openai/gpt-5.4-nano.toml +++ b/providers/zenmux/models/openai/gpt-5.4-nano.toml @@ -1,25 +1,19 @@ +# Effort: reasoning_effort = none|low|medium|high|xhigh +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "openai/gpt-5.4-nano" name = "GPT-5.4 Nano" -description = "Compact GPT model for low-latency assistance and high-volume workloads" -release_date = "2026-03-20" -last_updated = "2026-03-20" -attachment = false -reasoning = false -temperature = true -tool_call = true -knowledge = "2025-08-31" -open_weights = false + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh"] [cost] -input = 0.20 +input = 0.2 output = 1.25 - -[limit] -context = 400_000 -output = 128_000 +cache_read = 0.02 [modalities] -input = ["text"] -output = ["text"] +input = ["text", "image", "pdf"] [provider] npm = "@ai-sdk/openai" diff --git a/providers/zenmux/models/openai/gpt-5.4-pro.toml b/providers/zenmux/models/openai/gpt-5.4-pro.toml index 2a4733d120e..f394261d27c 100644 --- a/providers/zenmux/models/openai/gpt-5.4-pro.toml +++ b/providers/zenmux/models/openai/gpt-5.4-pro.toml @@ -1,26 +1,22 @@ -name = "GPT-5.4 Pro" -description = "Frontier GPT model for professional reasoning, coding, and multimodal work" -release_date = "2026-03-20" -last_updated = "2026-03-20" -attachment = true -reasoning = true -reasoning_options = [{ type = "effort", values = ["medium", "high", "xhigh"] }] -temperature = true -tool_call = true -knowledge = "2025-08-31" -open_weights = false +# Effort: reasoning_effort = medium|high|xhigh +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "openai/gpt-5.4-pro" + +[[reasoning_options]] +type = "effort" +values = ["medium", "high", "xhigh"] [cost] -input = 45.00 -output = 225.00 +input = 30 +output = 180 -[limit] -context = 1_050_000 -output = 128_000 +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 60 +output = 270 [modalities] -input = ["text", "image"] -output = ["text"] +input = ["text", "image", "pdf"] [provider] npm = "@ai-sdk/openai" diff --git a/providers/zenmux/models/openai/gpt-5.4.toml b/providers/zenmux/models/openai/gpt-5.4.toml index e728524101d..7d95dfc2d79 100644 --- a/providers/zenmux/models/openai/gpt-5.4.toml +++ b/providers/zenmux/models/openai/gpt-5.4.toml @@ -1,26 +1,21 @@ -name = "GPT-5.4" -description = "Frontier GPT model for professional reasoning, coding, and multimodal work" -release_date = "2026-03-20" -last_updated = "2026-03-20" -attachment = true -reasoning = true -reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] -temperature = true -tool_call = true -knowledge = "2025-08-31" -open_weights = false +# Effort: reasoning_effort = none|low|medium|high|xhigh +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "openai/gpt-5.4" -[cost] -input = 3.75 -output = 18.75 +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh"] -[limit] -context = 1_050_000 -output = 128_000 +[cost] +input = 2.5 +output = 15 +cache_read = 0.25 -[modalities] -input = ["text", "image"] -output = ["text"] +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 5 +output = 22.5 +cache_read = 0.5 [provider] npm = "@ai-sdk/openai" diff --git a/providers/zenmux/models/openai/gpt-5.5-instant.toml b/providers/zenmux/models/openai/gpt-5.5-instant.toml deleted file mode 100644 index ffcea046e7a..00000000000 --- a/providers/zenmux/models/openai/gpt-5.5-instant.toml +++ /dev/null @@ -1,7 +0,0 @@ -base_model = "openai/gpt-5.5-instant" -reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] - -[cost] -input = 5 -output = 30 -cache_read = 0.5 diff --git a/providers/zenmux/models/openai/gpt-5.5-pro.toml b/providers/zenmux/models/openai/gpt-5.5-pro.toml index 523edc05baa..00568d94c9b 100644 --- a/providers/zenmux/models/openai/gpt-5.5-pro.toml +++ b/providers/zenmux/models/openai/gpt-5.5-pro.toml @@ -1,5 +1,10 @@ +# Effort: reasoning_effort = medium|high|xhigh +# https://zenmux.ai/docs/guide/advanced/reasoning.html base_model = "openai/gpt-5.5-pro" -reasoning_options = [{ type = "effort", values = ["medium", "high", "xhigh"] }] + +[[reasoning_options]] +type = "effort" +values = ["medium", "high", "xhigh"] [cost] input = 30 diff --git a/providers/zenmux/models/openai/gpt-5.5.toml b/providers/zenmux/models/openai/gpt-5.5.toml index 0871b3e9a6d..655ffa5169a 100644 --- a/providers/zenmux/models/openai/gpt-5.5.toml +++ b/providers/zenmux/models/openai/gpt-5.5.toml @@ -1,5 +1,10 @@ +# Effort: reasoning_effort = none|low|medium|high|xhigh +# https://zenmux.ai/docs/guide/advanced/reasoning.html base_model = "openai/gpt-5.5" -reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh"] [cost] input = 5 diff --git a/providers/zenmux/models/openai/gpt-5.6-luna.toml b/providers/zenmux/models/openai/gpt-5.6-luna.toml index 89bfa05d30e..d17ca8c99ae 100644 --- a/providers/zenmux/models/openai/gpt-5.6-luna.toml +++ b/providers/zenmux/models/openai/gpt-5.6-luna.toml @@ -1,15 +1,20 @@ +# Effort: reasoning_effort = none|low|medium|high|xhigh|max +# https://zenmux.ai/docs/guide/advanced/reasoning.html base_model = "openai/gpt-5.6-luna" -reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh", "max"] }] + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh", "max"] [cost] -input = 1.00 -output = 6.00 -cache_read = 0.10 -cache_write = 1.25 +input = 0.2 +output = 1.2 +cache_read = 0.02 +cache_write = 0.25 [[cost.tiers]] -tier = { size = 272_000 } -input = 2.00 -output = 9.00 -cache_read = 0.20 -cache_write = 2.50 +tier = { type = "context", size = 272_000 } +input = 0.4 +output = 1.8 +cache_read = 0.04 +cache_write = 0.5 diff --git a/providers/zenmux/models/openai/gpt-5.6-sol.toml b/providers/zenmux/models/openai/gpt-5.6-sol.toml index c2917934cd6..c07bee7b579 100644 --- a/providers/zenmux/models/openai/gpt-5.6-sol.toml +++ b/providers/zenmux/models/openai/gpt-5.6-sol.toml @@ -1,15 +1,20 @@ +# Effort: reasoning_effort = none|low|medium|high|xhigh|max +# https://zenmux.ai/docs/guide/advanced/reasoning.html base_model = "openai/gpt-5.6-sol" -reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh", "max"] }] + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh", "max"] [cost] -input = 5.00 -output = 30.00 -cache_read = 0.50 -cache_write = 6.25 +input = 4 +output = 20 +cache_read = 0.4 +cache_write = 5 [[cost.tiers]] -tier = { size = 272_000 } -input = 10.00 -output = 45.00 -cache_read = 1.00 -cache_write = 12.50 +tier = { type = "context", size = 272_000 } +input = 8 +output = 30 +cache_read = 0.8 +cache_write = 10 diff --git a/providers/zenmux/models/openai/gpt-5.6-terra.toml b/providers/zenmux/models/openai/gpt-5.6-terra.toml index 6e8e33b9d70..c41799a88d9 100644 --- a/providers/zenmux/models/openai/gpt-5.6-terra.toml +++ b/providers/zenmux/models/openai/gpt-5.6-terra.toml @@ -1,15 +1,20 @@ +# Effort: reasoning_effort = none|low|medium|high|xhigh|max +# https://zenmux.ai/docs/guide/advanced/reasoning.html base_model = "openai/gpt-5.6-terra" -reasoning_options = [{ type = "effort", values = ["none", "low", "medium", "high", "xhigh", "max"] }] + +[[reasoning_options]] +type = "effort" +values = ["none", "low", "medium", "high", "xhigh", "max"] [cost] -input = 2.50 -output = 15.00 -cache_read = 0.25 -cache_write = 3.125 +input = 2 +output = 12 +cache_read = 0.2 +cache_write = 2.5 [[cost.tiers]] -tier = { size = 272_000 } -input = 5.00 -output = 22.50 -cache_read = 0.50 -cache_write = 6.25 +tier = { type = "context", size = 272_000 } +input = 4 +output = 18 +cache_read = 0.4 +cache_write = 5 diff --git a/providers/zenmux/models/openai/gpt-5.toml b/providers/zenmux/models/openai/gpt-5.toml index eff8cd1dd0d..52789e8112a 100644 --- a/providers/zenmux/models/openai/gpt-5.toml +++ b/providers/zenmux/models/openai/gpt-5.toml @@ -1,27 +1,18 @@ -name = "GPT-5" -description = "GPT model for general reasoning, writing, coding, and tool-assisted tasks" -release_date = "2025-08-07" -last_updated = "2025-08-07" -attachment = true -reasoning = true -reasoning_options = [{ type = "effort", values = ["low", "medium", "high"] }] -temperature = true -tool_call = true -knowledge = "2025-01-01" -open_weights = false +# Effort: reasoning_effort = minimal|low|medium|high +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "openai/gpt-5" + +[[reasoning_options]] +type = "effort" +values = ["minimal", "low", "medium", "high"] [cost] input = 1.25 -output = 10.00 -cache_read = 0.12 - -[limit] -context = 400_000 -output = 64_000 +output = 10 +cache_read = 0.125 [modalities] input = ["text", "image", "pdf"] -output = ["text"] [provider] npm = "@ai-sdk/openai" diff --git a/providers/zenmux/models/openai/gpt-6-astra.toml b/providers/zenmux/models/openai/gpt-6-astra.toml new file mode 100644 index 00000000000..88b339d94ce --- /dev/null +++ b/providers/zenmux/models/openai/gpt-6-astra.toml @@ -0,0 +1,20 @@ +# Effort: reasoning_effort = low|medium|high|xhigh|max +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "openai/gpt-6-astra" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high", "xhigh", "max"] + +[cost] +input = 10 +output = 50 +cache_read = 1 +cache_write = 12.5 + +[[cost.tiers]] +tier = { type = "context", size = 272_000 } +input = 20 +output = 75 +cache_read = 2 +cache_write = 25 diff --git a/providers/zenmux/models/openai/gpt-image-1.5.toml b/providers/zenmux/models/openai/gpt-image-1.5.toml new file mode 100644 index 00000000000..86f1dc8d549 --- /dev/null +++ b/providers/zenmux/models/openai/gpt-image-1.5.toml @@ -0,0 +1,12 @@ +base_model = "openai/gpt-image-1.5" + +[cost] +input = 5 +output = 32 +cache_read = 1.25 + +[limit] +context = 10_000 + +[modalities] +output = ["image"] diff --git a/providers/zenmux/models/openai/gpt-image-2.5-flare.toml b/providers/zenmux/models/openai/gpt-image-2.5-flare.toml new file mode 100644 index 00000000000..78d13aa01e2 --- /dev/null +++ b/providers/zenmux/models/openai/gpt-image-2.5-flare.toml @@ -0,0 +1,10 @@ +base_model = "openai/gpt-image-2.5-flare" +name = "GPT-Image-2.5-Flare" + +[cost] +input = 5 +output = 30 +cache_read = 1.25 + +[limit] +context = 10_000 diff --git a/providers/zenmux/models/openai/gpt-image-2.5-sunburst.toml b/providers/zenmux/models/openai/gpt-image-2.5-sunburst.toml new file mode 100644 index 00000000000..1370ade1ef9 --- /dev/null +++ b/providers/zenmux/models/openai/gpt-image-2.5-sunburst.toml @@ -0,0 +1,10 @@ +base_model = "openai/gpt-image-2.5-sunburst" +name = "GPT-Image-2.5-Sunburst" + +[cost] +input = 5 +output = 30 +cache_read = 1.25 + +[limit] +context = 10_000 diff --git a/providers/zenmux/models/openai/gpt-image-2.toml b/providers/zenmux/models/openai/gpt-image-2.toml new file mode 100644 index 00000000000..a1a10f4035c --- /dev/null +++ b/providers/zenmux/models/openai/gpt-image-2.toml @@ -0,0 +1,9 @@ +base_model = "openai/gpt-image-2" + +[cost] +input = 5 +output = 30 +cache_read = 1.25 + +[limit] +context = 10_000 diff --git a/providers/zenmux/models/openai/gpt-transcribe.toml b/providers/zenmux/models/openai/gpt-transcribe.toml new file mode 100644 index 00000000000..bf645f3a144 --- /dev/null +++ b/providers/zenmux/models/openai/gpt-transcribe.toml @@ -0,0 +1,4 @@ +# ZenMux bills this audio route at USD 0.000075 per second, not per token; +# the token-only cost schema cannot represent it, so cost is intentionally omitted. +# https://zenmux.ai/docs/api/openai/openai-list-models.html +base_model = "openai/gpt-transcribe" diff --git a/providers/zenmux/models/openai/o4-mini.toml b/providers/zenmux/models/openai/o4-mini.toml new file mode 100644 index 00000000000..5c4b20a0261 --- /dev/null +++ b/providers/zenmux/models/openai/o4-mini.toml @@ -0,0 +1,16 @@ +# Effort: reasoning_effort = low|medium|high +# https://zenmux.ai/docs/guide/advanced/reasoning.html +base_model = "openai/o4-mini" +name = "o4 Mini" + +[[reasoning_options]] +type = "effort" +values = ["low", "medium", "high"] + +[cost] +input = 1.1 +output = 4.4 +cache_read = 0.275 + +[modalities] +input = ["image", "text", "pdf"] diff --git a/providers/zenmux/models/openai/text-embedding-3-large.toml b/providers/zenmux/models/openai/text-embedding-3-large.toml new file mode 100644 index 00000000000..bac23d99196 --- /dev/null +++ b/providers/zenmux/models/openai/text-embedding-3-large.toml @@ -0,0 +1,12 @@ +# ZenMux's 2026-09-19 catalog reports context_length=8192. The inherited +# limit.output remains the lab embedding dimension rather than that token limit. +# https://zenmux.ai/docs/api/openai/openai-list-models.html +base_model = "openai/text-embedding-3-large" +name = "Text Embedding 3 Large" + +[cost] +input = 0.13 +output = 0 + +[limit] +context = 8_192 diff --git a/providers/zenmux/models/openai/text-embedding-3-small.toml b/providers/zenmux/models/openai/text-embedding-3-small.toml new file mode 100644 index 00000000000..7db9ea0c48a --- /dev/null +++ b/providers/zenmux/models/openai/text-embedding-3-small.toml @@ -0,0 +1,12 @@ +# ZenMux's 2026-09-19 catalog reports context_length=8192. The inherited +# limit.output remains the lab embedding dimension rather than that token limit. +# https://zenmux.ai/docs/api/openai/openai-list-models.html +base_model = "openai/text-embedding-3-small" +name = "Text Embedding 3 Small" + +[cost] +input = 0.02 +output = 0 + +[limit] +context = 8_192