diff --git a/models/mistral/mistral-embed.toml b/models/mistral/mistral-embed.toml new file mode 100644 index 00000000000..560b30b22fa --- /dev/null +++ b/models/mistral/mistral-embed.toml @@ -0,0 +1,18 @@ +name = "Mistral Embed" +description = "Embedding model for semantic search, retrieval, clustering, and ranking pipelines" +family = "mistral-embed" +release_date = "2023-12-11" +last_updated = "2023-12-11" +attachment = false +reasoning = false +temperature = false +tool_call = false +open_weights = false + +[limit] +context = 8_000 +output = 3_072 + +[modalities] +input = ["text"] +output = ["text"] diff --git a/providers/tempr/models/mistral/mistral-embed.toml b/providers/tempr/models/mistral/mistral-embed.toml new file mode 100644 index 00000000000..ee0d731f5d5 --- /dev/null +++ b/providers/tempr/models/mistral/mistral-embed.toml @@ -0,0 +1,5 @@ +base_model = "mistral/mistral-embed" + +[cost] +input = 0.1 +output = 0 diff --git a/providers/tempr/models/mistral/mistral-large-2512.toml b/providers/tempr/models/mistral/mistral-large-2512.toml new file mode 100644 index 00000000000..49df75a6342 --- /dev/null +++ b/providers/tempr/models/mistral/mistral-large-2512.toml @@ -0,0 +1,5 @@ +base_model = "mistral/mistral-large-2512" + +[cost] +input = 0.5 +output = 1.5 diff --git a/providers/tempr/models/mistral/mistral-large-latest.toml b/providers/tempr/models/mistral/mistral-large-latest.toml new file mode 100644 index 00000000000..c0c8d052341 --- /dev/null +++ b/providers/tempr/models/mistral/mistral-large-latest.toml @@ -0,0 +1,5 @@ +base_model = "mistral/mistral-large-latest" + +[cost] +input = 0.5 +output = 1.5 diff --git a/providers/tempr/models/mistral/mistral-medium-2604.toml b/providers/tempr/models/mistral/mistral-medium-2604.toml new file mode 100644 index 00000000000..8846aa0f71b --- /dev/null +++ b/providers/tempr/models/mistral/mistral-medium-2604.toml @@ -0,0 +1,14 @@ +# Tempr Gateway: https://temprhq.io/docs/gateway-chat-completions#reasoning +# Off: reasoning.effort = "none" (or reasoning.enabled = false) +# Effort: reasoning.effort = none|high (alias: top-level reasoning_effort) +# /v1/messages: thinking.type = enabled|disabled; output_config.effort = high +# /v1/responses: reasoning.effort = none|high +base_model = "mistral/mistral-medium-2604" + +reasoning_options = [ + { type = "effort", values = ["none", "high"] }, +] + +[cost] +input = 1.5 +output = 7.5 diff --git a/providers/tempr/models/mistral/mistral-medium-latest.toml b/providers/tempr/models/mistral/mistral-medium-latest.toml new file mode 100644 index 00000000000..626e92b789a --- /dev/null +++ b/providers/tempr/models/mistral/mistral-medium-latest.toml @@ -0,0 +1,14 @@ +# Tempr Gateway: https://temprhq.io/docs/gateway-chat-completions#reasoning +# Off: reasoning.effort = "none" (or reasoning.enabled = false) +# Effort: reasoning.effort = none|high (alias: top-level reasoning_effort) +# /v1/messages: thinking.type = enabled|disabled; output_config.effort = high +# /v1/responses: reasoning.effort = none|high +base_model = "mistral/mistral-medium-latest" + +reasoning_options = [ + { type = "effort", values = ["none", "high"] }, +] + +[cost] +input = 1.5 +output = 7.5 diff --git a/providers/tempr/models/mistral/mistral-small-2603.toml b/providers/tempr/models/mistral/mistral-small-2603.toml new file mode 100644 index 00000000000..c1d71163ed6 --- /dev/null +++ b/providers/tempr/models/mistral/mistral-small-2603.toml @@ -0,0 +1,14 @@ +# Tempr Gateway: https://temprhq.io/docs/gateway-chat-completions#reasoning +# Off: reasoning.effort = "none" (or reasoning.enabled = false) +# Effort: reasoning.effort = none|high (alias: top-level reasoning_effort) +# /v1/messages: thinking.type = enabled|disabled; output_config.effort = high +# /v1/responses: reasoning.effort = none|high +base_model = "mistral/mistral-small-2603" + +reasoning_options = [ + { type = "effort", values = ["none", "high"] }, +] + +[cost] +input = 0.15 +output = 0.6 diff --git a/providers/tempr/models/mistral/mistral-small-latest.toml b/providers/tempr/models/mistral/mistral-small-latest.toml new file mode 100644 index 00000000000..d1a5aaeca24 --- /dev/null +++ b/providers/tempr/models/mistral/mistral-small-latest.toml @@ -0,0 +1,14 @@ +# Tempr Gateway: https://temprhq.io/docs/gateway-chat-completions#reasoning +# Off: reasoning.effort = "none" (or reasoning.enabled = false) +# Effort: reasoning.effort = none|high (alias: top-level reasoning_effort) +# /v1/messages: thinking.type = enabled|disabled; output_config.effort = high +# /v1/responses: reasoning.effort = none|high +base_model = "mistral/mistral-small-latest" + +reasoning_options = [ + { type = "effort", values = ["none", "high"] }, +] + +[cost] +input = 0.15 +output = 0.6 diff --git a/providers/tempr/models/mistral/voxtral-small-latest.toml b/providers/tempr/models/mistral/voxtral-small-latest.toml new file mode 100644 index 00000000000..b15af0ee507 --- /dev/null +++ b/providers/tempr/models/mistral/voxtral-small-latest.toml @@ -0,0 +1,5 @@ +base_model = "mistral/voxtral-small-latest" + +[cost] +input = 0.1 +output = 0.3 diff --git a/providers/tempr/models/mistral/zai-glm-5-2.toml b/providers/tempr/models/mistral/zai-glm-5-2.toml new file mode 100644 index 00000000000..dcc6e184ad6 --- /dev/null +++ b/providers/tempr/models/mistral/zai-glm-5-2.toml @@ -0,0 +1,16 @@ +# Tempr Gateway: https://temprhq.io/docs/gateway-chat-completions#reasoning +# Off: reasoning.effort = "none" (or reasoning.enabled = false) +# Effort: reasoning.effort = none|high|max (alias: top-level reasoning_effort) +# /v1/messages: thinking.type = enabled|disabled; output_config.effort = high|max +# /v1/responses: reasoning.effort = none|high|max +base_model = "zhipuai/glm-5.2" +status = "beta" + +reasoning_options = [ + { type = "effort", values = ["none", "high", "max"] }, +] + +[cost] +input = 1.4 +output = 4.4 +cache_read = 0.14 diff --git a/providers/tempr/models/mistral/zai-glm-5-3.toml b/providers/tempr/models/mistral/zai-glm-5-3.toml new file mode 100644 index 00000000000..19974b909bd --- /dev/null +++ b/providers/tempr/models/mistral/zai-glm-5-3.toml @@ -0,0 +1,14 @@ +# Tempr Gateway: https://temprhq.io/docs/gateway-chat-completions#reasoning +# Effort: reasoning.effort = low|high|max (alias: top-level reasoning_effort) +# /v1/messages: output_config.effort = low|high|max +# /v1/responses: reasoning.effort = low|high|max +base_model = "zhipuai/glm-5.3" + +reasoning_options = [ + { type = "effort", values = ["low", "high", "max"] }, +] + +[cost] +input = 1.4 +output = 4.4 +cache_read = 0.14