fix: use max_tokens for OpenAI-compatible endpoints with custom base URL (#858)
Mistral (and several other providers) reject 'max_completion_tokens' with a 422 because they haven't adopted the newer OpenAI parameter name. When the openai provider is configured with a custom base_url (e.g. Mistral, Together AI), fall back to the widely-supported 'max_tokens' parameter. Native OpenAI (no custom base_url) and Groq still use 'max_completion_tokens'. Fixes #852
This commit is contained in:
parent
cd4b3e96e2
commit
cd99eef4c5
1 changed files with 19 additions and 4 deletions
|
|
@ -191,6 +191,23 @@ class OpenAICompatibleLLM(LLMInterface):
|
||||||
|
|
||||||
return None
|
return None
|
||||||
|
|
||||||
|
def _max_tokens_param_name(self) -> str:
|
||||||
|
"""Return the correct parameter name for limiting response tokens.
|
||||||
|
|
||||||
|
Native OpenAI and Groq accept 'max_completion_tokens'. Mistral and other
|
||||||
|
OpenAI-compatible endpoints that haven't adopted the newer parameter name
|
||||||
|
require 'max_tokens'. Using a custom base_url with the openai provider
|
||||||
|
signals a third-party compatible API, so fall back to 'max_tokens'.
|
||||||
|
"""
|
||||||
|
# Native OpenAI (no custom base URL) and Groq use max_completion_tokens
|
||||||
|
if self.provider == "groq":
|
||||||
|
return "max_completion_tokens"
|
||||||
|
if self.provider == "openai" and not self.base_url:
|
||||||
|
return "max_completion_tokens"
|
||||||
|
# openai with custom base_url, ollama, lmstudio, minimax, volcano —
|
||||||
|
# use the widely-supported max_tokens
|
||||||
|
return "max_tokens"
|
||||||
|
|
||||||
async def call(
|
async def call(
|
||||||
self,
|
self,
|
||||||
messages: list[dict[str, str]],
|
messages: list[dict[str, str]],
|
||||||
|
|
@ -263,9 +280,7 @@ class OpenAICompatibleLLM(LLMInterface):
|
||||||
# For reasoning models, enforce minimum to ensure space for reasoning + output
|
# For reasoning models, enforce minimum to ensure space for reasoning + output
|
||||||
if is_reasoning_model and max_completion_tokens < 16000:
|
if is_reasoning_model and max_completion_tokens < 16000:
|
||||||
max_completion_tokens = 16000
|
max_completion_tokens = 16000
|
||||||
call_params["max_completion_tokens"] = max_completion_tokens
|
call_params[self._max_tokens_param_name()] = max_completion_tokens
|
||||||
|
|
||||||
# Temperature - reasoning models don't support custom temperature
|
|
||||||
if temperature is not None and not is_reasoning_model:
|
if temperature is not None and not is_reasoning_model:
|
||||||
# MiniMax requires temperature in (0.0, 1.0] — clamp accordingly
|
# MiniMax requires temperature in (0.0, 1.0] — clamp accordingly
|
||||||
if self.provider == "minimax":
|
if self.provider == "minimax":
|
||||||
|
|
@ -577,7 +592,7 @@ class OpenAICompatibleLLM(LLMInterface):
|
||||||
}
|
}
|
||||||
|
|
||||||
if max_completion_tokens is not None:
|
if max_completion_tokens is not None:
|
||||||
call_params["max_completion_tokens"] = max_completion_tokens
|
call_params[self._max_tokens_param_name()] = max_completion_tokens
|
||||||
if temperature is not None:
|
if temperature is not None:
|
||||||
# MiniMax requires temperature in (0.0, 1.0] — clamp accordingly
|
# MiniMax requires temperature in (0.0, 1.0] — clamp accordingly
|
||||||
if self.provider == "minimax":
|
if self.provider == "minimax":
|
||||||
|
|
|
||||||
Loading…
Reference in a new issue