feat: rewrite llm config and model patch

This commit is contained in:
Simon
2026-07-02 21:31:44 +08:00
parent 83fee37fb4
commit 8ad719744a
10 changed files with 144 additions and 268 deletions
+69 -69
View File
@@ -24,8 +24,17 @@ export function zodToOpenAITool(name: string, tool: Tool) {
}
/**
* Patch model specific parameters
* @note in-place modification
* Patch model specific parameters. Only patches known models.
*
* @purpose
* - Reconcile the differences in the parameter schema each model accepts.
* - Disable thinking/reasoning, or lower it to the minimum where a full disable is impossible.
* - Minimize returned tokens.
* - Raise temperature for known smaller models to improve auto-recovery odds.
* @note Honor temperature if explicitly set by the user
*
* @todo Need vendor-specific patches.
* Local and 3rd-party hosted models may have different schema.
*/
export function modelPatch(body: Record<string, any>) {
const model: string = body.model || ''
@@ -34,13 +43,30 @@ export function modelPatch(body: Record<string, any>) {
const modelName = normalizeModelName(model)
if (modelName.startsWith('qwen')) {
debug('Applying Qwen patch: use higher temperature for auto fixing')
body.temperature = Math.max(body.temperature || 0, 1.0)
debug('Patch Qwen: disable thinking')
body.enable_thinking = false
if (body.temperature === undefined && !/max|plus/.test(modelName)) {
debug('Patch Qwen: raise temperature to 1.0')
body.temperature = 1.0
}
}
if (modelName.startsWith('claude')) {
debug('Applying Claude patch: disable thinking')
if (modelName.startsWith('deepseek')) {
debug('Patch DeepSeek: disable thinking, remove tool_choice')
body.thinking = { type: 'disabled' }
delete body.tool_choice
}
if (modelName.startsWith('gpt-5')) {
// verbosity trims output tokens across the whole GPT-5 family.
body.verbosity = 'low'
// The GPT-5.0 generation only supports "minimal"; 5.1+ supports "none".
body.reasoning_effort = /^gpt-5(-|$)/.test(modelName) ? 'minimal' : 'none'
debug(`Patch GPT-5: verbosity=low, reasoning_effort=${body.reasoning_effort}`)
}
if (modelName.startsWith('claude') && /opus|sonnet|haiku/.test(modelName)) {
debug('Patch Claude: disable thinking')
body.thinking = { type: 'disabled' }
// Convert tool_choice to Claude format
@@ -53,76 +79,50 @@ export function modelPatch(body: Record<string, any>) {
debug('Applying Claude patch: convert tool_choice format')
body.tool_choice = { type: 'tool', name: body.tool_choice.function.name }
}
// TODO: Claude naming pattern has changed
// needs proper handling
if (
modelName.startsWith('claude-opus-4-7') ||
modelName.startsWith('claude-opus-47') ||
modelName.startsWith('claude-opus-4-8') ||
modelName.startsWith('claude-opus-48')
) {
debug('Applying Claude-4.7/4.8 patch: remove temperature')
delete body.temperature
}
}
if (modelName.startsWith('grok')) {
debug('Applying Grok patch: removing tool_choice')
delete body.tool_choice
debug('Applying Grok patch: disable reasoning and thinking')
body.thinking = { type: 'disabled', effort: 'minimal' }
body.reasoning = { enabled: false, effort: 'low' }
}
if (modelName.startsWith('gpt')) {
debug('Applying GPT patch: set verbosity to low')
body.verbosity = 'low'
// *-chat-latest models don't support reasoning_effort — skip patches that set it
if (modelName.includes('chat-latest')) {
debug('Omitting reasoning_effort and temperature for chat-latest')
delete body.reasoning_effort
delete body.temperature
} else if (modelName.startsWith('gpt-52')) {
debug('Applying GPT-52 patch: disable reasoning')
body.reasoning_effort = 'none'
} else if (modelName.startsWith('gpt-51')) {
debug('Applying GPT-51 patch: disable reasoning')
body.reasoning_effort = 'none'
} else if (modelName.startsWith('gpt-54')) {
debug('Applying GPT-5.4 patch: remove reasoning_effort')
delete body.reasoning_effort
} else if (modelName.startsWith('gpt-55')) {
debug('Applying GPT-5.4 patch: remove reasoning_effort and temperature')
delete body.reasoning_effort
delete body.temperature
} else if (modelName.startsWith('gpt-5-mini')) {
debug('Applying GPT-5-mini patch: set reasoning effort to low, temperature to 1')
body.reasoning_effort = 'low'
body.temperature = 1
} else if (modelName.startsWith('gpt-5')) {
debug('Applying GPT-5 patch: set reasoning effort to low')
body.reasoning_effort = 'low'
}
}
if (modelName.startsWith('gemini')) {
debug('Applying Gemini patch: set reasoning effort to minimal')
body.reasoning_effort = 'minimal'
debug('Patch Gemini: reasoning_effort=low')
body.reasoning_effort = 'low'
if (/^gemini-25(?!.*pro)/.test(modelName)) {
debug('Patch Gemini 2.5 non-Pro: reasoning_effort=none')
body.reasoning_effort = 'none'
} else if (
modelName.startsWith('gemini-35-flash') ||
modelName.startsWith('gemini-31-flash-lite') ||
modelName.startsWith('gemini-3-flash')
) {
debug('Patch Gemini 3.x Flash/Lite: reasoning_effort=minimal')
body.reasoning_effort = 'minimal'
}
}
if (modelName.startsWith('deepseek')) {
debug('Applying DeepSeek patch: remove tool_choice')
delete body.tool_choice
if (modelName.startsWith('glm')) {
debug('Patch GLM: disable thinking')
body.thinking = { type: 'disabled' }
}
if (modelName.startsWith('grok')) {
if (/^grok-4-?3/.test(modelName)) {
debug('Patch Grok 4.3: reasoning_effort=none')
body.reasoning_effort = 'none'
} else if (modelName.startsWith('grok-3-mini') || modelName.startsWith('grok-code-fast')) {
debug('Patch Grok mini/code: reasoning_effort=low')
body.reasoning_effort = 'low'
}
}
if (modelName.startsWith('kimi') && !modelName.includes('code')) {
// kimi-k2.7-code cannot disable thinking (errors), hence the code exclusion.
debug('Patch Kimi: disable thinking')
body.thinking = { type: 'disabled' }
}
if (modelName.startsWith('minimax')) {
debug('Applying MiniMax patch: clamp temperature to (0, 1]')
// MiniMax API rejects temperature = 0; clamp to a small positive value
body.temperature = Math.max(body.temperature || 0, 0.01)
if (body.temperature > 1) body.temperature = 1
// MiniMax does not support parallel_tool_calls
// Only M3 can disable thinking; M2.x accepts the field as a silent no-op.
// parallel_tool_calls is unsupported.
debug('Patch MiniMax: disable thinking, remove parallel_tool_calls')
body.thinking = { type: 'disabled' }
delete body.parallel_tool_calls
}
@@ -144,7 +144,7 @@ export function modelPatch(body: Record<string, any>) {
* They should be treated as the same model.
* Normalize them to `gpt-52`
*/
function normalizeModelName(modelName: string): string {
export function normalizeModelName(modelName: string): string {
let normalizedName = modelName.toLowerCase()
// remove prefix before '/'