diff --git a/plugins/openai/body_test.go b/plugins/openai/body_test.go new file mode 100644 index 00000000..85ebef67 --- /dev/null +++ b/plugins/openai/body_test.go @@ -0,0 +1,30 @@ +package openai + +import "testing" + +func TestBuildChatGPTBodyTokenLimitField(t *testing.T) { + limit := 300 + messages := []map[string]interface{}{{"role": "user", "content": "hi"}} + + body := BuildChatGPTBody("gpt-5.6-terra", messages, &limit, nil, false, OpenAIType) + if _, ok := body["max_tokens"]; ok { + t.Errorf("gpt-5 body must not send max_tokens: %v", body) + } + if body["max_completion_tokens"] != 300 { + t.Errorf("gpt-5 body must send max_completion_tokens=300: %v", body) + } + + body = BuildChatGPTBody("gpt-4.1-mini", messages, &limit, nil, false, OpenAIType) + if body["max_tokens"] != 300 { + t.Errorf("gpt-4.1-mini body must keep max_tokens=300: %v", body) + } + if _, ok := body["max_completion_tokens"]; ok { + t.Errorf("gpt-4.1-mini body must not send max_completion_tokens: %v", body) + } +} + +func TestDefaultModelIsNotRetiring(t *testing.T) { + if defaultModel != "gpt-4.1-mini" { + t.Errorf("defaultModel = %q, want gpt-4.1-mini (gpt-3.5-turbo shuts down 23 Oct 2026)", defaultModel) + } +} diff --git a/plugins/openai/tokens_test.go b/plugins/openai/tokens_test.go index b6fd3040..361d290d 100644 --- a/plugins/openai/tokens_test.go +++ b/plugins/openai/tokens_test.go @@ -210,6 +210,13 @@ func TestTrimContextAsPerLimits(t *testing.T) { json.Unmarshal(messagesAsBytes, &messagesAsMap) openAIInstance := Instance() + // These cases are sized for a 4,096-token context window. The default + // model is no longer gpt-3.5-turbo, so name it explicitly. + previous := openAIInstance.GetConfig() + model := "gpt-3.5-turbo" + openAIInstance.SetConfig(OpenAIConfig{Model: &model}) + defer openAIInstance.SetConfig(previous) + Convey("Within Limit messages 1 trimmed", t, func() { messagesToPass := messagesAsMap updatedMessages, maxTokensToUse, err := openAIInstance.TrimContextAsPerLimits(messagesToPass, 2048, 3700, false) diff --git a/plugins/openai/util.go b/plugins/openai/util.go index 78ed24e9..edd35f07 100644 --- a/plugins/openai/util.go +++ b/plugins/openai/util.go @@ -27,7 +27,7 @@ var ( const ( OpenAIAPIURL string = "https://api.openai.com/v1" - defaultModel string = "gpt-3.5-turbo" + defaultModel string = "gpt-4.1-mini" defaultPrompt string = "You're a helpful assistant." defaultMaxTokens int = 300 defaultMinTokens int = 100 @@ -830,7 +830,7 @@ func (o OpenAIConfig) Key() string { // call. // // This will prioritize the model that user has configured, if -// any and fallback to `gpt-3.5-turbo` +// any and fallback to `gpt-4.1-mini` func (o OpenAIConfig) GetModel() string { modelSet := o.Model if modelSet == nil { @@ -1262,9 +1262,14 @@ func BuildChatGPTBody(model string, messages []map[string]interface{}, maxTokens requestBodyAsMap["model"] = model } - // If maxTokens is passed, inject it in the request body + // If maxTokens is passed, inject it in the request body. GPT-5 models + // reject `max_tokens` and take the limit as `max_completion_tokens`. if maxTokens != nil { - requestBodyAsMap["max_tokens"] = *maxTokens + if strings.HasPrefix(model, "gpt-5") { + requestBodyAsMap["max_completion_tokens"] = *maxTokens + } else { + requestBodyAsMap["max_tokens"] = *maxTokens + } } // If temperature is passed, inject it in the request body diff --git a/plugins/pipelines/ai_answer.go b/plugins/pipelines/ai_answer.go index dbe6b7d2..7a97a0a3 100644 --- a/plugins/pipelines/ai_answer.go +++ b/plugins/pipelines/ai_answer.go @@ -106,7 +106,7 @@ func executeAIAnswerStage( // Set default model if none is passed if inputs.Model == nil { - defaultModel := "gpt-3.5-turbo" + defaultModel := "gpt-4.1-mini" inputs.Model = &defaultModel }