Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
5 changes: 5 additions & 0 deletions .changeset/new-models-support.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,5 @@
---
"@reddb-io/redcode": patch
---

Claude Opus 5.5 and GPT-6 models now work with the same defaults as their predecessors. GPT-6 Sol, Luna and Astra get medium reasoning effort, reasoning summaries and encrypted reasoning on OpenAI, Azure, GitHub Copilot and OpenCode Zen, plus the Codex subscription context limits. Claude Opus 5.5 shows summarized thinking without picking a variant, and structured output on Opus 5.5, Fable and Mythos asks for the StructuredOutput tool instead of forcing it, which those models reject. GPT Luna is now the preferred small model for titles and summaries.
Original file line number Diff line number Diff line change
Expand Up @@ -1671,14 +1671,15 @@ type ResponsesModelConfig = {
}

function getResponsesModelConfig(modelId: string): ResponsesModelConfig {
// GPT-5 and every later generation (gpt-6-sol, gpt-6-luna, ...) follow the gpt-5 rules.
const gpt5OrNewer = Number(/^gpt-(\d+)/.exec(modelId)?.[1]) >= 5
const gptChat = /^gpt-\d+-chat/.test(modelId)
const supportsFlexProcessing =
modelId.startsWith("o3") ||
modelId.startsWith("o4-mini") ||
(modelId.startsWith("gpt-5") && !modelId.startsWith("gpt-5-chat"))
modelId.startsWith("o3") || modelId.startsWith("o4-mini") || (gpt5OrNewer && !gptChat)
const supportsPriorityProcessing =
modelId.startsWith("gpt-4") ||
modelId.startsWith("gpt-5-mini") ||
(modelId.startsWith("gpt-5") && !modelId.startsWith("gpt-5-nano") && !modelId.startsWith("gpt-5-chat")) ||
(gpt5OrNewer && !/^gpt-\d+-nano/.test(modelId) && !gptChat) ||
modelId.startsWith("o3") ||
modelId.startsWith("o4-mini")
const defaults = {
Expand All @@ -1688,8 +1689,8 @@ function getResponsesModelConfig(modelId: string): ResponsesModelConfig {
supportsPriorityProcessing,
}

// gpt-5-chat models are non-reasoning
if (modelId.startsWith("gpt-5-chat")) {
// gpt-5-chat and later gpt-N-chat models are non-reasoning
if (gpt5OrNewer && gptChat) {
return {
...defaults,
isReasoningModel: false,
Expand All @@ -1699,7 +1700,7 @@ function getResponsesModelConfig(modelId: string): ResponsesModelConfig {
// o series reasoning models:
if (
modelId.startsWith("o") ||
modelId.startsWith("gpt-5") ||
gpt5OrNewer ||
modelId.startsWith("codex-") ||
modelId.startsWith("computer-use")
) {
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -79,6 +79,51 @@ describe("doGenerate", () => {
})
})

describe("reasoning model classification", () => {
test.each([
["gpt-5.5", true],
["gpt-6-sol", true],
["gpt-6-luna", true],
["gpt-5-chat-latest", false],
["gpt-6-chat-latest", false],
["gpt-4.1", false],
])("%s is a reasoning model: %p", async (modelId, reasoning) => {
const mockFetch = createMockFetch({
id: "resp_1",
created_at: 0,
model: modelId,
output: [
{
type: "message",
role: "assistant",
id: "msg_1",
content: [{ type: "output_text", text: "Hello there", annotations: [] }],
},
],
usage: { input_tokens: 10, output_tokens: 5 },
})
const model = new OpenAIResponsesLanguageModel(modelId, {
provider: "copilot",
url: () => "https://api.test.com/responses",
headers: () => ({ Authorization: "Bearer test-token" }),
fetch: mockFetch as any,
})

await model.doGenerate({
prompt: [{ role: "system", content: "Be brief." }, ...TEST_PROMPT],
temperature: 0.5,
providerOptions: { copilot: { reasoningEffort: "high" } },
includeRawChunks: false,
} as any)

const calls = mockFetch.mock.calls as unknown as [string, RequestInit][]
const body = JSON.parse(String(calls[0][1].body))
expect(body.reasoning).toEqual(reasoning ? { effort: "high" } : undefined)
expect(body.temperature).toEqual(reasoning ? undefined : 0.5)
expect(body.input[0].role).toBe(reasoning ? "developer" : "system")
})
})

describe("convertToOpenAIResponsesInput", () => {
test("echoes a stale tool-call itemId from the copilot namespace as the function_call id", async () => {
const { input } = await convertToOpenAIResponsesInput({
Expand Down
3 changes: 2 additions & 1 deletion packages/llm/src/providers/openai-options.ts
Original file line number Diff line number Diff line change
Expand Up @@ -46,7 +46,8 @@ export const gpt5DefaultOptions = (
options: { readonly textVerbosity?: boolean } = {},
): ProviderOptions | undefined => {
const id = modelID.toLowerCase()
if (!id.includes("gpt-5") || id.includes("gpt-5-chat") || id.includes("gpt-5-pro")) return undefined
// GPT-5 and every later generation (gpt-6-sol, gpt-6-luna, ...) share these defaults.
if (!(Number(/gpt-(\d+)/.exec(id)?.[1]) >= 5) || /gpt-\d+-(?:chat|pro)/.test(id)) return undefined
return openAIProviderOptions({
reasoningEffort: "medium",
reasoningSummary: "auto",
Expand Down
16 changes: 16 additions & 0 deletions packages/llm/test/provider/openai-responses.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -659,6 +659,22 @@ describe("OpenAI Responses route", () => {
}),
)

it.effect("applies the GPT-5 reasoning defaults to GPT-6 models", () =>
Effect.gen(function* () {
const prepared = yield* LLMClient.prepare<OpenAIResponses.OpenAIResponsesBody>(
LLM.request({
model: OpenAI.configure({ baseURL: "https://api.openai.test/v1/", apiKey: "test" }).responses("gpt-6-sol"),
prompt: "hi",
}),
)

expect(prepared.body.store).toBe(false)
expect(prepared.body.include).toEqual(["reasoning.encrypted_content"])
expect(prepared.body.reasoning).toEqual({ effort: "medium", summary: "auto" })
expect(prepared.body.text).toBeUndefined()
}),
)

it.effect("lets callers opt out of the GPT-5 default include", () =>
Effect.gen(function* () {
const prepared = yield* LLMClient.prepare<OpenAIResponses.OpenAIResponsesBody>(
Expand Down
15 changes: 7 additions & 8 deletions packages/redcode/src/plugin/openai/codex.ts
Original file line number Diff line number Diff line change
Expand Up @@ -312,14 +312,13 @@ export async function CodexAuthPlugin(input: PluginInput, options: CodexAuthPlug
output: 0,
cache: { read: 0, write: 0 },
},
limit:
model.id.includes("gpt-5.5") || model.id.includes("gpt-5.6")
? {
context: 400_000,
input: 272_000,
output: 128_000,
}
: model.limit,
limit: ["gpt-5.5", "gpt-5.6", "gpt-6-"].some((prefix) => model.id.includes(prefix))
? {
context: 400_000,
input: 272_000,
output: 128_000,
}
: model.limit,
},
]),
)
Expand Down
4 changes: 2 additions & 2 deletions packages/redcode/src/provider/provider.ts
Original file line number Diff line number Diff line change
Expand Up @@ -2066,7 +2066,7 @@ const layer = Layer.effect(
}

const priority = providerID.startsWith("opencode")
? ["gpt-nano"]
? ["gpt-luna", "gpt-nano"]
: providerID.startsWith("github-copilot")
? ["gpt-mini", ...smallModelFamilyPriority]
: smallModelFamilyPriority
Expand Down Expand Up @@ -2142,7 +2142,7 @@ const layer = Layer.effect(
)

const priority = ["gpt-5", "claude-sonnet-4", "big-pickle", "gemini-3-pro"]
const smallModelFamilyPriority = ["gemini-flash", "gpt-nano", "claude-haiku"]
const smallModelFamilyPriority = ["gpt-luna", "gemini-flash", "gpt-nano", "claude-haiku"]
export function sort<T extends { id: string }>(models: T[]) {
return sortBy(
models,
Expand Down
36 changes: 30 additions & 6 deletions packages/redcode/src/provider/transform.ts
Original file line number Diff line number Diff line change
Expand Up @@ -740,13 +740,13 @@ function anthropicBlockBinding(model: Provider.Model, options: { [x: string]: an
switch (model.api.npm) {
case "@ai-sdk/anthropic":
case "@ai-sdk/google-vertex/anthropic": {
const thinking = options.thinking ?? { type: "adaptive" }
const thinking = options.thinking ?? adaptiveThinkingDefault(model.api.id)
if (thinking.type !== "adaptive" && thinking.type !== "enabled") return options
if (thinking.blockBinding !== undefined) return options
return { ...options, thinking: { ...thinking, blockBinding: ANTHROPIC_BLOCK_BINDING } }
}
case "@ai-sdk/amazon-bedrock": {
const reasoningConfig = options.reasoningConfig ?? { type: "adaptive" }
const reasoningConfig = options.reasoningConfig ?? adaptiveThinkingDefault(model.api.id)
if (reasoningConfig.type !== "adaptive" && reasoningConfig.type !== "enabled") return options
if (reasoningConfig.blockBinding !== undefined) return options
return { ...options, reasoningConfig: { ...reasoningConfig, blockBinding: ANTHROPIC_BLOCK_BINDING } }
Expand All @@ -755,6 +755,23 @@ function anthropicBlockBinding(model: Provider.Model, options: { [x: string]: an
return options
}

// Claude 4.7+ omits thinking text unless asked, so a request without a variant would stream none.
function adaptiveThinkingDefault(apiId: string) {
return { type: "adaptive", ...(anthropicOmitsThinking(apiId) ? { display: "summarized" } : {}) }
}

// Claude Opus 5.5 and the Fable and Mythos models always think, and the API answers a forced
// tool_choice ("any" or a named tool) with a 400 for them.
export function supportsForcedToolChoice(model: Provider.Model) {
const id = model.api.id.toLowerCase()
if (!id.includes("claude-")) return true
if (/(?:^|[^a-z])(?:fable|mythos)(?:[^a-z]|$)/.test(id)) return false
const version = /claude-(?:([a-z]+)-)?(\d+)(?:[.-](\d{1,2}))?(?:-([a-z]+))?(?:[.@-]|$)/.exec(id)
if (!version || (version[1] ?? version[4]) !== "opus") return true
const major = Number(version[2])
return major < 5 || (major === 5 && Number(version[3] ?? 0) < 5)
}

function googleThinkingLevelEfforts(apiId: string) {
const id = apiId.toLowerCase()
if (!id.includes("gemini-3")) return ["low", "high"]
Expand Down Expand Up @@ -1348,17 +1365,18 @@ export function options(input: {

// Any gpt version above 5.4 in combination with azure does not support reasoningEffort
// so we should return early here.
const [, gptMajorVersion, gptMinorVersion] = input.model.api.id.match(/gpt-(\d+)\.(\d+)/) ?? []
const isGpt55OrNewer = Number(gptMajorVersion) > 5 || (Number(gptMajorVersion) === 5 && Number(gptMinorVersion) >= 5)
const [, gptMajorVersion, gptMinorVersion] = input.model.api.id.match(/gpt-(\d+)(?:\.(\d+))?/) ?? []
const isGpt55OrNewer =
Number(gptMajorVersion) > 5 || (Number(gptMajorVersion) === 5 && Number(gptMinorVersion ?? 0) >= 5)
if (input.model.api.npm === "@ai-sdk/azure" && input.providerOptions?.useCompletionUrls) {
if (!isGpt55OrNewer) {
result["reasoningEffort"] = "medium"
}
return result
}

if (input.model.api.id.includes("gpt-5") && !input.model.api.id.includes("gpt-5-chat")) {
if (!input.model.api.id.includes("gpt-5-pro")) {
if (openaiReasoningGeneration(input.model.api.id)) {
if (!/gpt-\d+-pro/.test(input.model.api.id)) {
result["reasoningEffort"] = "medium"
if (
input.model.api.npm === "@ai-sdk/openai" ||
Expand Down Expand Up @@ -1396,6 +1414,12 @@ export function options(input: {
return result
}

// GPT-5 and every later generation (gpt-6-sol, gpt-6-luna, ...) share the reasoning defaults;
// gpt-N-chat ids are the non-reasoning chat snapshots.
function openaiReasoningGeneration(apiId: string) {
return Number(/gpt-(\d+)/.exec(apiId)?.[1]) >= 5 && !/gpt-\d+-chat/.test(apiId)
}

export function smallOptions(model: Provider.Model) {
const small = Object.values(model.variants ?? {})[0] ?? {}
if (
Expand Down
46 changes: 44 additions & 2 deletions packages/redcode/src/session/prompt.ts
Original file line number Diff line number Diff line change
Expand Up @@ -201,6 +201,12 @@ IMPORTANT:

const STRUCTURED_OUTPUT_SYSTEM_PROMPT = `IMPORTANT: The user has requested structured output. You MUST use the StructuredOutput tool to provide your final response. Do NOT respond with plain text - you MUST call the StructuredOutput tool with your answer formatted according to the schema.`

// Stored messages read back as plain JSON, but publishing a user message encodes its format as the
// schema class, so a stored format is decoded again before the message is republished.
const decodeFormat = Schema.decodeUnknownSync(SessionV1.Format)

const STRUCTURED_OUTPUT_REMINDER = `Your last response was plain text, but structured output was requested. Call the StructuredOutput tool now with your final answer formatted according to the schema.`

function mcpResourceBase64Size(value: string) {
const trimmed = value.replace(/\s/g, "")
const padding = trimmed.endsWith("==") ? 2 : trimmed.endsWith("=") ? 1 : 0
Expand Down Expand Up @@ -1432,6 +1438,7 @@ const layer = Layer.effect(
const info: SessionV1.User = {
...message.value.info,
time: { ...message.value.info.time, created: DateTime.toEpochMillis(now) },
...(message.value.info.format ? { format: decodeFormat(message.value.info.format) } : {}),
}
yield* sessions.updateMessage(info)
yield* events.publish(SessionV1.Event.MessagePromoted, { sessionID, messageID })
Expand Down Expand Up @@ -1608,6 +1615,8 @@ const layer = Layer.effect(
let todoContinuations = 0
let reconnects = 0
let responseRepairs = 0
// Reminders sent to a model that cannot be forced to call StructuredOutput and answered in text.
let structuredReminders = 0
const promptAssessments = new Map<string, Intelligence.Evaluation | undefined>()
const intelligenceAttempts = new Map<string, Intelligence.Evaluation | undefined>()
const toolAssessments = new Map<string, Intelligence.Evaluation | undefined>()
Expand Down Expand Up @@ -1643,6 +1652,7 @@ const layer = Layer.effect(
todoContinuations = 0
reconnects = 0
responseRepairs = 0
structuredReminders = 0
reviewed = undefined
ineffectiveCompactions = 0
overflowRecoveries = 0
Expand Down Expand Up @@ -2786,7 +2796,13 @@ const layer = Layer.effect(
messages: stepMessages,
tools,
model,
toolChoice: format.type === "json_schema" ? "required" : undefined,
// Models that always think reject a forced tool choice; they are asked for the tool instead.
toolChoice:
format.type === "json_schema"
? ProviderTransform.supportsForcedToolChoice(model)
? "required"
: "auto"
: undefined,
estimate: requestEstimate,
// System One already chose this turn's tools and skills; a RedRouter must not choose again.
// Its hint lets a RedRouter combo pick the model for the turn; Redcode never switches it.
Expand Down Expand Up @@ -2841,9 +2857,35 @@ const layer = Layer.effect(
return "break" as const
}
if (format.type === "json_schema") {
// Without a forced tool choice the model may answer in text; remind it before failing.
if (
!ProviderTransform.supportsForcedToolChoice(model) &&
structuredReminders < (format.retryCount ?? 2)
) {
structuredReminders++
const reminder: SessionV1.User = {
id: MessageID.ascending(),
sessionID,
role: "user",
time: { created: Date.now() },
agent: lastUser.agent,
model: lastUser.model,
format: decodeFormat(format),
}
yield* sessions.updateMessage(reminder)
yield* sessions.updatePart({
id: PartID.ascending(),
sessionID,
messageID: reminder.id,
type: "text",
text: STRUCTURED_OUTPUT_REMINDER,
synthetic: true,
})
return "continue" as const
}
handle.message.error = new SessionV1.StructuredOutputError({
message: "Model did not produce structured output",
retries: 0,
retries: structuredReminders,
}).toObject()
yield* sessions.updateMessage(handle.message)
return "break" as const
Expand Down
15 changes: 14 additions & 1 deletion packages/redcode/test/plugin/codex.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -289,7 +289,17 @@ describe("plugin.codex", () => {
const provider = {
models: {
...Object.fromEntries(
["gpt-5.4", "gpt-5.5", "gpt-5.6-sol", "gpt-5.6-terra", "gpt-5.6-luna", "gpt-5.7-pro"].map((id) => [
[
"gpt-5.4",
"gpt-5.5",
"gpt-5.6-sol",
"gpt-5.6-terra",
"gpt-5.6-luna",
"gpt-5.7-pro",
"gpt-6-sol",
"gpt-6-luna",
"gpt-6-astra",
].map((id) => [
id,
{ id, api: { id }, limit, cost: {}, options: {} },
]),
Expand Down Expand Up @@ -318,6 +328,9 @@ describe("plugin.codex", () => {
expect(models["gpt-5.6-sol"]?.limit).toEqual({ context: 400_000, input: 272_000, output: 128_000 })
expect(models["gpt-5.6-terra"]?.limit).toEqual({ context: 400_000, input: 272_000, output: 128_000 })
expect(models["gpt-5.6-luna"]?.limit).toEqual({ context: 400_000, input: 272_000, output: 128_000 })
expect(models["gpt-6-sol"]?.limit).toEqual({ context: 400_000, input: 272_000, output: 128_000 })
expect(models["gpt-6-luna"]?.limit).toEqual({ context: 400_000, input: 272_000, output: 128_000 })
expect(models["gpt-6-astra"]?.limit).toEqual({ context: 400_000, input: 272_000, output: 128_000 })
expect(models["gpt-5.4-pro"]).toBeUndefined()
expect(models["gpt-5.7-pro"]).toBeDefined()
expect(models["gpt-5.6-sol-high"]).toBeDefined()
Expand Down
Loading
Loading