Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
5 changes: 5 additions & 0 deletions .changeset/add-claude-sonnet-5-5-support.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,5 @@
---
"zoo-code": minor
---

Add Claude Sonnet 5.5 (`claude-sonnet-5-5`) support across the Anthropic, Amazon Bedrock, Vertex AI, OpenRouter, Requesty, and Vercel AI Gateway providers. Sonnet 5.5 has a native 1M-token context window, 128K max output tokens, $2/$10 per-million-token input/output pricing, and keeps the Sonnet 5 adaptive-thinking contract (no sampling parameters, effort-based reasoning).
18 changes: 18 additions & 0 deletions packages/types/src/providers/anthropic.ts
Original file line number Diff line number Diff line change
Expand Up @@ -48,6 +48,24 @@ export const anthropicModels = {
description:
"Claude Sonnet 5 is the best combination of speed and intelligence, optimized for coding, tool use, and agentic workflows.",
},
"claude-sonnet-5-5": {
maxTokens: 128_000, // Overridden to 8k if `enableReasoningEffort` is false.
contextWindow: 1_000_000, // 1M context window native (no beta header required)
supportsImages: true,
supportsPromptCache: true,
inputPrice: 2.0, // $2 per million input tokens
outputPrice: 10.0, // $10 per million output tokens
cacheWritesPrice: 2.5, // $2.50 per million tokens (5m cache write)
cacheReadsPrice: 0.2, // $0.20 per million tokens
// Sonnet 5.5 keeps the Sonnet 5 adaptive-thinking / binary-toggle
// convention on the direct Anthropic provider path: manual budget_tokens
// and non-default sampling parameters return a 400.
supportsReasoningBudget: true,
supportsReasoningBinary: true,
supportsTemperature: false,
description:
"Claude Sonnet 5.5 is the best combination of speed and intelligence, optimized for coding, tool use, and agentic workflows.",
},
"claude-sonnet-4-5": {
maxTokens: 64_000, // Overridden to 8k if `enableReasoningEffort` is false.
contextWindow: 200_000, // Default 200K, extendable to 1M with beta flag 'context-1m-2025-08-07'
Expand Down
29 changes: 28 additions & 1 deletion packages/types/src/providers/bedrock.ts
Original file line number Diff line number Diff line change
Expand Up @@ -70,6 +70,27 @@ export const bedrockModels = {
description:
"Claude Sonnet 5 is the best combination of speed and intelligence, optimized for coding, tool use, and agentic workflows.",
},
"anthropic.claude-sonnet-5-5": {
// Undated model ID exactly as documented for Bedrock; no date-suffix
// variant has been published yet.
maxTokens: 128_000,
supportsMaxTokens: true,
contextWindow: 1_000_000, // 1M context window native (no beta header required)
supportsImages: true,
supportsPromptCache: true,
supportsReasoningBudget: true,
supportsReasoningBinary: true,
supportsTemperature: false,
inputPrice: 2.0, // $2 per million input tokens
outputPrice: 10.0, // $10 per million output tokens
cacheWritesPrice: 2.5, // $2.50 per million tokens (5m cache write)
cacheReadsPrice: 0.2, // $0.20 per million tokens
minTokensPerCachePoint: 1024,
maxCachePoints: 4,
cachableFields: ["system", "messages", "tools"],
description:
"Claude Sonnet 5.5 is the best combination of speed and intelligence, optimized for coding, tool use, and agentic workflows.",
},
"amazon.nova-pro-v1:0": {
maxTokens: 5000,
contextWindow: 300_000,
Expand Down Expand Up @@ -680,14 +701,19 @@ export const BEDROCK_1M_CONTEXT_MODEL_IDS = [
// intentionally absent.
// https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-anthropic-claude-sonnet-5.html
// https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-anthropic-claude-opus-5.html
export const BEDROCK_THINKING_DISABLE_MODEL_IDS = ["anthropic.claude-sonnet-5", "anthropic.claude-opus-5"] as const
export const BEDROCK_THINKING_DISABLE_MODEL_IDS = [
"anthropic.claude-sonnet-5",
"anthropic.claude-sonnet-5-5",
"anthropic.claude-opus-5",
] as const

// Amazon Bedrock models that support Global Inference profiles
// As of Nov 2025, AWS supports Global Inference for:
// - Claude Sonnet 4
// - Claude Sonnet 4.5
// - Claude Sonnet 4.6
// - Claude Sonnet 5
// - Claude Sonnet 5.5
// - Claude Haiku 4.5
// - Claude Opus 4.5
// - Claude Opus 4.6
Expand All @@ -700,6 +726,7 @@ export const BEDROCK_GLOBAL_INFERENCE_MODEL_IDS = [
"anthropic.claude-sonnet-4-5-20250929-v1:0",
"anthropic.claude-sonnet-4-6",
"anthropic.claude-sonnet-5",
"anthropic.claude-sonnet-5-5",
"anthropic.claude-haiku-4-5-20251001-v1:0",
"anthropic.claude-opus-4-5-20251101-v1:0",
"anthropic.claude-opus-4-6-v1",
Expand Down
2 changes: 2 additions & 0 deletions packages/types/src/providers/openrouter.ts
Original file line number Diff line number Diff line change
Expand Up @@ -40,6 +40,7 @@ export const OPEN_ROUTER_PROMPT_CACHING_MODELS = new Set([
"anthropic/claude-sonnet-4.5",
"anthropic/claude-sonnet-4.6",
"anthropic/claude-sonnet-5",
"anthropic/claude-sonnet-5-5",
"anthropic/claude-opus-4",
"anthropic/claude-opus-4.1",
"anthropic/claude-opus-4.5",
Expand Down Expand Up @@ -87,6 +88,7 @@ export const OPEN_ROUTER_REASONING_BUDGET_MODELS = new Set([
"anthropic/claude-sonnet-4.5",
"anthropic/claude-sonnet-4.6",
"anthropic/claude-sonnet-5",
"anthropic/claude-sonnet-5-5",
"anthropic/claude-haiku-4.5",
"google/gemini-2.5-pro-preview",
"google/gemini-2.5-pro",
Expand Down
2 changes: 2 additions & 0 deletions packages/types/src/providers/vercel-ai-gateway.ts
Original file line number Diff line number Diff line change
Expand Up @@ -20,6 +20,7 @@ export const VERCEL_AI_GATEWAY_PROMPT_CACHING_MODELS = new Set([
"anthropic/claude-sonnet-4",
"anthropic/claude-sonnet-4.6",
"anthropic/claude-sonnet-5",
"anthropic/claude-sonnet-5-5",
"openai/gpt-4.1",
"openai/gpt-4.1-mini",
"openai/gpt-4.1-nano",
Expand Down Expand Up @@ -69,6 +70,7 @@ export const VERCEL_AI_GATEWAY_VISION_AND_TOOLS_MODELS = new Set([
"anthropic/claude-sonnet-4.5",
"anthropic/claude-sonnet-4.6",
"anthropic/claude-sonnet-5",
"anthropic/claude-sonnet-5-5",
"google/gemini-1.5-flash",
"google/gemini-1.5-pro",
"google/gemini-2.0-flash",
Expand Down
15 changes: 15 additions & 0 deletions packages/types/src/providers/vertex.ts
Original file line number Diff line number Diff line change
Expand Up @@ -396,6 +396,21 @@ export const vertexModels = {
description:
"Claude Sonnet 5 is the best combination of speed and intelligence, optimized for coding, tool use, and agentic workflows.",
},
"claude-sonnet-5-5": {
maxTokens: 128_000, // 128K max output tokens per the Anthropic model card (same across all platforms).
contextWindow: 1_000_000, // 1M context window native (no beta header required)
supportsImages: true,
supportsPromptCache: true,
inputPrice: 2.0, // $2 per million input tokens
outputPrice: 10.0, // $10 per million output tokens
cacheWritesPrice: 2.5, // $2.50 per million tokens (5m cache write)
cacheReadsPrice: 0.2, // $0.20 per million tokens
supportsReasoningBudget: true,
supportsReasoningBinary: true,
supportsTemperature: false,
description:
"Claude Sonnet 5.5 is the best combination of speed and intelligence, optimized for coding, tool use, and agentic workflows.",
},
"claude-haiku-4-5@20251001": {
maxTokens: 8192,
contextWindow: 200_000,
Expand Down
54 changes: 54 additions & 0 deletions src/api/providers/__tests__/anthropic-vertex.spec.ts
Original file line number Diff line number Diff line change
Expand Up @@ -1027,6 +1027,25 @@ describe("VertexHandler", () => {
expect(model.info.supportsTemperature).toBe(false)
})

it("should return Claude Sonnet 5.5 model info", () => {
const handler = new AnthropicVertexHandler({
apiModelId: "claude-sonnet-5-5",
vertexProjectId: "test-project",
vertexRegion: "us-central1",
})

const model = handler.getModel()
expect(model.id).toBe("claude-sonnet-5-5")
expect(model.info.maxTokens).toBe(128_000)
expect(model.info.contextWindow).toBe(1_000_000)
expect(model.info.inputPrice).toBe(2.0)
expect(model.info.outputPrice).toBe(10.0)
expect(model.info.supportsReasoningBinary).toBe(true)
expect(model.info.supportsReasoningBudget).toBe(true)
expect(model.info.supportsPromptCache).toBe(true)
expect(model.info.supportsTemperature).toBe(false)
})

it("should return Claude Opus 5 model info", () => {
const handler = new AnthropicVertexHandler({
apiModelId: "claude-opus-5",
Expand Down Expand Up @@ -1390,6 +1409,41 @@ describe("VertexHandler", () => {
expect(request.temperature).toBeUndefined()
})

it("should use adaptive thinking for Claude Sonnet 5.5", async () => {
const sonnetHandler = new AnthropicVertexHandler({
apiModelId: "claude-sonnet-5-5",
vertexProjectId: "test-project",
vertexRegion: "us-central1",
enableReasoningEffort: true,
})

const mockCreate = vitest
.fn()
.mockImplementation(async () =>
asyncStreamFrom([
{ type: "message_start", message: { usage: { input_tokens: 10, output_tokens: 5 } } },
]),
)
// The SDK client's overloaded `create` signature can't be assigned a
// vitest mock directly, so a structural double assertion is required.
;(sonnetHandler["client"].messages as unknown as { create: typeof mockCreate }).create = mockCreate

await sonnetHandler
.createMessage("You are a helpful assistant", [{ role: "user", content: "Hello" }])
.next()

expect(mockCreate).toHaveBeenCalledWith(
expect.objectContaining({
thinking: { type: "adaptive" },
}),
undefined,
)

const request = mockCreate.mock.calls[0][0]
expect(request.thinking).not.toHaveProperty("budget_tokens")
expect(request.temperature).toBeUndefined()
})

it("should use adaptive thinking for Claude Opus 5", async () => {
const opusHandler = new AnthropicVertexHandler({
apiModelId: "claude-opus-5",
Expand Down
46 changes: 46 additions & 0 deletions src/api/providers/__tests__/anthropic.spec.ts
Original file line number Diff line number Diff line change
Expand Up @@ -456,6 +456,31 @@ describe("AnthropicHandler", () => {
expect(requestOptions?.headers?.["anthropic-beta"]).toContain("prompt-caching-2024-07-31")
})

it("should use adaptive thinking for Claude Sonnet 5.5 when reasoning is enabled", async () => {
const sonnetHandler = new AnthropicHandler({
apiKey: "test-api-key",
apiModelId: "claude-sonnet-5-5",
enableReasoningEffort: true,
modelMaxTokens: 32768,
})

const stream = sonnetHandler.createMessage(systemPrompt, [
{
role: "user",
content: [{ type: "text" as const, text: "Hello" }],
},
])

await collectStream(stream)

const requestBody = mockCreate.mock.calls[mockCreate.mock.calls.length - 1]?.[0]
const requestOptions = mockCreate.mock.calls[mockCreate.mock.calls.length - 1]?.[1]
expect(requestBody?.thinking).toEqual({ type: "adaptive" })
expect(requestBody?.temperature).toBeUndefined()
expect(requestBody?.max_tokens).toBe(32768)
expect(requestOptions?.headers?.["anthropic-beta"]).toContain("prompt-caching-2024-07-31")
})

it("should use adaptive thinking for Claude Opus 5 when reasoning is enabled", async () => {
const opusHandler = new AnthropicHandler({
apiKey: "test-api-key",
Expand Down Expand Up @@ -725,6 +750,27 @@ describe("AnthropicHandler", () => {
expect(model.reasoningBudget).toBeUndefined()
})

it("should handle Claude Sonnet 5.5 model correctly", () => {
const handler = new AnthropicHandler({
apiKey: "test-api-key",
apiModelId: "claude-sonnet-5-5",
})
const model = handler.getModel()
expect(model.id).toBe("claude-sonnet-5-5")
expect(model.info.maxTokens).toBe(128000)
expect(model.info.contextWindow).toBe(1000000)
expect(model.info.inputPrice).toBe(2.0)
expect(model.info.outputPrice).toBe(10.0)
expect(model.info.cacheWritesPrice).toBe(2.5)
expect(model.info.cacheReadsPrice).toBe(0.2)
expect(model.maxTokens).toBe(8192)
expect(model.info.supportsReasoningBinary).toBe(true)
expect(model.info.supportsReasoningBudget).toBe(true)
expect(model.info.supportsPromptCache).toBe(true)
expect(model.info.supportsTemperature).toBe(false)
expect(model.reasoningBudget).toBeUndefined()
})

it("should handle Claude Opus 5 model correctly", () => {
const handler = new AnthropicHandler({
apiKey: "test-api-key",
Expand Down
Loading
Loading