fix(providers): update the v6 SDK packages and fix Claude and Gemini settings

- Update ai to 6.0.300 and the @ai-sdk providers to their latest v6-line
  versions. @ai-sdk/anthropic 3.0.47 did not know claude-opus-4-7/4-8
  and capped their output at 32000 tokens; 3.0.127 allows 128000
- Drop the fine-grained-tool-streaming beta header for the Anthropic API:
  the provider now streams tool input per tool (eager_input_streaming)
- Claude 4.7 and later reject a non-default temperature/top_p/top_k and
  the extended thinking budget with a 400. A middleware retries once
  without them, so TEMPERATURE and *_THINKING_BUDGET_TOKENS no longer
  break those models
- Prompt caching also reaches Claude on the Anthropic API and OpenRouter;
  before, only Bedrock got a cache marker
- GOOGLE_TOP_K and GOOGLE_TOP_P never reached Gemini: they were sent as
  Google provider options, which drops them. They are call settings now.
  GOOGLE_CANDIDATE_COUNT and GOOGLE_REASONING_EFFORT, which the provider
  does not support, are removed
- Add @ai-sdk/openai-compatible as a direct dependency
This commit is contained in:
dayuan.jiang
2026-10-04 13:07:56 +09:00
parent d32eb25523
commit ac62a58c9f
10 changed files with 536 additions and 304 deletions
+10 -17
View File
@@ -13,6 +13,7 @@ import path from "path"
import { z } from "zod"
import { checkAccessCode } from "@/lib/access-code"
import {
CACHE_POINT,
getAIModel,
SINGLE_SYSTEM_PROVIDERS,
supportsPromptCaching,
@@ -25,6 +26,7 @@ import {
replaceHistoricalToolInputs,
validateFileParts,
} from "@/lib/chat-helpers"
import { withDeprecatedParamsFallback } from "@/lib/deprecated-params"
import {
checkAndIncrementRequest,
isQuotaEnabled,
@@ -266,7 +268,6 @@ async function handleChatRequest(req: Request): Promise<Response> {
const {
model: baseModel,
providerOptions,
headers,
modelId,
provider: resolvedProvider,
} = getAIModel(clientOverrides)
@@ -289,8 +290,11 @@ async function handleChatRequest(req: Request): Promise<Response> {
)
}
// Retry with a smaller budget if the provider rejects the requested one
const model = withOutputTokenLimitFallback(baseModel)
// Retry once if the provider rejects the requested budget, or (newer
// Claude models) the sampling or thinking settings
const model = withOutputTokenLimitFallback(
withDeprecatedParamsFallback(baseModel),
)
// The user setting can raise the budget only on their own key (desktop users
// can still raise it themselves); on the server's keys it can only lower it
@@ -444,9 +448,7 @@ ${userInputText}
if (enhancedMessages[i].role === "assistant") {
enhancedMessages[i] = {
...enhancedMessages[i],
providerOptions: {
bedrock: { cachePoint: { type: "default" } },
},
providerOptions: CACHE_POINT,
}
break // Only cache the last assistant message
}
@@ -499,21 +501,13 @@ IMPORTANT: The "Current diagram XML" is the SINGLE SOURCE OF TRUTH for what's on
{
role: "system" as const,
content: finalSystemMessage,
...(shouldCache && {
providerOptions: {
bedrock: { cachePoint: { type: "default" } },
},
}),
...(shouldCache && { providerOptions: CACHE_POINT }),
},
// Cache breakpoint 2: Previous and Current diagram XML context
{
role: "system" as const,
content: xmlContext,
...(shouldCache && {
providerOptions: {
bedrock: { cachePoint: { type: "default" } },
},
}),
...(shouldCache && { providerOptions: CACHE_POINT }),
},
]
@@ -566,7 +560,6 @@ IMPORTANT: The "Current diagram XML" is the SINGLE SOURCE OF TRUTH for what's on
},
messages: allMessages,
...(providerOptions && { providerOptions }), // This now includes all reasoning configs
...(headers && { headers }),
// Langfuse telemetry config (returns undefined if not configured)
...(getTelemetryConfig({ sessionId: validSessionId, userId }) && {
experimental_telemetry: getTelemetryConfig({