fix(server): use the keys the user sent, and more review fixes

Found by the second PR review:
- With AWS_BEARER_TOKEN_BEDROCK set on the server, a request with the
  user's AWS keys ran on the server's token: the Bedrock SDK prefers it.
  Checked with Bedrock: invalid user keys used to get an answer.
- An OpenAI key with the official URL filled in (the settings form does
  that) went to the Responses API. Back to main's rule: a configured base
  URL uses Chat Completions.
- A user's Ollama key went to the server's OLLAMA_BASE_URL, for chat and
  for the model list. Like every other provider, it goes to the user's
  base URL or Ollama Cloud.
- The server's keyless Ollama and EdgeOne were not counted in the quota.
- AI_MODEL models on the server's keys ran on any provider with a server
  key, not only on AI_PROVIDER.
- A user's Azure key without a base URL used the server's resource name.
- The admin panel's Test button failed whenever access codes were set.
- DeepSeek's errors in the stream (plain text) were shown as they were,
  without a hint and also on the server's keys. Bedrock's throttling in
  the stream was not recognised as a rate limit.
- The EdgeOne function accepted text/plain; x=application/json, which
  other sites can send without a CORS preflight.
- Desktop app: a launch that found the old port taken for a moment (the
  previous version still quitting after an update) remembered the new
  port for good. The new port is kept only when Windows reserves the old
  one. A failed read of the presets file moved it aside as corrupt, and a
  save could then replace the presets. Switching presets on the same port
  now reloads the page. The dev launcher no longer misses a preset change
  made before or during a restart.
This commit is contained in:
dayuan.jiang
2026-10-05 10:52:37 +09:00
parent 080f44716f
commit d5f31cb253
19 changed files with 499 additions and 91 deletions
+5 -1
View File
@@ -50,7 +50,11 @@ export async function POST(req: Request) {
return validateModel(
new Request(new URL("/api/validate-model", req.url), {
method: "POST",
headers: { "Content-Type": "application/json" },
headers: {
"Content-Type": "application/json",
// Checked again there, in place of an access code
"x-admin-password": req.headers.get("x-admin-password") || "",
},
body: JSON.stringify({
provider: resolved.provider,
apiKey: resolved.apiKey,
+16 -6
View File
@@ -14,6 +14,7 @@ import { checkAccessCode } from "@/lib/access-code"
import {
CACHE_POINT,
getAIModel,
getServerProvider,
SINGLE_SYSTEM_PROVIDERS,
supportsPromptCaching,
usesServerCredentials,
@@ -49,6 +50,7 @@ import {
} from "@/lib/server-model-config"
import { allowPrivateUrls, isPrivateUrl } from "@/lib/ssrf-protection"
import { getSystemPrompt } from "@/lib/system-prompts"
import { normalizeBaseUrl } from "@/lib/types/model-config"
import { getUserIdFromRequest } from "@/lib/user-id"
import { hasCells } from "@/packages/mcp-server/src/pages.ts"
import {
@@ -245,15 +247,17 @@ async function handleChatRequest(req: Request): Promise<Response> {
} = getAIModel(clientOverrides)
// On the server's own keys, only run models the server offers: a server
// model picked by id (its model name is fixed above) or one in AI_MODEL.
// With their own key, users can run any model.
// model picked by id (its model name is fixed above) or one in AI_MODEL
// on AI_PROVIDER. With their own key, users can run any model.
const onServerCredentials = usesServerCredentials(
resolvedProvider,
clientOverrides,
)
const envModels =
process.env.AI_MODEL?.split(",").map((m) => m.trim()) || []
if (onServerCredentials && !serverModel && !envModels.includes(modelId)) {
const offeredInEnv =
envModels.includes(modelId) && resolvedProvider === getServerProvider()
if (onServerCredentials && !serverModel && !offeredInEnv) {
return Response.json(
{
error: `Model "${modelId}" is not available on this server. Add your own API key in Settings to use it.`,
@@ -264,10 +268,16 @@ async function handleChatRequest(req: Request): Promise<Response> {
// === SERVER-SIDE QUOTA CHECK START ===
// Quota is opt-in (DYNAMODB_QUOTA_TABLE) and counts what runs on the
// server's keys. Decided by the key actually used: a key header the
// provider never reads must not skip it.
// server's keys, or on its keyless Ollama or EdgeOne. Decided by the key
// actually used: a key header the provider never reads must not skip it.
const onServerEndpoint =
(resolvedProvider === "ollama" || resolvedProvider === "edgeone") &&
!clientOverrides.apiKey &&
!normalizeBaseUrl(req.headers.get("x-ai-base-url") ?? "")
const countsQuota =
isQuotaEnabled() && onServerCredentials && userId !== "anonymous"
isQuotaEnabled() &&
(onServerCredentials || onServerEndpoint) &&
userId !== "anonymous"
if (countsQuota) {
const quotaCheck = await checkAndIncrementRequest(userId, {
requests: Number(process.env.DAILY_REQUEST_LIMIT) || 10,
+4 -2
View File
@@ -2,6 +2,7 @@ import { streamText, tool } from "ai"
import { NextResponse } from "next/server"
import { z } from "zod"
import { checkAccessCode } from "@/lib/access-code"
import { checkAdminAuth } from "@/lib/admin/auth"
import { getAIModel, usesServerCredentials } from "@/lib/ai-providers"
import { classifyLLMError } from "@/lib/llm-errors"
import { allowPrivateUrls, isPrivateUrl } from "@/lib/ssrf-protection"
@@ -34,9 +35,10 @@ const NO_TOOL_CALL_WARNING =
"Connected, but the model answered without calling a tool. It may not support tool calls, which drawing needs."
export async function POST(req: Request) {
// Lets the server send requests to arbitrary URLs, so require the access code
// Lets the server send requests to arbitrary URLs, so require the access
// code, or the admin password (the admin panel's Test button)
const accessError = checkAccessCode(req)
if (accessError) return accessError
if (accessError && checkAdminAuth(req)) return accessError
try {
const body: ValidateRequest = await req.json()