mirror of
https://github.com/DayuanJiang/next-ai-draw-io.git
synced 2026-10-09 03:07:46 +08:00
fix(server): count quota by the key actually used, and more review fixes
Found by the PR review, each with a test that failed first: - Quota: any key header skipped it, even one the provider never reads (x-aws-access-key-id with OpenAI), so a request ran on the server's key without being counted. The check now runs after the model is resolved and uses usesServerCredentials. On main already. - usesServerCredentials read the raw base URL; "/" cleans up to none, so an Ollama request ran on the server's key past the server-model check. - SGLang's default 127.0.0.1:8000 only fills the settings form. Chat and the model list used it as a real address, so the server called its own machine even with private URLs blocked. Now a base URL is required. - With a user's OpenAI key and no base URL, the SDK read the server's OPENAI_BASE_URL. The official endpoint is now passed. On main already. - The Test button refused nothing on the server's keys (Ollama Cloud), and a 15 s timeout reported "connected, no tool call". - The model list for Ollama without a base URL came from ollama.com while chat went to the server's Ollama. - Bedrock's "Too many tokens, please wait" counted as context too long. - On the server's keys the provider's error text stays in the server log; it can name the server's AWS account, role or internal hosts. - Desktop app: the preset keys are the user's own (NEXT_AI_DRAWIO_DESKTOP), so Max Output Tokens can be raised and keyless models in settings work again. A launch that found the remembered port taken no longer replaces it, which hid the user's chats and settings for good.
This commit is contained in:
+11
-3
@@ -36,7 +36,8 @@ export interface LLMError {
|
||||
// error can come as 403 or 429, a context or image error as a plain 400
|
||||
const SPECIFIC_TEXTS: Array<[RegExp, LLMErrorCode]> = [
|
||||
[
|
||||
/context length|context window|maximum context|prompt is too long|input is too long|too many (?:input )?tokens/i,
|
||||
// Not "too many tokens": that is Bedrock's throttling message
|
||||
/context length|context window|maximum context|prompt is too long|input is too long|too many input tokens/i,
|
||||
"context_too_long",
|
||||
],
|
||||
[
|
||||
@@ -110,12 +111,19 @@ function problemDetail(body: string): string | undefined {
|
||||
/**
|
||||
* The error text for the chat stream: what went wrong with the provider as
|
||||
* JSON for the hint, or the text the model must read to fix a tool call.
|
||||
* On the server's keys the provider's own text stays in the server log:
|
||||
* it can name the server's account, role or internal hosts.
|
||||
*/
|
||||
export function streamErrorText(error: unknown): string {
|
||||
export function streamErrorText(error: unknown, hideDetails = false): string {
|
||||
// The SDK passes an invalid tool call's error as a plain string
|
||||
if (typeof error === "string") return error
|
||||
if (isToolCallError(error)) return (error as Error).message
|
||||
return JSON.stringify(classifyLLMError(error))
|
||||
const classified = classifyLLMError(error)
|
||||
if (hideDetails) {
|
||||
console.error("[chat] Provider error:", error)
|
||||
classified.message = "The provider returned an error."
|
||||
}
|
||||
return JSON.stringify(classified)
|
||||
}
|
||||
|
||||
/**
|
||||
|
||||
Reference in New Issue
Block a user