Fix local LLM prompt truncation and context window limit guard

This commit is contained in:
2026-07-22 17:18:34 -04:00
parent 550ba0d29f
commit 20e9349786
2 changed files with 11 additions and 6 deletions
+1 -1
View File
@@ -1 +1 @@
97312 98070
+10 -5
View File
@@ -92,10 +92,15 @@ const generateTextWithLLMFallback = async (
const modelName = llmConfig?.lmStudioModel || "local-model"; const modelName = llmConfig?.lmStudioModel || "local-model";
const apiKey = llmConfig?.customApiKey; const apiKey = llmConfig?.customApiKey;
// 1. Direct LM Studio / Local LLM request // Safely limit prompt length for local LLMs (max ~12,000 chars ~ 3000 tokens context limit)
if (primary === 'lmstudio' || primary === 'openai_compatible') { const safePrompt = prompt.length > 12000 ? prompt.substring(0, 12000) + "\n\n[Content truncated for context size]" : prompt;
const hasValidGeminiKey = !!process.env.GEMINI_API_KEY && process.env.GEMINI_API_KEY.trim().length > 10;
// 1. Direct LM Studio / Local LLM request or Gemini key is missing
if (primary === 'lmstudio' || primary === 'openai_compatible' || !hasValidGeminiKey) {
console.log(`[LLM Router] Routing directly to Local LLM (${baseUrl})...`); console.log(`[LLM Router] Routing directly to Local LLM (${baseUrl})...`);
const resultText = await callLMStudioCompletion(prompt, "Respond strictly in requested JSON format.", baseUrl, modelName, apiKey); const resultText = await callLMStudioCompletion(safePrompt, "Respond strictly in requested JSON format.", baseUrl, modelName, apiKey);
return { text: resultText, providerUsed: `Local LLM (${modelName})` }; return { text: resultText, providerUsed: `Local LLM (${modelName})` };
} }
@@ -112,7 +117,7 @@ const generateTextWithLLMFallback = async (
const response = await ai.models.generateContent({ const response = await ai.models.generateContent({
model: "gemini-3.6-flash", model: "gemini-3.6-flash",
contents: prompt, contents: safePrompt,
config config
}); });
@@ -127,7 +132,7 @@ const generateTextWithLLMFallback = async (
console.log(`[LLM Router] Falling back to Local LM Studio endpoint at ${baseUrl}...`); console.log(`[LLM Router] Falling back to Local LM Studio endpoint at ${baseUrl}...`);
try { try {
const localResult = await callLMStudioCompletion( const localResult = await callLMStudioCompletion(
`${prompt}\n\nIMPORTANT: Respond with clean JSON matching the requested structure.`, `${safePrompt}\n\nIMPORTANT: Respond with clean JSON matching the requested structure.`,
"You are a backup AI assistant filling in for Gemini. Return structured JSON.", "You are a backup AI assistant filling in for Gemini. Return structured JSON.",
baseUrl, baseUrl,
modelName, modelName,