Fix local LLM prompt truncation and context window limit guard
This commit is contained in:
@@ -92,10 +92,15 @@ const generateTextWithLLMFallback = async (
|
|||||||
const modelName = llmConfig?.lmStudioModel || "local-model";
|
const modelName = llmConfig?.lmStudioModel || "local-model";
|
||||||
const apiKey = llmConfig?.customApiKey;
|
const apiKey = llmConfig?.customApiKey;
|
||||||
|
|
||||||
// 1. Direct LM Studio / Local LLM request
|
// Safely limit prompt length for local LLMs (max ~12,000 chars ~ 3000 tokens context limit)
|
||||||
if (primary === 'lmstudio' || primary === 'openai_compatible') {
|
const safePrompt = prompt.length > 12000 ? prompt.substring(0, 12000) + "\n\n[Content truncated for context size]" : prompt;
|
||||||
|
|
||||||
|
const hasValidGeminiKey = !!process.env.GEMINI_API_KEY && process.env.GEMINI_API_KEY.trim().length > 10;
|
||||||
|
|
||||||
|
// 1. Direct LM Studio / Local LLM request or Gemini key is missing
|
||||||
|
if (primary === 'lmstudio' || primary === 'openai_compatible' || !hasValidGeminiKey) {
|
||||||
console.log(`[LLM Router] Routing directly to Local LLM (${baseUrl})...`);
|
console.log(`[LLM Router] Routing directly to Local LLM (${baseUrl})...`);
|
||||||
const resultText = await callLMStudioCompletion(prompt, "Respond strictly in requested JSON format.", baseUrl, modelName, apiKey);
|
const resultText = await callLMStudioCompletion(safePrompt, "Respond strictly in requested JSON format.", baseUrl, modelName, apiKey);
|
||||||
return { text: resultText, providerUsed: `Local LLM (${modelName})` };
|
return { text: resultText, providerUsed: `Local LLM (${modelName})` };
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -112,7 +117,7 @@ const generateTextWithLLMFallback = async (
|
|||||||
|
|
||||||
const response = await ai.models.generateContent({
|
const response = await ai.models.generateContent({
|
||||||
model: "gemini-3.6-flash",
|
model: "gemini-3.6-flash",
|
||||||
contents: prompt,
|
contents: safePrompt,
|
||||||
config
|
config
|
||||||
});
|
});
|
||||||
|
|
||||||
@@ -127,7 +132,7 @@ const generateTextWithLLMFallback = async (
|
|||||||
console.log(`[LLM Router] Falling back to Local LM Studio endpoint at ${baseUrl}...`);
|
console.log(`[LLM Router] Falling back to Local LM Studio endpoint at ${baseUrl}...`);
|
||||||
try {
|
try {
|
||||||
const localResult = await callLMStudioCompletion(
|
const localResult = await callLMStudioCompletion(
|
||||||
`${prompt}\n\nIMPORTANT: Respond with clean JSON matching the requested structure.`,
|
`${safePrompt}\n\nIMPORTANT: Respond with clean JSON matching the requested structure.`,
|
||||||
"You are a backup AI assistant filling in for Gemini. Return structured JSON.",
|
"You are a backup AI assistant filling in for Gemini. Return structured JSON.",
|
||||||
baseUrl,
|
baseUrl,
|
||||||
modelName,
|
modelName,
|
||||||
|
|||||||
Reference in New Issue
Block a user