fix: remove artificial MAX_STREAM_CHUNKS limit that cut off long responses
The plugin hard-capped streaming responses at 1000 chunks. For large models like qwen2.5:32b generating detailed answers, this limit was easily exceeded, causing the response to stop mid-sentence. The Ollama stream already terminates naturally when the model sends the final done signal, so the artificial chunk limit served no purpose. - src/chat-view.ts: Remove chunkCount tracking and MAX_STREAM_CHUNKS constant. Let the stream run until Ollama signals completion.
This commit is contained in:
@@ -487,7 +487,6 @@ export class ChatView extends ItemView {
|
||||
|
||||
let fullResponse = '';
|
||||
let toolCalls: OllamaToolCall[] = [];
|
||||
let chunkCount = 0;
|
||||
|
||||
for await (const chunk of stream) {
|
||||
if (chunk.content) {
|
||||
@@ -502,11 +501,6 @@ export class ChatView extends ItemView {
|
||||
if (chunk.tool_calls) {
|
||||
toolCalls = [...toolCalls, ...chunk.tool_calls];
|
||||
}
|
||||
|
||||
chunkCount++;
|
||||
if (chunkCount > MAX_STREAM_CHUNKS) {
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
// Process tool calls if any
|
||||
@@ -575,5 +569,4 @@ export class ChatView extends ItemView {
|
||||
private conversationStateManager: ConversationStateManager;
|
||||
}
|
||||
|
||||
const MAX_STREAM_CHUNKS = 1000;
|
||||
const MAX_TOOL_CALLS = 5;
|
||||
|
||||
Reference in New Issue
Block a user