diff --git a/main.js b/main.js index 8bc6351..0c65215 100644 --- a/main.js +++ b/main.js @@ -9507,10 +9507,10 @@ var ContentVectorizer = class { const parts = [ chunk.title, chunk.firstParagraph, - chunk.content.substring(0, 1e3), + chunk.content.substring(0, 500), // Limit content to avoid long prompts - chunk.headings.join(" "), - JSON.stringify(chunk.frontmatter) + chunk.headings.slice(0, 5).join(" ") + // Limit headings ].filter(Boolean); return parts.join("\n\n"); } @@ -10033,7 +10033,7 @@ var OllamaPlugin = class extends import_obsidian4.Plugin { const files = this.app.vault.getMarkdownFiles(); Logger.info(`Starting background vault indexing for ${files.length} files...`, "main"); let indexed = 0; - const BATCH_SIZE = 2; + const BATCH_SIZE = 1; const DELAY_MS = 500; for (let i = 0; i < files.length; i += BATCH_SIZE) { if (signal.aborted) { @@ -10041,20 +10041,18 @@ var OllamaPlugin = class extends import_obsidian4.Plugin { return; } const batch = files.slice(i, i + BATCH_SIZE); - await Promise.all( - batch.map(async (file) => { - if (signal.aborted) return; - try { - const content = await this.app.vault.read(file); - if (signal.aborted) return; - await this.vaultVectorStore.indexFile(file, content); - indexed++; - } catch (error) { - const errorMessage = error instanceof Error ? error.message : String(error); - Logger.warn(`Failed to index ${file.path}: ${errorMessage}`, "main"); - } - }) - ); + for (const file of batch) { + if (signal.aborted) break; + try { + const content = await this.app.vault.read(file); + if (signal.aborted) break; + await this.vaultVectorStore.indexFile(file, content); + indexed++; + } catch (error) { + const errorMessage = error instanceof Error ? error.message : String(error); + Logger.warn(`Failed to index ${file.path}: ${errorMessage}`, "main"); + } + } if (i + BATCH_SIZE < files.length) { await new Promise((resolve) => setTimeout(resolve, DELAY_MS)); } diff --git a/src/indexing-pipeline/vectorization.ts b/src/indexing-pipeline/vectorization.ts index fbbd2e1..11f6c42 100644 --- a/src/indexing-pipeline/vectorization.ts +++ b/src/indexing-pipeline/vectorization.ts @@ -92,13 +92,13 @@ export class ContentVectorizer { * Creates a prompt from content chunk for embedding */ private createPrompt(chunk: ContentChunk): string { - // Combine important elements for embedding + // Combine important elements for embedding, keeping it concise + // to avoid exceeding the embedding model's context window const parts = [ chunk.title, chunk.firstParagraph, - chunk.content.substring(0, 1000), // Limit content to avoid long prompts - chunk.headings.join(' '), - JSON.stringify(chunk.frontmatter), + chunk.content.substring(0, 500), // Limit content to avoid long prompts + chunk.headings.slice(0, 5).join(' '), // Limit headings ].filter(Boolean); return parts.join('\n\n'); diff --git a/src/main.ts b/src/main.ts index 2550640..f8049c2 100755 --- a/src/main.ts +++ b/src/main.ts @@ -166,7 +166,7 @@ export default class OllamaPlugin extends Plugin { Logger.info(`Starting background vault indexing for ${files.length} files...`, 'main'); let indexed = 0; - const BATCH_SIZE = 2; + const BATCH_SIZE = 1; const DELAY_MS = 500; for (let i = 0; i < files.length; i += BATCH_SIZE) { @@ -176,20 +176,19 @@ export default class OllamaPlugin extends Plugin { } const batch = files.slice(i, i + BATCH_SIZE); - await Promise.all( - batch.map(async (file) => { - if (signal.aborted) return; - try { - const content = await this.app.vault.read(file); - if (signal.aborted) return; - await this.vaultVectorStore!.indexFile(file, content); - indexed++; - } catch (error) { - const errorMessage = error instanceof Error ? error.message : String(error); - Logger.warn(`Failed to index ${file.path}: ${errorMessage}`, 'main'); - } - }) - ); + // Process files sequentially to avoid concurrent embedding requests + for (const file of batch) { + if (signal.aborted) break; + try { + const content = await this.app.vault.read(file); + if (signal.aborted) break; + await this.vaultVectorStore!.indexFile(file, content); + indexed++; + } catch (error) { + const errorMessage = error instanceof Error ? error.message : String(error); + Logger.warn(`Failed to index ${file.path}: ${errorMessage}`, 'main'); + } + } // Delay between batches to avoid overloading Ollama if (i + BATCH_SIZE < files.length) { diff --git a/tests/indexing-pipeline.test.ts b/tests/indexing-pipeline.test.ts index 205f796..e2a39a6 100644 --- a/tests/indexing-pipeline.test.ts +++ b/tests/indexing-pipeline.test.ts @@ -207,7 +207,7 @@ Content`; firstParagraph: 'First paragraph', wordCount: 2, chunkIndex: 0, - chunkSize: 100 + chunkSize: 100, }; const prompt = (vectorizer as any).createPrompt(mockChunk); @@ -215,7 +215,7 @@ Content`; expect(prompt).toContain('Test'); expect(prompt).toContain('First paragraph'); expect(prompt).toContain('Heading'); - expect(prompt).toContain('tags'); + // Frontmatter is no longer included in embedding prompts }); // Note: Actual embedding tests would require mocking fetch or integration testing @@ -233,7 +233,7 @@ Content`; firstParagraph: 'First paragraph', wordCount: 2, chunkIndex: 0, - chunkSize: 100 + chunkSize: 100, }; // Mock fetch to simulate an error @@ -290,12 +290,12 @@ This is a test document for pipeline processing.`; it('should process files in batches', async () => { const files: MockVaultFile[] = [ { basename: 'file1', path: 'file1.md' }, - { basename: 'file2', path: 'file2.md' } + { basename: 'file2', path: 'file2.md' }, ]; const fileContents = { 'file1.md': '# File 1\n\nContent 1', - 'file2.md': '# File 2\n\nContent 2' + 'file2.md': '# File 2\n\nContent 2', }; const results = await pipeline.processFilesInBatches(files, fileContents, 1); diff --git a/tests/vectorization.test.ts b/tests/vectorization.test.ts index abe03a1..1f225a5 100644 --- a/tests/vectorization.test.ts +++ b/tests/vectorization.test.ts @@ -200,8 +200,7 @@ describe('ContentVectorizer', () => { expect(prompt).toContain('This is the first paragraph'); expect(prompt).toContain('Main Heading'); expect(prompt).toContain('Sub Heading'); - expect(prompt).toContain('test'); - expect(prompt).toContain('2024-01-01'); + // Frontmatter is no longer included in embedding prompts }); it('should handle empty content fields gracefully', () => { @@ -221,8 +220,7 @@ describe('ContentVectorizer', () => { const prompt = (vectorizer as any).createPrompt(chunk); expect(prompt).toContain('Only content'); - // JSON.stringify({}) produces "{}", which is truthy so it's included - expect(prompt).toContain('{}'); + // Frontmatter is no longer included in embedding prompts }); it('should limit content length', () => { @@ -243,7 +241,7 @@ describe('ContentVectorizer', () => { const prompt = (vectorizer as any).createPrompt(chunk); expect(prompt).not.toContain('a'.repeat(1500)); - expect(prompt).toContain('a'.repeat(1000)); + expect(prompt).toContain('a'.repeat(500)); }); it('should handle missing frontmatter gracefully', () => {