cb621c83b5
- main.ts: process files sequentially (not Promise.all) to avoid concurrent embedding requests hammering Ollama; batch size reduced to 1 - vectorization.ts: shorten embedding prompts from 1000 to 500 chars, limit headings to 5, remove frontmatter from prompt to stay well within embedding model context window - Update vectorization and indexing-pipeline tests for new prompt format
309 lines
9.0 KiB
TypeScript
309 lines
9.0 KiB
TypeScript
import { ContentExtractor } from '../src/indexing-pipeline/extraction';
|
|
import { ContentNormalizer } from '../src/indexing-pipeline/normalization';
|
|
import { ContentVectorizer } from '../src/indexing-pipeline/vectorization';
|
|
import { IndexingPipeline } from '../src/indexing-pipeline/pipeline';
|
|
|
|
// Mock VaultFile interface for testing
|
|
interface MockVaultFile {
|
|
basename: string;
|
|
path: string;
|
|
}
|
|
|
|
describe('Indexing Pipeline Components', () => {
|
|
describe('ContentExtractor', () => {
|
|
let extractor: ContentExtractor;
|
|
|
|
beforeEach(() => {
|
|
extractor = new ContentExtractor();
|
|
});
|
|
|
|
it('should extract frontmatter correctly', () => {
|
|
const content = `---
|
|
title: Test Title
|
|
tags: algorithm, programming
|
|
date: 2023-01-01
|
|
---
|
|
|
|
# Heading
|
|
|
|
Content here`;
|
|
|
|
const file: MockVaultFile = { basename: 'test', path: 'test.md' };
|
|
const extracted = extractor.extractFromFile(file, content);
|
|
|
|
expect(extracted.frontmatter.title).toBe('Test Title');
|
|
expect(extracted.frontmatter.tags).toBe('algorithm, programming');
|
|
expect(extracted.frontmatter.date).toBe('2023-01-01');
|
|
expect(extracted.headings).toContain('Heading');
|
|
});
|
|
|
|
it('should extract headings correctly', () => {
|
|
const content = `# Heading 1
|
|
## Heading 2
|
|
### Heading 3
|
|
|
|
Content`;
|
|
|
|
const file: MockVaultFile = { basename: 'test', path: 'test.md' };
|
|
const extracted = extractor.extractFromFile(file, content);
|
|
|
|
expect(extracted.headings).toEqual(['Heading 1', 'Heading 2', 'Heading 3']);
|
|
});
|
|
|
|
it('should extract embedded code blocks', () => {
|
|
const content = `# Code Example
|
|
|
|
\`\`\`javascript
|
|
console.log('hello world');
|
|
\`\`\`
|
|
|
|
Some content`;
|
|
|
|
const file: MockVaultFile = { basename: 'test', path: 'test.md' };
|
|
const extracted = extractor.extractFromFile(file, content);
|
|
|
|
expect(extracted.embeddedCodeBlocks).toHaveLength(1);
|
|
expect(extracted.embeddedCodeBlocks[0]).toContain('console.log');
|
|
});
|
|
|
|
it('should extract first paragraph', () => {
|
|
const content = `First paragraph here.
|
|
|
|
Second paragraph here.
|
|
|
|
# Heading`;
|
|
|
|
const file: MockVaultFile = { basename: 'test', path: 'test.md' };
|
|
const extracted = extractor.extractFromFile(file, content);
|
|
|
|
expect(extracted.firstParagraph).toBe('First paragraph here.');
|
|
});
|
|
|
|
it('should extract raw text correctly', () => {
|
|
const content = `---
|
|
title: Test
|
|
---
|
|
|
|
# Heading
|
|
|
|
Content with **bold** and [link](url).
|
|
|
|
\`\`\`javascript
|
|
code
|
|
\`\`\``;
|
|
|
|
const rawText = extractor.extractRawText(content);
|
|
expect(rawText).not.toContain('---');
|
|
expect(rawText).not.toContain('# Heading');
|
|
expect(rawText).not.toContain('```javascript');
|
|
expect(rawText).toContain('Content with bold and link');
|
|
});
|
|
});
|
|
|
|
describe('ContentNormalizer', () => {
|
|
let normalizer: ContentNormalizer;
|
|
let extractor: ContentExtractor;
|
|
|
|
beforeEach(() => {
|
|
normalizer = new ContentNormalizer();
|
|
extractor = new ContentExtractor();
|
|
});
|
|
|
|
it('should normalize frontmatter dates to ISO format', () => {
|
|
const content = `---
|
|
title: Test
|
|
date: 2023-01-01
|
|
created: 2023-06-15
|
|
updated: invalid-date
|
|
tags: algorithm
|
|
---
|
|
|
|
Content`;
|
|
|
|
const file: MockVaultFile = { basename: 'test', path: 'test.md' };
|
|
const extracted = extractor.extractFromFile(file, content);
|
|
const normalized = normalizer.normalize(extracted);
|
|
|
|
expect(normalized.frontmatter.date).toBe('2023-01-01T00:00:00.000Z');
|
|
expect(normalized.frontmatter.created).toBe('2023-06-15T00:00:00.000Z');
|
|
expect(normalized.frontmatter.updated).toBe('invalid-date'); // Should preserve invalid dates
|
|
});
|
|
|
|
it('should convert tags to array format', () => {
|
|
const content = `---
|
|
title: Test
|
|
tags: algorithm, programming, javascript
|
|
---
|
|
|
|
Content`;
|
|
|
|
const file: MockVaultFile = { basename: 'test', path: 'test.md' };
|
|
const extracted = extractor.extractFromFile(file, content);
|
|
const normalized = normalizer.normalize(extracted);
|
|
|
|
expect(normalized.frontmatter.tags).toEqual(['algorithm', 'programming', 'javascript']);
|
|
});
|
|
|
|
it('should calculate word count correctly', () => {
|
|
const content = `# Title
|
|
|
|
This is a test document with several words to count.
|
|
It has multiple sentences and words to make it longer.`;
|
|
|
|
const file: MockVaultFile = { basename: 'test', path: 'test.md' };
|
|
const extracted = extractor.extractFromFile(file, content);
|
|
const normalized = normalizer.normalize(extracted);
|
|
|
|
expect(normalized.wordCount).toBeGreaterThan(0);
|
|
});
|
|
|
|
it('should extract tokens correctly', () => {
|
|
const content = `# Test Document
|
|
|
|
This is a test document with important keywords.`;
|
|
|
|
const file: MockVaultFile = { basename: 'test', path: 'test.md' };
|
|
const extracted = extractor.extractFromFile(file, content);
|
|
const normalized = normalizer.normalize(extracted);
|
|
|
|
expect(normalized.tokens).toContain('test');
|
|
expect(normalized.tokens).toContain('document');
|
|
expect(normalized.tokens).toContain('important');
|
|
expect(normalized.tokens).toContain('keywords');
|
|
});
|
|
|
|
it('should extract title correctly', () => {
|
|
const content = `# Test Document
|
|
|
|
Content`;
|
|
|
|
const file: MockVaultFile = { basename: 'test.md', path: 'test.md' };
|
|
const extracted = extractor.extractFromFile(file, content);
|
|
const normalized = normalizer.normalize(extracted);
|
|
|
|
expect(normalized.title).toBe('test');
|
|
});
|
|
});
|
|
|
|
describe('ContentVectorizer', () => {
|
|
let vectorizer: ContentVectorizer;
|
|
|
|
beforeEach(() => {
|
|
vectorizer = new ContentVectorizer({
|
|
model: 'nomic-embed-text',
|
|
ollamaUrl: 'http://localhost:11434',
|
|
});
|
|
});
|
|
|
|
it('should create a proper prompt from content chunk', () => {
|
|
const mockChunk = {
|
|
id: 'test',
|
|
path: 'test.md',
|
|
title: 'Test',
|
|
content: 'Test content',
|
|
tokens: ['test', 'content'],
|
|
headings: ['Heading'],
|
|
frontmatter: { tags: ['test'] },
|
|
firstParagraph: 'First paragraph',
|
|
wordCount: 2,
|
|
chunkIndex: 0,
|
|
chunkSize: 100,
|
|
};
|
|
|
|
const prompt = (vectorizer as any).createPrompt(mockChunk);
|
|
|
|
expect(prompt).toContain('Test');
|
|
expect(prompt).toContain('First paragraph');
|
|
expect(prompt).toContain('Heading');
|
|
// Frontmatter is no longer included in embedding prompts
|
|
});
|
|
|
|
// Note: Actual embedding tests would require mocking fetch or integration testing
|
|
it('should handle vectorization errors gracefully', async () => {
|
|
// This test would require mocking fetch to simulate error responses
|
|
// For now, we're just ensuring the method exists and doesn't crash
|
|
const mockChunk = {
|
|
id: 'test',
|
|
path: 'test.md',
|
|
title: 'Test',
|
|
content: 'Test content',
|
|
tokens: ['test', 'content'],
|
|
headings: ['Heading'],
|
|
frontmatter: { tags: ['test'] },
|
|
firstParagraph: 'First paragraph',
|
|
wordCount: 2,
|
|
chunkIndex: 0,
|
|
chunkSize: 100,
|
|
};
|
|
|
|
// Mock fetch to simulate an error
|
|
const originalFetch = global.fetch;
|
|
(global.fetch as any) = jest.fn().mockRejectedValue(new Error('Network error'));
|
|
|
|
try {
|
|
const result = await vectorizer.vectorize(mockChunk);
|
|
expect(result).toEqual([]);
|
|
} finally {
|
|
global.fetch = originalFetch;
|
|
}
|
|
});
|
|
});
|
|
|
|
describe('IndexingPipeline', () => {
|
|
let pipeline: IndexingPipeline;
|
|
|
|
beforeEach(() => {
|
|
pipeline = new IndexingPipeline({
|
|
ollamaUrl: 'http://localhost:11434',
|
|
embeddingModel: 'nomic-embed-text',
|
|
});
|
|
});
|
|
|
|
it('should process files through the pipeline', async () => {
|
|
const file: MockVaultFile = { basename: 'test', path: 'test.md' };
|
|
const content = `---
|
|
title: Test Document
|
|
tags: test, example
|
|
---
|
|
|
|
# Introduction
|
|
|
|
This is a test document for pipeline processing.`;
|
|
|
|
const result = await pipeline.processFile(file, content);
|
|
|
|
expect(result).not.toBeNull();
|
|
expect(result?.title).toBe('test');
|
|
expect(result?.path).toBe('test.md');
|
|
expect(result?.content).toContain('This is a test document for pipeline processing');
|
|
});
|
|
|
|
it('should handle processing errors gracefully', async () => {
|
|
const file: MockVaultFile = { basename: 'test', path: 'test.md' };
|
|
|
|
const result = await pipeline.processFile(file, '');
|
|
|
|
// Should not crash, but might return null or incomplete result
|
|
expect(result).toBeNull(); // Empty content should return null
|
|
});
|
|
|
|
it('should process files in batches', async () => {
|
|
const files: MockVaultFile[] = [
|
|
{ basename: 'file1', path: 'file1.md' },
|
|
{ basename: 'file2', path: 'file2.md' },
|
|
];
|
|
|
|
const fileContents = {
|
|
'file1.md': '# File 1\n\nContent 1',
|
|
'file2.md': '# File 2\n\nContent 2',
|
|
};
|
|
|
|
const results = await pipeline.processFilesInBatches(files, fileContents, 1);
|
|
|
|
expect(results).toHaveLength(2);
|
|
expect(results[0].title).toBe('file1');
|
|
expect(results[1].title).toBe('file2');
|
|
});
|
|
});
|
|
});
|