import { VaultIndexer } from '../src/vault-indexer'; import { VaultIndexEntry } from '../src/types'; // Mock Obsidian types interface MockTFile { basename: string; path: string; } interface MockVault { getMarkdownFiles: () => MockTFile[]; read: (file: MockTFile) => Promise; } describe('VaultIndexer', () => { let indexer: VaultIndexer; let mockVault: MockVault; beforeEach(() => { mockVault = { getMarkdownFiles: jest.fn().mockReturnValue([]), read: jest.fn(), }; indexer = new VaultIndexer(mockVault as unknown as any); jest.clearAllMocks(); }); describe('searchVault', () => { it('should return empty array when no files exist', async () => { mockVault.getMarkdownFiles = jest.fn().mockReturnValue([]); const results = await indexer.searchVault('test', 5); expect(results).toEqual([]); }); it('should return empty array for empty or whitespace-only query', async () => { const file: MockTFile = { basename: 'test', path: 'test.md' }; mockVault.getMarkdownFiles = jest.fn().mockReturnValue([file]); mockVault.read = jest.fn().mockResolvedValue('some content'); const results1 = await indexer.searchVault('', 5); const results2 = await indexer.searchVault(' ', 5); expect(results1).toEqual([]); expect(results2).toEqual([]); }); it('should return files matching the query', async () => { const file1: MockTFile = { basename: 'notes', path: 'notes.md' }; const file2: MockTFile = { basename: 'todo', path: 'todo.md' }; mockVault.getMarkdownFiles = jest.fn().mockReturnValue([file1, file2]); mockVault.read = jest.fn().mockImplementation(async (file: MockTFile) => { if (file.basename === 'notes') { return 'These are my important notes about programming algorithms'; } return 'Buy milk and eggs'; }); const results = await indexer.searchVault('programming', 5); expect(results.length).toBe(1); expect(results[0].title).toBe('notes'); expect(results[0].score).toBeGreaterThan(0); }); it('should respect the limit parameter', async () => { const files: MockTFile[] = []; for (let i = 0; i < 10; i++) { files.push({ basename: `file${i}`, path: `file${i}.md` }); } mockVault.getMarkdownFiles = jest.fn().mockReturnValue(files); mockVault.read = jest.fn().mockResolvedValue('important keyword test'); const results = await indexer.searchVault('keyword', 3); expect(results.length).toBeLessThanOrEqual(3); }); it('should return results sorted by score descending', async () => { const file1: MockTFile = { basename: 'one', path: 'one.md' }; const file2: MockTFile = { basename: 'two', path: 'two.md' }; mockVault.getMarkdownFiles = jest.fn().mockReturnValue([file1, file2]); mockVault.read = jest.fn().mockImplementation(async (file: MockTFile) => { if (file.basename === 'one') { return 'keyword keyword keyword important'; } return 'keyword'; }); const results = await indexer.searchVault('keyword', 5); if (results.length >= 2) { expect(results[0].score).toBeGreaterThanOrEqual(results[1].score); } }); it('should truncate content previews to 500 characters', async () => { const file: MockTFile = { basename: 'long', path: 'long.md' }; const longContent = 'content '.repeat(100); // Use meaningful words, not just 'a' mockVault.getMarkdownFiles = jest.fn().mockReturnValue([file]); mockVault.read = jest.fn().mockResolvedValue(longContent); const results = await indexer.searchVault('content', 5); expect(results.length).toBeGreaterThan(0); expect(results[0].content.length).toBeLessThanOrEqual(500); }); it('should process files in batches to handle large vaults', async () => { const files: MockTFile[] = []; for (let i = 0; i < 25; i++) { files.push({ basename: `file${i}`, path: `file${i}.md` }); } mockVault.getMarkdownFiles = jest.fn().mockReturnValue(files); mockVault.read = jest.fn().mockResolvedValue('important test keyword'); const results = await indexer.searchVault('test', 5); expect(mockVault.read).toHaveBeenCalledTimes(25); expect(results.length).toBeGreaterThan(0); }); it('should filter out files with zero score', async () => { const file1: MockTFile = { basename: 'match', path: 'match.md' }; const file2: MockTFile = { basename: 'nomatch', path: 'nomatch.md' }; mockVault.getMarkdownFiles = jest.fn().mockReturnValue([file1, file2]); mockVault.read = jest.fn().mockImplementation(async (file: MockTFile) => { if (file.basename === 'match') { return 'relevant keyword algorithm'; } return 'nothing relevant here at all'; }); const results = await indexer.searchVault('keyword', 5); expect(results.length).toBe(1); expect(results[0].title).toBe('match'); }); it('should handle vault.read errors gracefully', async () => { const file1: MockTFile = { basename: 'good', path: 'good.md' }; const file2: MockTFile = { basename: 'bad', path: 'bad.md' }; mockVault.getMarkdownFiles = jest.fn().mockReturnValue([file1, file2]); mockVault.read = jest.fn().mockImplementation(async (file: MockTFile) => { if (file.basename === 'good') { return 'important keyword test'; } throw new Error('Permission denied'); }); const consoleWarnSpy = jest.spyOn(console, 'warn').mockImplementation(); const results = await indexer.searchVault('keyword', 5); expect(results.length).toBe(1); expect(results[0].title).toBe('good'); expect(consoleWarnSpy).toHaveBeenCalledWith(expect.stringContaining('Permission denied')); consoleWarnSpy.mockRestore(); }); it('should give higher scores to title matches', async () => { const file1: MockTFile = { basename: 'algorithm', path: 'algorithm.md' }; const file2: MockTFile = { basename: 'other', path: 'other.md' }; mockVault.getMarkdownFiles = jest.fn().mockReturnValue([file1, file2]); mockVault.read = jest.fn().mockImplementation(async (file: MockTFile) => { if (file.basename === 'algorithm') { return 'Some other content here'; } return 'This file discusses algorithm design patterns'; }); const results = await indexer.searchVault('algorithm', 5); expect(results.length).toBe(2); // File with title match should be first expect(results[0].title).toBe('algorithm'); }); it('should give higher scores to heading matches', async () => { const file1: MockTFile = { basename: 'file1', path: 'file1.md' }; const file2: MockTFile = { basename: 'file2', path: 'file2.md' }; mockVault.getMarkdownFiles = jest.fn().mockReturnValue([file1, file2]); mockVault.read = jest.fn().mockImplementation(async (file: MockTFile) => { if (file.basename === 'file1') { return '# Algorithm Design\n\nThis discusses design patterns'; } return 'This file mentions algorithm somewhere in the body text'; }); const results = await indexer.searchVault('algorithm', 5); expect(results.length).toBe(2); // File with heading match should score higher expect(results[0].title).toBe('file1'); }); it('should give higher scores to frontmatter matches', async () => { const file1: MockTFile = { basename: 'file1', path: 'file1.md' }; const file2: MockTFile = { basename: 'file2', path: 'file2.md' }; mockVault.getMarkdownFiles = jest.fn().mockReturnValue([file1, file2]); mockVault.read = jest.fn().mockImplementation(async (file: MockTFile) => { if (file.basename === 'file1') { return '---\ntags: algorithm design\n---\n\nSome content here'; } return 'This file mentions algorithm in the body'; }); const results = await indexer.searchVault('algorithm', 5); expect(results.length).toBe(2); // File with frontmatter match should score higher expect(results[0].title).toBe('file1'); }); it('should handle phrase matching with bonus', async () => { const file1: MockTFile = { basename: 'file1', path: 'file1.md' }; const file2: MockTFile = { basename: 'file2', path: 'file2.md' }; mockVault.getMarkdownFiles = jest.fn().mockReturnValue([file1, file2]); mockVault.read = jest.fn().mockImplementation(async (file: MockTFile) => { if (file.basename === 'file1') { return 'This discusses the design pattern algorithm'; } return 'This discusses design and pattern and algorithm separately'; }); const results = await indexer.searchVault('design pattern', 5); expect(results.length).toBe(2); }); it('should filter out stop words from query', async () => { const file: MockTFile = { basename: 'test', path: 'test.md' }; mockVault.getMarkdownFiles = jest.fn().mockReturnValue([file]); mockVault.read = jest.fn().mockResolvedValue('important keyword here'); // Query with stop words should still find the keyword const results = await indexer.searchVault('the important keyword', 5); expect(results.length).toBe(1); expect(results[0].title).toBe('test'); }); it('should handle files with no matching content', async () => { const file: MockTFile = { basename: 'test', path: 'test.md' }; mockVault.getMarkdownFiles = jest.fn().mockReturnValue([file]); mockVault.read = jest.fn().mockResolvedValue('nothing relevant'); const results = await indexer.searchVault('nonexistent', 5); expect(results.length).toBe(0); }); }); describe('tokenize', () => { it('should filter out stop words', () => { const tokens = (indexer as any).tokenize('the quick brown fox'); expect(tokens).not.toContain('the'); expect(tokens).toContain('quick'); expect(tokens).toContain('brown'); expect(tokens).toContain('fox'); }); it('should convert to lowercase', () => { const tokens = (indexer as any).tokenize('Hello WORLD'); expect(tokens).toEqual(['hello', 'world']); }); it('should handle punctuation', () => { const tokens = (indexer as any).tokenize('Hello, world!'); expect(tokens).toEqual(['hello', 'world']); }); it('should filter very short tokens', () => { const tokens = (indexer as any).tokenize('a b test word'); expect(tokens).not.toContain('a'); expect(tokens).not.toContain('b'); expect(tokens).toContain('test'); expect(tokens).toContain('word'); }); }); describe('calculateWeightedScore', () => { it('should return 0 when no tokens match', () => { const content = 'important algorithm design'; const queryTokens = (indexer as any).tokenize('nonexistent'); const tokenized = (indexer as any).tokenizeContent(content, { basename: 'test', path: 'test.md', } as any); const score = (indexer as any).calculateWeightedScore(tokenized, '', queryTokens); expect(score).toBe(0); }); it('should score higher when more tokens match', () => { const content = 'algorithm design pattern implementation'; const tokenized = (indexer as any).tokenizeContent(content, { basename: 'test', path: 'test.md', } as any); const query1 = 'algorithm'; const query2 = 'algorithm design pattern'; const score1 = (indexer as any).calculateWeightedScore( tokenized, query1, (indexer as any).tokenize(query1) ); const score2 = (indexer as any).calculateWeightedScore( tokenized, query2, (indexer as any).tokenize(query2) ); expect(score2).toBeGreaterThan(score1); }); it('should be case insensitive', () => { const content = 'Important Algorithm Design'; const tokenized = (indexer as any).tokenizeContent(content, { basename: 'test', path: 'test.md', } as any); const query = 'important algorithm'; const score = (indexer as any).calculateWeightedScore( tokenized, query, (indexer as any).tokenize(query) ); expect(score).toBeGreaterThan(0); }); it('should handle word boundary matching', () => { const content = 'algorithm'; const tokenized = (indexer as any).tokenizeContent(content, { basename: 'test', path: 'test.md', } as any); const query = 'algorithm'; const score = (indexer as any).calculateWeightedScore( tokenized, query, (indexer as any).tokenize(query) ); expect(score).toBeGreaterThan(0); }); }); describe('exactMatch with stemming', () => { it('should match exact words', () => { expect((indexer as any).exactMatch('test', 'test')).toBe(true); }); it('should match plurals', () => { expect((indexer as any).exactMatch('tests', 'test')).toBe(true); expect((indexer as any).exactMatch('test', 'tests')).toBe(true); }); it('should handle -ed suffix', () => { expect((indexer as any).exactMatch('tested', 'test')).toBe(true); }); it('should handle -ing suffix', () => { expect((indexer as any).exactMatch('testing', 'test')).toBe(true); }); it('should not match unrelated words', () => { expect((indexer as any).exactMatch('apple', 'banana')).toBe(false); }); }); describe('tokenizeContent', () => { it('should extract headings from markdown', () => { const content = '# Heading 1\n\n# Heading 2\n\nSome content'; const tokenized = (indexer as any).tokenizeContent(content, { basename: 'test', path: 'test.md', } as any); expect(tokenized.headings).toContain('Heading 1'); expect(tokenized.headings).toContain('Heading 2'); }); it('should extract frontmatter', () => { const content = '---\ntags: algorithm\ntitle: test\n---\n\nSome content'; const tokenized = (indexer as any).tokenizeContent(content, { basename: 'test', path: 'test.md', } as any); expect(tokenized.frontmatter.tags).toBe('algorithm'); expect(tokenized.frontmatter.title).toBe('test'); }); it('should extract first paragraph', () => { const content = 'First paragraph here.\n\nSecond paragraph here.'; const tokenized = (indexer as any).tokenizeContent(content, { basename: 'test', path: 'test.md', } as any); expect(tokenized.firstParagraph).toContain('First'); expect(tokenized.firstParagraph).not.toContain('Second'); }); }); });