diff --git a/src/modules/search/infrastructure/InMemoryVectorDatabase.ts b/src/modules/search/infrastructure/InMemoryVectorDatabase.ts index 1b687e5e..47d5bff5 100644 --- a/src/modules/search/infrastructure/InMemoryVectorDatabase.ts +++ b/src/modules/search/infrastructure/InMemoryVectorDatabase.ts @@ -71,11 +71,11 @@ export class InMemoryVectorDatabase implements IVectorDatabase { params: SemanticSearchUrlsParams, ): Promise> { try { - console.log('all urls to compare', this.urls); - const threshold = params.threshold || 0; // Lower default threshold for more matches + console.log('Searching through', this.urls.size, 'indexed URLs'); const results: UrlSearchResult[] = []; + const queryTerms = this.extractSearchTerms(params.query); - console.log('Query content for similarity:', params.query); + console.log('Search terms:', queryTerms); for (const [url, indexed] of this.urls.entries()) { // Filter by URL type if specified @@ -83,31 +83,25 @@ export class InMemoryVectorDatabase implements IVectorDatabase { continue; } - const similarity = this.calculateSimilarity( - params.query, - indexed.content, - ); + const matchScore = this.calculateTextMatch(queryTerms, indexed.content); - console.log( - `Similarity between "${params.query}" and "${indexed.content}": ${similarity}`, - ); + console.log(`Match score for "${indexed.content}": ${matchScore}`); - if (similarity >= threshold) { + // Include result if any search terms match + if (matchScore > 0) { results.push({ url: indexed.url, - similarity, + similarity: matchScore, // Use match score as similarity metadata: indexed.metadata, }); } } - // Sort by similarity (highest first) and limit results + // Sort by match score (highest first) and limit results results.sort((a, b) => b.similarity - a.similarity); const limitedResults = results.slice(0, params.limit); - console.log( - `Found ${limitedResults.length} similar URLs above threshold ${threshold}`, - ); + console.log(`Found ${limitedResults.length} matching URLs`); return ok(limitedResults); } catch (error) { @@ -137,56 +131,35 @@ export class InMemoryVectorDatabase implements IVectorDatabase { } /** - * Simple text similarity calculation based on shared words - * Uses a more lenient scoring system to increase likelihood of matches + * Extract search terms from query string */ - private calculateSimilarity(text1: string, text2: string): number { - const words1 = this.tokenize(text1); - const words2 = this.tokenize(text2); - - if (words1.length === 0 && words2.length === 0) return 1; - if (words1.length === 0 || words2.length === 0) return 0; - - // Count shared words (with frequency) - const freq1 = this.getWordFrequency(words1); - const freq2 = this.getWordFrequency(words2); - - let sharedWords = 0; - let totalWords = 0; - - // Count shared words based on minimum frequency - for (const word of new Set([...words1, ...words2])) { - const count1 = freq1.get(word) || 0; - const count2 = freq2.get(word) || 0; - - if (count1 > 0 && count2 > 0) { - sharedWords += Math.min(count1, count2); - } - totalWords += Math.max(count1, count2); - } - - // Return ratio of shared words to total words - // This is more lenient than Jaccard similarity - return totalWords > 0 ? sharedWords / totalWords : 0; + private extractSearchTerms(query: string): string[] { + return query + .toLowerCase() + .replace(/[^\w\s]/g, ' ') + .split(/\s+/) + .filter((term) => term.length > 0); } /** - * Get word frequency map + * Calculate text match score based on how many search terms are found + * Returns a score from 0 to 1 based on the percentage of terms that match */ - private getWordFrequency(words: string[]): Map { - const freq = new Map(); - for (const word of words) { - freq.set(word, (freq.get(word) || 0) + 1); + private calculateTextMatch(searchTerms: string[], content: string): number { + if (searchTerms.length === 0) return 0; + + const contentLower = content.toLowerCase(); + let matchedTerms = 0; + + for (const term of searchTerms) { + // Check for exact word match or partial match + if (contentLower.includes(term)) { + matchedTerms++; + } } - return freq; - } - private tokenize(text: string): string[] { - return text - .toLowerCase() - .replace(/[^\w\s]/g, ' ') - .split(/\s+/) - .filter((word) => word.length > 1); // Allow shorter words for more matches + // Return percentage of terms that matched + return matchedTerms / searchTerms.length; } /**