1/** 2 * Compute cosine similarity between two normalized vectors 3 * @param {number[]} a - First vector 4 * @param {number[]} b - Second vector 5 * @returns {number} Cosine similarity score between -1 and 1 6 */ 7function cosineSim(a, b) { 8 if (a.length !== b.length) { 9 throw new Error('Vectors must have the same length'); 10 } 11 12 let dot = 0; 13 let normA = 0; 14 let normB = 0; 15 16 for (let i = 0; i < a.length; i++) { 17 dot += a[i] * b[i]; 18 normA += a[i] * a[i]; 19 normB += b[i] * b[i]; 20 } 21 22 const norm = Math.sqrt(normA) * Math.sqrt(normB); 23 return norm === 0 ? 0 : dot / norm; 24} 25 26/** 27 * Find top K most similar items to a query vector 28 * @param {number[]} queryVec - Query embedding vector 29 * @param {Object} vectorData - Object with file paths as keys and {vector, metadata} as values 30 * @param {number} K - Number of top results to return 31 * @returns {Array} Array of {score, metadata} objects sorted by similarity score 32 */ 33export function topK(queryVec, vectorData, K = 10) { 34 if (!queryVec || !Array.isArray(queryVec)) { 35 throw new Error('Query vector must be a non-empty array'); 36 } 37 38 if (!vectorData || typeof vectorData !== 'object') { 39 throw new Error('Vector data must be an object'); 40 } 41 42 const results = []; 43 44 for (const [filePath, item] of Object.entries(vectorData)) { 45 if (!item.vector || !Array.isArray(item.vector)) { 46 console.warn(`Skipping item with invalid vector: ${filePath}`); 47 continue; 48 } 49 50 try { 51 const score = cosineSim(queryVec, item.vector); 52 results.push({ 53 score, 54 metadata: { 55 ...item.metadata, 56 filePath 57 } 58 }); 59 } catch (error) { 60 console.warn(`Error computing similarity for ${filePath}:`, error); 61 } 62 } 63 64 // Sort by score descending and return top K 65 return results 66 .sort((a, b) => b.score - a.score) 67 .slice(0, K); 68} 69 70/** 71 * Search for posts similar to a query 72 * @param {string} query - Search query 73 * @param {Object} vectorData - Pre-loaded vector data 74 * @param {Function} embedQuery - Function to embed the query 75 * @param {number} limit - Maximum number of results 76 * @returns {Promise<Array>} Search results with scores and metadata 77 */ 78export async function searchPosts(query, vectorData, embedQuery, limit = 10) { 79 if (!query || query.trim().length === 0) { 80 return []; 81 } 82 83 try { 84 const queryVector = await embedQuery(query); 85 const results = topK(queryVector, vectorData, limit); 86 87 // Add relevance threshold - only return results with score > 0.1 88 const relevantResults = results.filter(result => result.score > 0.1); 89 90 return relevantResults.map(result => ({ 91 ...result, 92 // Add percentage score for display 93 scorePercent: Math.round(result.score * 100) 94 })); 95 } catch (error) { 96 console.error('Search error:', error); 97 throw error; 98 } 99}
Line numbers count LF bytes from the start of the resource, as the search results do. Vendor segments are library code the classifier recognised; they are stored but not indexed. Bytes are shown as Latin1 characters, one per byte.