PageSourceSearch

https://jonmagic.com/js/search.js

js jonmagic.com collected 2026-09-25 14:06:12 UTC 2,852 bytes, 99 lines download raw bytes

1/**
2 * Compute cosine similarity between two normalized vectors
3 * @param {number[]} a - First vector
4 * @param {number[]} b - Second vector
5 * @returns {number} Cosine similarity score between -1 and 1
6 */
7function cosineSim(a, b) {
8  if (a.length !== b.length) {
9    throw new Error('Vectors must have the same length');
10  }
11
12  let dot = 0;
13  let normA = 0;
14  let normB = 0;
15
16  for (let i = 0; i < a.length; i++) {
17    dot += a[i] * b[i];
18    normA += a[i] * a[i];
19    normB += b[i] * b[i];
20  }
21
22  const norm = Math.sqrt(normA) * Math.sqrt(normB);
23  return norm === 0 ? 0 : dot / norm;
24}
25
26/**
27 * Find top K most similar items to a query vector
28 * @param {number[]} queryVec - Query embedding vector
29 * @param {Object} vectorData - Object with file paths as keys and {vector, metadata} as values
30 * @param {number} K - Number of top results to return
31 * @returns {Array} Array of {score, metadata} objects sorted by similarity score
32 */
33export function topK(queryVec, vectorData, K = 10) {
34  if (!queryVec || !Array.isArray(queryVec)) {
35    throw new Error('Query vector must be a non-empty array');
36  }
37
38  if (!vectorData || typeof vectorData !== 'object') {
39    throw new Error('Vector data must be an object');
40  }
41
42  const results = [];
43
44  for (const [filePath, item] of Object.entries(vectorData)) {
45    if (!item.vector || !Array.isArray(item.vector)) {
46      console.warn(`Skipping item with invalid vector: ${filePath}`);
47      continue;
48    }
49
50    try {
51      const score = cosineSim(queryVec, item.vector);
52      results.push({
53        score,
54        metadata: {
55          ...item.metadata,
56          filePath
57        }
58      });
59    } catch (error) {
60      console.warn(`Error computing similarity for ${filePath}:`, error);
61    }
62  }
63
64  // Sort by score descending and return top K
65  return results
66    .sort((a, b) => b.score - a.score)
67    .slice(0, K);
68}
69
70/**
71 * Search for posts similar to a query
72 * @param {string} query - Search query
73 * @param {Object} vectorData - Pre-loaded vector data
74 * @param {Function} embedQuery - Function to embed the query
75 * @param {number} limit - Maximum number of results
76 * @returns {Promise<Array>} Search results with scores and metadata
77 */
78export async function searchPosts(query, vectorData, embedQuery, limit = 10) {
79  if (!query || query.trim().length === 0) {
80    return [];
81  }
82
83  try {
84    const queryVector = await embedQuery(query);
85    const results = topK(queryVector, vectorData, limit);
86
87    // Add relevance threshold - only return results with score > 0.1
88    const relevantResults = results.filter(result => result.score > 0.1);
89
90    return relevantResults.map(result => ({
91      ...result,
92      // Add percentage score for display
93      scorePercent: Math.round(result.score * 100)
94    }));
95  } catch (error) {
96    console.error('Search error:', error);
97    throw error;
98  }
99}

Line numbers count LF bytes from the start of the resource, as the search results do. Vendor segments are library code the classifier recognised; they are stored but not indexed. Bytes are shown as Latin1 characters, one per byte.