1/** 2 * Main-thread client for the free-text semantic search worker (#279). 3 * 4 * Mirrors `semanticRecall.js`, with one deliberate difference: an aborted 5 * request does NOT terminate the worker. Search aborts on every keystroke, and 6 * the worker holds the loaded model, so terminating it would throw away the 7 * download the reader just paid for. 8 * 9 * The worker is created only when the reader turns the toggle on, so a visit 10 * that never uses semantic search costs nothing. 11 */ 12const DEFAULT_WORKER_URL = new URL( 13 './semantic-search-worker.js?v=3.8.36', 14 import.meta.url, 15); 16 17// Cold start loads the model, which is tens of megabytes on a first use and can 18// take several seconds on a phone. A warm query encodes in well under a second. 19// One generous timeout covers both rather than reporting a slow first load as a 20// failure. 21export const SEMANTIC_SEARCH_TIMEOUT_MS = 60000; 22 23function makeAbortError() { 24 if (typeof DOMException === 'function') { 25 return new DOMException('Semantic search request aborted', 'AbortError'); 26 } 27 const error = new Error('Semantic search request aborted'); 28 error.name = 'AbortError'; 29 return error; 30} 31 32function defaultWorkerFactory() { 33 if (typeof Worker !== 'function') { 34 throw new Error('Semantic search requires Web Worker support'); 35 } 36 return new Worker(DEFAULT_WORKER_URL, { 37 type: 'module', 38 name: 'archive-semantic-search', 39 }); 40} 41 42export function createSemanticSearchClient({ 43 workerFactory = defaultWorkerFactory, 44 requestTimeoutMs = SEMANTIC_SEARCH_TIMEOUT_MS, 45} = {}) { 46 let worker = null; 47 let nextRequestId = 0; 48 const pending = new Map(); 49 50 const cleanupRequest = (requestId) => { 51 const request = pending.get(requestId); 52 if (!request) return null; 53 pending.delete(requestId); 54 clearTimeout(request.timer); 55 request.signal?.removeEventListener('abort', request.onAbort); 56 return request; 57 }; 58 59 const rejectAll = (error) => { 60 for (const requestId of [...pending.keys()]) { 61 cleanupRequest(requestId)?.reject(error); 62 } 63 }; 64 65 const dropWorker = (error) => { 66 rejectAll(error); 67 worker?.terminate(); 68 worker = null; 69 }; 70 71 const handleMessage = (event) => { 72 const message = event?.data; 73 if (!message || !pending.has(message.requestId)) return; 74 const request = cleanupRequest(message.requestId); 75 if (message.type === 'semantic-warmup-result') { 76 request.resolve({ count: message.count ?? 0 }); 77 return; 78 } 79 if (message.type === 'semantic-query-result') { 80 request.resolve({ 81 query: message.query, 82 matches: Array.isArray(message.matches) ? message.matches : [], 83 count: message.count ?? 0, 84 }); 85 return; 86 } 87 if (message.type === 'semantic-query-error') { 88 request.reject(new Error(message.error || 'Semantic search worker failed')); 89 } 90 }; 91 92 const handleWorkerError = (event) => { 93 // A worker that fails to start (no module worker support, a blocked or 94 // missing script) reports here. Drop it so the toggle can fall back to 95 // lexical search instead of waiting for a timeout. 96 dropWorker(new Error(event?.message || 'Semantic search worker failed')); 97 }; 98 99 const ensureWorker = () => { 100 if (worker) return worker; 101 worker = workerFactory(); 102 worker.addEventListener('message', handleMessage); 103 worker.addEventListener('error', handleWorkerError); 104 return worker; 105 }; 106 107 const post = (payload, { signal, timeoutMs = requestTimeoutMs } = {}) => { 108 if (signal?.aborted) return Promise.reject(makeAbortError()); 109 110 let searchWorker; 111 try { 112 searchWorker = ensureWorker(); 113 } catch (error) { 114 return Promise.reject(error); 115 } 116 117 const requestId = `semantic-search-${++nextRequestId}`; 118 return new Promise((resolve, reject) => { 119 // Abort forgets this one request and leaves the worker, and its loaded 120 // model, in place for the next keystroke. 121 const onAbort = () => cleanupRequest(requestId)?.reject(makeAbortError()); 122 const timer = setTimeout(() => { 123 if (!pending.has(requestId)) return; 124 // A hung worker cannot be recovered by waiting, so replace it. The 125 // browser has the model cached by then, so the retry reloads it fast. 126 dropWorker(new Error('Semantic search request timed out')); 127 }, timeoutMs); 128 129 pending.set(requestId, { resolve, reject, timer, signal, onAbort }); 130 signal?.addEventListener('abort', onAbort, { once: true }); 131
132 try { 133 searchWorker.postMessage({ ...payload, requestId }); 134 } catch (error) { 135 cleanupRequest(requestId)?.reject(error); 136 } 137 }); 138 }; 139 140 /** Load the artifact and the model without ranking anything. */ 141 const warmup = (options) => post({ type: 'semantic-warmup' }, options); 142 143 /** Rank the archive against one query. Resolves { query, matches, count }. */ 144 const search = (query, { k, minScore, ...options } = {}) => { 145 if (typeof query !== 'string' || !query.trim()) { 146 return Promise.reject(new Error('Semantic search requires a query')); 147 } 148 return post({ type: 'semantic-query', query, k, minScore }, options); 149 }; 150 151 // Drop the worker and the model it holds. Called when the reader switches the 152 // toggle off: the download stays in the browser cache, so turning it back on 153 // reloads from disk. In-flight requests are rejected as aborts, not failures, 154 // so an answer that arrives after the reader has left cannot report the 155 // feature as broken. 156 const terminate = () => { 157 dropWorker(makeAbortError()); 158 }; 159 160 return { warmup, search, terminate }; 161} 162 163let defaultClient; 164 165export function warmupSemanticSearch(options) { 166 defaultClient ||= createSemanticSearchClient(); 167 return defaultClient.warmup(options); 168} 169 170export function requestSemanticSearch(query, options) { 171 defaultClient ||= createSemanticSearchClient(); 172 return defaultClient.search(query, options); 173} 174 175export function terminateSemanticSearch() { 176 defaultClient?.terminate(); 177 defaultClient = undefined; 178}
Line numbers count LF bytes from the start of the resource, as the search results do. Vendor segments are library code the classifier recognised; they are stored but not indexed. Bytes are shown as Latin1 characters, one per byte.