PageSourceSearch

https://pressthink.org/j/rosen-archive/frontend/services/semanticSearch.js?v=3.8.36

js pressthink.org collected 2026-10-02 04:26:21 UTC 6,022 bytes, 178 lines download raw bytes

1/**
2 * Main-thread client for the free-text semantic search worker (#279).
3 *
4 * Mirrors `semanticRecall.js`, with one deliberate difference: an aborted
5 * request does NOT terminate the worker. Search aborts on every keystroke, and
6 * the worker holds the loaded model, so terminating it would throw away the
7 * download the reader just paid for.
8 *
9 * The worker is created only when the reader turns the toggle on, so a visit
10 * that never uses semantic search costs nothing.
11 */
12const DEFAULT_WORKER_URL = new URL(
13  './semantic-search-worker.js?v=3.8.36',
14  import.meta.url,
15);
16
17// Cold start loads the model, which is tens of megabytes on a first use and can
18// take several seconds on a phone. A warm query encodes in well under a second.
19// One generous timeout covers both rather than reporting a slow first load as a
20// failure.
21export const SEMANTIC_SEARCH_TIMEOUT_MS = 60000;
22
23function makeAbortError() {
24  if (typeof DOMException === 'function') {
25    return new DOMException('Semantic search request aborted', 'AbortError');
26  }
27  const error = new Error('Semantic search request aborted');
28  error.name = 'AbortError';
29  return error;
30}
31
32function defaultWorkerFactory() {
33  if (typeof Worker !== 'function') {
34    throw new Error('Semantic search requires Web Worker support');
35  }
36  return new Worker(DEFAULT_WORKER_URL, {
37    type: 'module',
38    name: 'archive-semantic-search',
39  });
40}
41
42export function createSemanticSearchClient({
43  workerFactory = defaultWorkerFactory,
44  requestTimeoutMs = SEMANTIC_SEARCH_TIMEOUT_MS,
45} = {}) {
46  let worker = null;
47  let nextRequestId = 0;
48  const pending = new Map();
49
50  const cleanupRequest = (requestId) => {
51    const request = pending.get(requestId);
52    if (!request) return null;
53    pending.delete(requestId);
54    clearTimeout(request.timer);
55    request.signal?.removeEventListener('abort', request.onAbort);
56    return request;
57  };
58
59  const rejectAll = (error) => {
60    for (const requestId of [...pending.keys()]) {
61      cleanupRequest(requestId)?.reject(error);
62    }
63  };
64
65  const dropWorker = (error) => {
66    rejectAll(error);
67    worker?.terminate();
68    worker = null;
69  };
70
71  const handleMessage = (event) => {
72    const message = event?.data;
73    if (!message || !pending.has(message.requestId)) return;
74    const request = cleanupRequest(message.requestId);
75    if (message.type === 'semantic-warmup-result') {
76      request.resolve({ count: message.count ?? 0 });
77      return;
78    }
79    if (message.type === 'semantic-query-result') {
80      request.resolve({
81        query: message.query,
82        matches: Array.isArray(message.matches) ? message.matches : [],
83        count: message.count ?? 0,
84      });
85      return;
86    }
87    if (message.type === 'semantic-query-error') {
88      request.reject(new Error(message.error || 'Semantic search worker failed'));
89    }
90  };
91
92  const handleWorkerError = (event) => {
93    // A worker that fails to start (no module worker support, a blocked or
94    // missing script) reports here. Drop it so the toggle can fall back to
95    // lexical search instead of waiting for a timeout.
96    dropWorker(new Error(event?.message || 'Semantic search worker failed'));
97  };
98
99  const ensureWorker = () => {
100    if (worker) return worker;
101    worker = workerFactory();
102    worker.addEventListener('message', handleMessage);
103    worker.addEventListener('error', handleWorkerError);
104    return worker;
105  };
106
107  const post = (payload, { signal, timeoutMs = requestTimeoutMs } = {}) => {
108    if (signal?.aborted) return Promise.reject(makeAbortError());
109
110    let searchWorker;
111    try {
112      searchWorker = ensureWorker();
113    } catch (error) {
114      return Promise.reject(error);
115    }
116
117    const requestId = `semantic-search-${++nextRequestId}`;
118    return new Promise((resolve, reject) => {
119      // Abort forgets this one request and leaves the worker, and its loaded
120      // model, in place for the next keystroke.
121      const onAbort = () => cleanupRequest(requestId)?.reject(makeAbortError());
122      const timer = setTimeout(() => {
123        if (!pending.has(requestId)) return;
124        // A hung worker cannot be recovered by waiting, so replace it. The
125        // browser has the model cached by then, so the retry reloads it fast.
126        dropWorker(new Error('Semantic search request timed out'));
127      }, timeoutMs);
128
129      pending.set(requestId, { resolve, reject, timer, signal, onAbort });
130      signal?.addEventListener('abort', onAbort, { once: true });
131
132      try {
133        searchWorker.postMessage({ ...payload, requestId });
134      } catch (error) {
135        cleanupRequest(requestId)?.reject(error);
136      }
137    });
138  };
139
140  /** Load the artifact and the model without ranking anything. */
141  const warmup = (options) => post({ type: 'semantic-warmup' }, options);
142
143  /** Rank the archive against one query. Resolves { query, matches, count }. */
144  const search = (query, { k, minScore, ...options } = {}) => {
145    if (typeof query !== 'string' || !query.trim()) {
146      return Promise.reject(new Error('Semantic search requires a query'));
147    }
148    return post({ type: 'semantic-query', query, k, minScore }, options);
149  };
150
151  // Drop the worker and the model it holds. Called when the reader switches the
152  // toggle off: the download stays in the browser cache, so turning it back on
153  // reloads from disk. In-flight requests are rejected as aborts, not failures,
154  // so an answer that arrives after the reader has left cannot report the
155  // feature as broken.
156  const terminate = () => {
157    dropWorker(makeAbortError());
158  };
159
160  return { warmup, search, terminate };
161}
162
163let defaultClient;
164
165export function warmupSemanticSearch(options) {
166  defaultClient ||= createSemanticSearchClient();
167  return defaultClient.warmup(options);
168}
169
170export function requestSemanticSearch(query, options) {
171  defaultClient ||= createSemanticSearchClient();
172  return defaultClient.search(query, options);
173}
174
175export function terminateSemanticSearch() {
176  defaultClient?.terminate();
177  defaultClient = undefined;
178}

Line numbers count LF bytes from the start of the resource, as the search results do. Vendor segments are library code the classifier recognised; they are stored but not indexed. Bytes are shown as Latin1 characters, one per byte.