PageSourceSearch

https://pressthink.org/j/rosen-archive/frontend/utils/searchNormalize.js?v=3.8.36

js pressthink.org collected 2026-10-02 04:26:18 UTC 9,576 bytes, 259 lines download raw bytes

1/**
2 * Search-text normalization for the archive filter (#456).
3 *
4 * The archive filter matched query against title/summary/categories with a
5 * plain `toLowerCase().includes()`. That is exact-codepoint matching, so a
6 * record whose title uses typographic punctuation -- a curly apostrophe
7 * (U+2019), curly quotes (U+201C/D), an em or en dash (U+2014/2013), or an
8 * ellipsis (U+2026) -- never matched the ASCII a user actually types
9 * ('  "  -  ...). Jay reported the symptom on the 2026-06-19 call: "I'd put in
10 * the headline of a piece I know is in there, and it says we don't have any
11 * such thing." 33 non-social article titles in the archive carry such
12 * characters.
13 *
14 * The fix is one normalization boundary: fold both the stored text and the
15 * query through `normalizeForSearch` before comparing, so the whole class of
16 * typographic mismatches collapses to ASCII and can't silently fail to match.
17 * Diacritics are folded too (cafe matches an accented cafe), which only widens
18 * recall. The prime (U+2032) is folded to an apostrophe too -- not present in
19 * today's data, but a common apostrophe lookalike in social and OCR text the
20 * archive ingests, so the boundary covers the whole class. (Double-prime
21 * U+2033 NFKD-decomposes to two primes, so it folds the same way for free.)
22 *
23 * Special characters are referenced by numeric code point (0x2019, etc.) rather
24 * than as literal glyphs, so the whole module is reviewable in plain ASCII.
25 */
26
27// Record-internal field separator (U+0001). A real query never contains a
28// control character, so joining a record's fields with this and running a
29// single `includes()` can never match a phrase that spans two fields (e.g. a
30// title's last word plus a summary's first word). Keeps the old per-field
31// matching semantics while reducing the matcher to one substring test.
32const FIELD_SEP = String.fromCharCode(0x01);
33export const MIN_EXACT_PHRASE_WORDS = 2;
34
35/**
36 * Fold a string to a normalized, lower-case, ASCII-punctuation form for search.
37 * Lower-cases, NFKD-decomposes (so accented letters split into base + combining
38 * marks), then walks code points: typographic quotes/dashes become their ASCII
39 * equivalents, combining marks are dropped, everything else passes through.
40 * Whitespace runs are collapsed last.
41 * @param {*} str - any value; null/undefined/non-strings are coerced safely
42 * @returns {string} normalized text ('' for empty input)
43 */
44export function normalizeForSearch(str) {
45  if (str == null) return '';
46  const decomposed = String(str).toLowerCase().normalize('NFKD');
47  let out = '';
48  for (const ch of decomposed) {
49    const code = ch.codePointAt(0);
50    if ((code >= 0x2018 && code <= 0x201b) || code === 0x2032) { // single/low/high quotes, prime
51      out += "'";
52    } else if (code >= 0x201c && code <= 0x201f) { // low/high double quotes
53      out += '"';
54    } else if (code === 0x2013 || code === 0x2014 || code === 0x2015) { // en/em dash, bar
55      out += '-';
56    } else if (code === 0x2026) {                  // horizontal ellipsis
57      out += '...';
58    } else if (code >= 0x0300 && code <= 0x036f) { // combining diacritical marks
59      // drop
60    } else {
61      out += ch;
62    }
63  }
64  return out.replace(/\s+/g, ' ').trim();
65}
66
67/**
68 * Tokenize text at non-letter/digit boundaries for exact phrases. This covers
69 * MiniSearch's whitespace/punctuation boundaries plus symbols, while reusing
70 * the archive's case, quote, and diacritic folding.
71 */
72export function tokenizeSearchWords(value) {
73  return normalizeForSearch(value).split(/[^\p{L}\p{N}
73]+/u).filter(Boolean);
74}
75
76/**
77 * Parse supported quoted phrases without changing ordinary query text.
78 * MiniSearch supplies broad token recall; phrase postings verify adjacency.
79 */
80export function parseSearchQuery(rawQuery) {
81  const source = rawQuery == null ? '' : String(rawQuery).trim();
82  const normalized = normalizeForSearch(source);
83  const quoteCount = [...normalized].filter(character => character === '"').length;
84  if (quoteCount % 2 !== 0) {
85    return {
86      miniQuery: source,
87      phraseKeys: [],
88      phraseTokens: [],
89      unquotedTokens: tokenizeSearchWords(source),
90    };
91  }
92  const phraseTokens = [];
93  const unquotedParts = [];
94  const quotePattern = /"([^"]+)"/g;
95  let cursor = 0;
96  let match;
97
98  while ((match = quotePattern.exec(normalized)) !== null) {
99    unquotedParts.push(normalized.slice(cursor, match.index));
100    const tokens = tokenizeSearchWords(match[1]);
101    if (tokens.length >= MIN_EXACT_PHRASE_WORDS) {
102      phraseTokens.push(tokens);
103    } else {
104      unquotedParts.push(match[1]);
105    }
106    cursor = match.index + match[0].length;
107  }
108  unquotedParts.push(normalized.slice(cursor));
109
110  if (phraseTokens.length === 0) {
111    return {
112      miniQuery: source,
113      phraseKeys: [],
114      phraseTokens: [],
115      unquotedTokens: tokenizeSearchWords(source),
116    };
117  }
118
119  return {
120    miniQuery: source
121      .replace(/["\u201c\u201d\u201e\u201f]/gu, '')
122      .replace(/\s+/g, ' ')
123      .trim(),
124    phraseKeys: phraseTokens.map(tokens => tokens.join('~')),
125    phraseTokens,
126    unquotedTokens: tokenizeSearchWords(unquotedParts.join(' ')),
127  };
128}
129
130function includesTokenSequence(fieldTokens, expectedTokens) {
131  if (expectedTokens.length > fieldTokens.length) return false;
132  const lastStart = fieldTokens.length - expectedTokens.length;
133  for (let start = 0; start <= lastStart; start += 1) {
134    if (expectedTokens.every((token, offset) => fieldTokens[start + offset] === token)) {
135      return true;
136    }
137  }
138  return false;
139}
140
141/**
142 * Apply quoted-query semantics to the already normalized in-memory card text.
143 * A phrase cannot cross a record field boundary.
144 */
145export function matchesParsedSearchText(searchText, parsedQuery) {
146  const parsed = parsedQuery || parseSearchQuery('');
147  if (parsed.phraseTokens.length === 0) {
148    return !parsed.miniQuery || searchText.includes(normalizeForSearch(parsed.miniQuery));
149  }
150
151  const fieldTokens = String(searchText || '')
152    .split(FIELD_SEP)
153    .map(tokenizeSearchWords);
154  const phrasesMatch = parsed.phraseTokens.every(tokens => (
155    fieldTokens.some(field => includesTokenSequence(field, tokens))
156  ));
157  if (!phrasesMatch) return false;
158
159  const allTokens = fieldTokens.flat();
160  return parsed.unquotedTokens.every(expected => (
161    allTokens.some(token => token.startsWith(expected))
162  ));
163}
164
165/**
166 * Search every loaded MiniSearch artifact. Quoted queries require each hit to
167 * be present in every exact phrase posting list. An artifact without postings
168 * cannot verify adjacency and contributes no quoted full-text hits.
169 */
170export function searchLoadedIndexes(indexes, rawQuery) {
171  const parsed = parseSearchQuery(rawQuery);
172  if (!parsed.miniQuery) return [];
173
174  return (Array.isArray(indexes) ? indexes : []).flatMap((mini) => {
175    const hits = mini.search(parsed.miniQuery, { prefix: true, combineWith: 'AND' });
176    if (parsed.phraseKeys.length === 0) return hits;
177    if (!(mini.phrasePostings instanceof Map)) return [];
178
179    return hits.filter(hit => parsed.phraseKeys.every(key => (
180      mini.phrasePostings.get(key)?.has(hit.id) === true
181    )));
182  });
183}
184
185/**
186 * Find ordered autocomplete matches, deduplicated by normalized search text.
187 * The first spelling of an equivalent term is preserved for display.
188 * @param {Array<*>} terms - autocomplete terms in display-priority order
189 * @param {*} rawQuery - the user's unnormalized query
190 * @param {number} limit - maximum number of unique suggestions
191 * @returns {string[]} first-seen display terms matching the query
192 */
193export function findSearchSuggestions(terms, rawQuery, limit = 8) {
194  if (!Array.isArray(terms) || !Number.isFinite(limit)) return [];
195
196  const query = normalizeForSearch(rawQuery);
197  const max = Math.max(0, Math.floor(limit));
198  if (!query || max === 0) return [];
199
200  const seen = new Set();
201  const suggestions = [];
202
203  for (const term of terms) {
204    if (typeof term !== 'string') continue;
205    const normalizedTerm = normalizeForSearch(term);
206    if (
207      !normalizedTerm ||
208      !normalizedTerm.includes(query) ||
209      seen.has(normalizedTerm)
210    ) {
211      continue;
212    }
213
214    seen.add(normalizedTerm);
215    suggestions.push(term);
216    if (suggestions.length === max) break;
217  }
218
219  return suggestions;
220}
221
222/**
223 * Build the normalized, searchable blob for a record: title, summary, and each
224 * category, normalized and joined with a field separator so matches stay within
225 * a single field. Precompute this once per record on data load, then test each
226 * keystroke's normalized term against the blob.
227 * @param {Object} record - an archive record
228 * @returns {string} normalized searchable text
229 */
230export function buildSearchText(record) {
231  if (!record) return '';
232  const summary = record.summaryPreview || record.summary || '';
233  const parts = [record.title || '', summary, ...(record.categories || [])];
234  return parts.map(normalizeForSearch).join(FIELD_SEP);
235}
236
237/**
238 * Whether a record matches a raw search term. An empty term matches everything,
239 * mirroring the filter's "no search" state.
240 * @param {Object} record - an archive record
241 * @param {string} rawTerm - the user's unnormalized query
242 * @returns {boolean}
243 */
244export function matchesSearch(record, rawTerm) {
245  const term = normalizeForSearch(rawTerm);
246  if (!term) return true;
247  return buildSearchText(record).includes(term);
248}
249
250export default {
251  normalizeForSearch,
252  tokenizeSearchWords,
253  parseSearchQuery,
254  matchesParsedSearchText,
255  searchLoadedIndexes,
256  findSearchSuggestions,
257  buildSearchText,
258  matchesSearch,
259};

Line numbers count LF bytes from the start of the resource, as the search results do. Vendor segments are library code the classifier recognised; they are stored but not indexed. Bytes are shown as Latin1 characters, one per byte.