1/** 2 * Search-text normalization for the archive filter (#456). 3 * 4 * The archive filter matched query against title/summary/categories with a 5 * plain `toLowerCase().includes()`. That is exact-codepoint matching, so a 6 * record whose title uses typographic punctuation -- a curly apostrophe 7 * (U+2019), curly quotes (U+201C/D), an em or en dash (U+2014/2013), or an 8 * ellipsis (U+2026) -- never matched the ASCII a user actually types 9 * (' " - ...). Jay reported the symptom on the 2026-06-19 call: "I'd put in 10 * the headline of a piece I know is in there, and it says we don't have any 11 * such thing." 33 non-social article titles in the archive carry such 12 * characters. 13 * 14 * The fix is one normalization boundary: fold both the stored text and the 15 * query through `normalizeForSearch` before comparing, so the whole class of 16 * typographic mismatches collapses to ASCII and can't silently fail to match. 17 * Diacritics are folded too (cafe matches an accented cafe), which only widens 18 * recall. The prime (U+2032) is folded to an apostrophe too -- not present in 19 * today's data, but a common apostrophe lookalike in social and OCR text the 20 * archive ingests, so the boundary covers the whole class. (Double-prime 21 * U+2033 NFKD-decomposes to two primes, so it folds the same way for free.) 22 * 23 * Special characters are referenced by numeric code point (0x2019, etc.) rather 24 * than as literal glyphs, so the whole module is reviewable in plain ASCII. 25 */ 26 27// Record-internal field separator (U+0001). A real query never contains a 28// control character, so joining a record's fields with this and running a 29// single `includes()` can never match a phrase that spans two fields (e.g. a 30// title's last word plus a summary's first word). Keeps the old per-field 31// matching semantics while reducing the matcher to one substring test. 32const FIELD_SEP = String.fromCharCode(0x01); 33export const MIN_EXACT_PHRASE_WORDS = 2; 34 35/** 36 * Fold a string to a normalized, lower-case, ASCII-punctuation form for search. 37 * Lower-cases, NFKD-decomposes (so accented letters split into base + combining 38 * marks), then walks code points: typographic quotes/dashes become their ASCII 39 * equivalents, combining marks are dropped, everything else passes through. 40 * Whitespace runs are collapsed last. 41 * @param {*} str - any value; null/undefined/non-strings are coerced safely 42 * @returns {string} normalized text ('' for empty input) 43 */ 44export function normalizeForSearch(str) { 45 if (str == null) return ''; 46 const decomposed = String(str).toLowerCase().normalize('NFKD'); 47 let out = ''; 48 for (const ch of decomposed) { 49 const code = ch.codePointAt(0); 50 if ((code >= 0x2018 && code <= 0x201b) || code === 0x2032) { // single/low/high quotes, prime 51 out += "'"; 52 } else if (code >= 0x201c && code <= 0x201f) { // low/high double quotes 53 out += '"'; 54 } else if (code === 0x2013 || code === 0x2014 || code === 0x2015) { // en/em dash, bar 55 out += '-'; 56 } else if (code === 0x2026) { // horizontal ellipsis 57 out += '...'; 58 } else if (code >= 0x0300 && code <= 0x036f) { // combining diacritical marks 59 // drop 60 } else { 61 out += ch; 62 } 63 } 64 return out.replace(/\s+/g, ' ').trim(); 65} 66 67/** 68 * Tokenize text at non-letter/digit boundaries for exact phrases. This covers 69 * MiniSearch's whitespace/punctuation boundaries plus symbols, while reusing 70 * the archive's case, quote, and diacritic folding. 71 */ 72export function tokenizeSearchWords(value) { 73 return normalizeForSearch(value).split(/[^\p{L}\p{N}
73]+/u).filter(Boolean); 74} 75 76/** 77 * Parse supported quoted phrases without changing ordinary query text. 78 * MiniSearch supplies broad token recall; phrase postings verify adjacency. 79 */ 80export function parseSearchQuery(rawQuery) { 81 const source = rawQuery == null ? '' : String(rawQuery).trim(); 82 const normalized = normalizeForSearch(source); 83 const quoteCount = [...normalized].filter(character => character === '"').length; 84 if (quoteCount % 2 !== 0) { 85 return { 86 miniQuery: source, 87 phraseKeys: [], 88 phraseTokens: [], 89 unquotedTokens: tokenizeSearchWords(source), 90 }; 91 } 92 const phraseTokens = []; 93 const unquotedParts = []; 94 const quotePattern = /"([^"]+)"/g; 95 let cursor = 0; 96 let match; 97 98 while ((match = quotePattern.exec(normalized)) !== null) { 99 unquotedParts.push(normalized.slice(cursor, match.index)); 100 const tokens = tokenizeSearchWords(match[1]); 101 if (tokens.length >= MIN_EXACT_PHRASE_WORDS) { 102 phraseTokens.push(tokens); 103 } else { 104 unquotedParts.push(match[1]); 105 } 106 cursor = match.index + match[0].length; 107 } 108 unquotedParts.push(normalized.slice(cursor)); 109 110 if (phraseTokens.length === 0) { 111 return { 112 miniQuery: source, 113 phraseKeys: [], 114 phraseTokens: [], 115 unquotedTokens: tokenizeSearchWords(source), 116 }; 117 } 118 119 return { 120 miniQuery: source 121 .replace(/["\u201c\u201d\u201e\u201f]/gu, '') 122 .replace(/\s+/g, ' ') 123 .trim(), 124 phraseKeys: phraseTokens.map(tokens => tokens.join('~')), 125 phraseTokens, 126 unquotedTokens: tokenizeSearchWords(unquotedParts.join(' ')), 127 }; 128} 129 130function includesTokenSequence(fieldTokens, expectedTokens) { 131 if (expectedTokens.length > fieldTokens.length) return false; 132 const lastStart = fieldTokens.length - expectedTokens.length; 133 for (let start = 0; start <= lastStart; start += 1) { 134 if (expectedTokens.every((token, offset) => fieldTokens[start + offset] === token)) { 135 return true; 136 } 137 } 138 return false; 139} 140 141/** 142 * Apply quoted-query semantics to the already normalized in-memory card text. 143 * A phrase cannot cross a record field boundary. 144 */ 145export function matchesParsedSearchText(searchText, parsedQuery) { 146 const parsed = parsedQuery || parseSearchQuery(''); 147 if (parsed.phraseTokens.length === 0) { 148 return !parsed.miniQuery || searchText.includes(normalizeForSearch(parsed.miniQuery)); 149 } 150 151 const fieldTokens = String(searchText || '') 152 .split(FIELD_SEP) 153 .map(tokenizeSearchWords); 154 const phrasesMatch = parsed.phraseTokens.every(tokens => ( 155 fieldTokens.some(field => includesTokenSequence(field, tokens)) 156 )); 157 if (!phrasesMatch) return false; 158 159 const allTokens = fieldTokens.flat(); 160 return parsed.unquotedTokens.every(expected => ( 161 allTokens.some(token => token.startsWith(expected)) 162 )); 163} 164 165/** 166 * Search every loaded MiniSearch artifact. Quoted queries require each hit to 167 * be present in every exact phrase posting list. An artifact without postings 168 * cannot verify adjacency and contributes no quoted full-text hits. 169 */ 170export function searchLoadedIndexes(indexes, rawQuery) { 171 const parsed = parseSearchQuery(rawQuery); 172 if (!parsed.miniQuery) return []; 173 174 return (Array.isArray(indexes) ? indexes : []).flatMap((mini) => { 175 const hits = mini.search(parsed.miniQuery, { prefix: true, combineWith: 'AND' }); 176 if (parsed.phraseKeys.length === 0) return hits; 177 if (!(mini.phrasePostings instanceof Map)) return []; 178 179 return hits.filter(hit => parsed.phraseKeys.every(key => ( 180 mini.phrasePostings.get(key)?.has(hit.id) === true 181 ))); 182 }); 183} 184 185/** 186 * Find ordered autocomplete matches, deduplicated by normalized search text. 187 * The first spelling of an equivalent term is preserved for display. 188 * @param {Array<*>} terms - autocomplete terms in display-priority order 189 * @param {*} rawQuery - the user's unnormalized query 190 * @param {number} limit - maximum number of unique suggestions 191 * @returns {string[]} first-seen display terms matching the query 192 */ 193export function findSearchSuggestions(terms, rawQuery, limit = 8) { 194 if (!Array.isArray(terms) || !Number.isFinite(limit)) return []; 195 196 const query = normalizeForSearch(rawQuery); 197 const max = Math.max(0, Math.floor(limit)); 198 if (!query || max === 0) return []; 199 200 const seen = new Set(); 201 const suggestions = []; 202 203 for (const term of terms) { 204 if (typeof term !== 'string') continue;
205 const normalizedTerm = normalizeForSearch(term); 206 if ( 207 !normalizedTerm || 208 !normalizedTerm.includes(query) || 209 seen.has(normalizedTerm) 210 ) { 211 continue; 212 } 213 214 seen.add(normalizedTerm); 215 suggestions.push(term); 216 if (suggestions.length === max) break; 217 } 218 219 return suggestions; 220} 221 222/** 223 * Build the normalized, searchable blob for a record: title, summary, and each 224 * category, normalized and joined with a field separator so matches stay within 225 * a single field. Precompute this once per record on data load, then test each 226 * keystroke's normalized term against the blob. 227 * @param {Object} record - an archive record 228 * @returns {string} normalized searchable text 229 */ 230export function buildSearchText(record) { 231 if (!record) return ''; 232 const summary = record.summaryPreview || record.summary || ''; 233 const parts = [record.title || '', summary, ...(record.categories || [])]; 234 return parts.map(normalizeForSearch).join(FIELD_SEP); 235} 236 237/** 238 * Whether a record matches a raw search term. An empty term matches everything, 239 * mirroring the filter's "no search" state. 240 * @param {Object} record - an archive record 241 * @param {string} rawTerm - the user's unnormalized query 242 * @returns {boolean} 243 */ 244export function matchesSearch(record, rawTerm) { 245 const term = normalizeForSearch(rawTerm); 246 if (!term) return true; 247 return buildSearchText(record).includes(term); 248} 249 250export default { 251 normalizeForSearch, 252 tokenizeSearchWords, 253 parseSearchQuery, 254 matchesParsedSearchText, 255 searchLoadedIndexes, 256 findSearchSuggestions, 257 buildSearchText, 258 matchesSearch, 259};
Line numbers count LF bytes from the start of the resource, as the search results do. Vendor segments are library code the classifier recognised; they are stored but not indexed. Bytes are shown as Latin1 characters, one per byte.