PageSourceSearch

https://pressthink.org/j/rosen-archive/frontend/services/archiveService.js?v=3.8.36

js pressthink.org collected 2026-10-02 04:26:05 UTC 32,930 bytes, 909 lines download raw bytes

1
2import { DATA_CONFIG } from '../constants.js?v=3.8.36';
3import {
4  initDatabase,
5  loadArchiveData as loadSqliteData,
6  isReady as isSqliteReady,
7  queryAsObjects,
8  getRecordCountByYear,
9  getRecordCountByCategory,
10  getRecordCountByEra,
11  getMostMentionedEntities,
12  getMostCommonConcepts,
13  getCategoryCoOccurrence,
14  searchRecords as sqlSearchRecords,
15  getStats as getSqliteStats
16} from './sqliteService.js?v=3.8.36';
17import { IS_LOCAL, BASE_PATH } from '../utils/pathResolver.js?v=3.8.36';
18import { searchIndexOptions, socialSearchIndexOptions } from '../utils/searchConfig.js?v=3.8.36';
19import { escapeCsvCell } from '../utils/csvSafety.js?v=3.8.36';
20import { idbGet, idbSet, idbClear } from './idbCache.js?v=3.8.36';
21import { CACHE_VERSION, CACHE_TTL_MS, MAX_LOCALSTORAGE_SIZE, cacheKeyFor } from './cacheConfig.js?v=3.8.36';
22import { raceTimeout } from '../utils/raceTimeout.js?v=3.8.36';
23import { createResilientSearchIndexLoader, loadSearchIndexArtifact } from './searchIndexLoader.js?v=3.8.36';
24import { loadReleaseMetadata } from './releaseMetadata.js?v=3.8.36';
25
26// Routine cache-hit / fetch-start logs are silent in production. Set
27// `localStorage.jrda_debug = '1'` in DevTools and reload to opt in (#170).
28// Wrapped in try/catch because localStorage access can throw SecurityError
29// in privacy modes or when storage is blocked by browser policy; a throw
30// here would prevent this module (and therefore the whole app) from loading.
31const DEBUG = (() => {
32  try {
33    return typeof localStorage !== 'undefined' && localStorage.getItem('jrda_debug') === '1';
34  } catch {
35    return false;
36  }
37})();
38const debug = DEBUG ? console.log.bind(console) : () => {};
39
40// Simple hash function for UI color selection (djb1 variant)
41// Used by App.js to deterministically assign colors to categories
42export const hashString = (str) => {
43  let hash = 0;
44  for (let i = 0; i < str.length; i++) hash = (hash << 5) - hash + str.charCodeAt(i);
45  return Math.abs(hash);
46};
47
48// ============================================
49// LAZY LOADING STATE
50// ============================================
51
52// Cache for loaded details (populated on demand)
53let detailsCache = null;
54let detailsLoading = false;
55let detailsLoadPromise = null;
56
57// Cache for loaded entities (populated on demand)
58let entitiesCache = null;
59let entitiesLoading = false;
60let entitiesLoadPromise = null;
61
62// Entity lookup maps - populated when data is loaded
63let entityById = new Map();        // entity_id -> entity object
64let entityToRecords = new Map();   // entity_id -> Set of record ids
65let recordToEntities = new Map();  // record_id -> Set of entity ids
66
67/**
68 * Build entity lookup maps from loaded archive data
69 * Called after archive data is fetched
70 */
71export const buildEntityMaps = (data) => {
72  entityById.clear();
73  entityToRecords.clear();
74  recordToEntities.clear();
75
76  // Build entity ID -> entity object map
77  if (data.entities && Array.isArray(data.entities)) {
78    data.entities.forEach(entity => {
79      entityById.set(entity.id, entity);
80    });
81  }
82
83  // Build bidirectional record <-> entity maps
84  if (data.records && Array.isArray(data.records)) {
85    data.records.forEach(record => {
86      const entityIds = record.relatedIds || [];
87      recordToEntities.set(record.id, new Set(entityIds));
88
89      entityIds.forEach(entityId => {
90        if (!entityToRecords.has(entityId)) {
91          entityToRecords.set(entityId, new Set());
92        }
93        entityToRecords.get(entityId).add(record.id);
94      });
95    });
96  }
97
98  console.log(`Built entity maps: ${entityById.size} entities, ${recordToEntities.size} records`);
99};
100
101/**
102 * Get entity by ID
103 */
104export const getEntityById = (entityId) => entityById.get(entityId);
105
106/**
107 * Get all records that mention a specific entity
108 */
109export const getRecordsByEntity = (entityId) => {
110  const recordIds = entityToRecords.get(entityId);
111  return recordIds ? Array.from(recordIds) : [];
112};
113
114/**
115 * Get all entities mentioned in a record
116 */
117export const getEntitiesByRecord = (recordId) => {
118  const entityIds = recordToEntities.get(recordId);
119  if (!entityIds) return [];
120  return Array.from(entityIds).map(id =>
120 entityById.get(id)).filter(Boolean);
121};
122
123/**
124 * Find shared entities between two records
125 * @param {string} recordId1 - First record ID
126 * @param {string} recordId2 - Second record ID
127 * @param {string|null} entityTypeFilter - Optional entity type filter (Person, Organization, Concept, etc.)
128 * @returns {Array} Array of shared entity objects with prominence scores
129 */
130export const findSharedEntities = (recordId1, recordId2, entityTypeFilter = null) => {
131  const entities1 = recordToEntities.get(recordId1);
132  const entities2 = recordToEntities.get(recordId2);
133
134  if (!entities1 || !entities2) return [];
135
136  const shared = [];
137  entities1.forEach(entityId => {
138    if (entities2.has(entityId)) {
139      const entity = entityById.get(entityId);
140      if (entity) {
141        // Apply type filter if specified
142        if (!entityTypeFilter || entity.type === entityTypeFilter) {
143          shared.push(entity);
144        }
145      }
146    }
147  });
148
149  // Sort by prominence (higher first)
150  return shared.sort((a, b) => (b.prominence || 0) - (a.prominence || 0));
151};
152
153/**
154 * Calculate connection strength between two records based on shared entities
155 * @param {string} recordId1 - First record ID
156 * @param {string} recordId2 - Second record ID
157 * @param {string|null} entityTypeFilter - Optional entity type filter
158 * @returns {Object} { strength: number, sharedEntities: Array }
159 */
160export const calculateEntityConnectionStrength = (recordId1, recordId2, entityTypeFilter = null) => {
161  const sharedEntities = findSharedEntities(recordId1, recordId2, entityTypeFilter);
162
163  if (sharedEntities.length === 0) {
164    return { strength: 0, sharedEntities: [], prominenceScore: 0 };
165  }
166
167  // Calculate weighted strength based on prominence scores
168  const prominenceScore = sharedEntities.reduce((sum, e) => sum + (e.prominence || 1), 0);
169
170  return {
171    strength: sharedEntities.length,
172    sharedEntities,
173    prominenceScore
174  };
175};
176
177const DISSERTATION_RECORD = {
178  id: 'dissertation-1986',
179  title: 'The Impossible Press: American Journalism and the Decline of Public Life',
180  author: 'Jay Rosen',
181  date: '1986-01-01',
182  year: '1986',
183  era: 'Public Journalism (90s)',
184  pub: 'New York University (Ph.D. Dissertation)',
185  url: IS_LOCAL ? './dissertation/reader/' : `${BASE_PATH}/dissertation/reader/`,
186  summary: 'Rosen\'s doctoral dissertation traces the history of the idea that the function of the press is to inform the public. It argues that the rise of the mass circulation newspaper, while creating a technical ability to reach everyone, actually undermined the conditions necessary for a "universal town meeting." Drawing heavily on Walter Lippmann and John Dewey, it suggests that the professionalization of journalism ("objectivity") was a retreat from the problem of creating a genuine public life in a complex society. It contrasts news as "symptom" vs. news as "symbol" and explores how the press creates a "pseudo-environment" of public opinion.',
187  quote: 'An impossible press was born, one which sought to solve the whole problem of public life simply by controlling the conduct of journalists.',
188  categories: ['Journalism Theory & Practice', 'Politics & Democracy', 'Press & Media Criticism', 'Audience & Public Engagement'],
189  concepts: ['Public Sphere', 'Omnicompetent Citizen', 'Objectivity', 'Mass Society', 'Professionalism', 'Communication vs Community', 'Democracy and Distance'],
190  tags: ['Walter Lippmann', 'John Dewey', 'James Gordon Bennett', 'Joseph Pulitzer', 'Penny Press', 'Yellow Journalism', 'Robert Park', 'Tocqueville'],
191  verified: true,
192  type: 'Dissertation'
193};
194
195// Cache configuration (CACHE_VERSION / CACHE_TTL_MS / MAX_LOCALSTORAGE_SIZE
196// and the cacheKeyFor hash) lives in cacheConfig.js rather than inline.
197
198/**
199 * Check version.json on the server. If the version has changed,
200 * clear all caches so users get fresh data after deploys.
201 */
202// Memoise the in-flight Promise so concurrent callers all await the same
203// fetch instead of racing past a sync boolean (#171). Released on settle
204// so a hung or failed check doesn't poison every future call in the session.
205let versionCheckPromise = null;
206const checkVersion = () => {
207  if (versionCheckPromise) return versionCheckPromise;
208  const pending = (async () => {
209    try {
210      const { version } = await loadReleaseMetadata();
211      const stored = localStorage.getItem('jrda_deploy_version');
212      if (stored && stored !== version) {
213        console.log('[Cache] Deploy version changed, clearing caches');
214        clearArchiveCache();
215      }
216      localStorage.setItem('jrda_deploy_version', version);
217    } catch { /* version.json not available, skip */ }
218  })();
219  versionCheckPromise = pending;
220  pending.finally(() => {
221    if (versionCheckPromise === pending) versionCheckPromise = null;
222  });
223  return pending;
224};
225
226// Bound the wait at the call site so a slow or hung version.json can't
227// stall a load a good cache could satisfy. The check stays in flight in
228// the background and clears the cache if it eventually returns.
229const VERSION_CHECK_TIMEOUT_MS = 4000;
230
231const getCachedData = (url) => {
232  try {
233    const cacheKey = cacheKeyFor(url);
234    // localStorage is the only writer now (#337); still read a legacy
235    // sessionStorage entry first so caches from older builds expire cleanly.
236    let cached = sessionStorage.getItem(cacheKey) || localStorage.getItem(cacheKey);
237    if (!cached) return null;
238
239    const entry = JSON.parse(cached);
240    const now = Date.now();
241
242    if (entry.version !== CACHE_VERSION || (now - entry.timestamp) > CACHE_TTL_MS) {
243      try { sessionStorage.removeItem(cacheKey); } catch {}
244      try { localStorage.removeItem(cacheKey); } catch {}
245      return null;
246    }
247
248    return entry.data;
249  } catch (e) {
250    console.warn('Cache read error:', e);
251    return null;
252  }
253};
254
255const setCachedData = (url, data) => {
256  const cacheKey = cacheKeyFor(url);
257  const entry = {
258    data,
259    timestamp: Date.now(),
260    version: CACHE_VERSION
261  };
262
263  try {
264    const serialized = JSON.stringify(entry);
265
266    // Web Storage gives localStorage and sessionStorage a ~5 MB quota each, so a
267    // payload too large for localStorage will not fit sessionStorage either. The
268    // old overflow-to-sessionStorage attempt therefore threw QuotaExceededError on
269    // every load for the large data dumps (archive-core ~13 MB, archive-details
270    // ~13 MB, archive-data ~30 MB) and cached nothing. Skip it: archive-core also
271    // has an IndexedDB cache (fetchCoreData, far larger quota), and all of these
272    // payloads are served from the service-worker Cache Storage on refetch
273    // (sw.js stale-while-revalidate), so Web Storage is redundant for them. (#337)
274    // These files are not in Web Storage and only archive-core is in IndexedDB,
275    // so when the service worker is unavailable they refetch on every
276    // navigation. That residual gap is surfaced once at startup (the
277    // SW-registration block in index.html), not cached here (#428).
278    if (serialized.length > MAX_LOCALSTORAGE_SIZE) {
279      return;
280    }
281
282    // Small data goes to localStorage (persists across sessions)
283    localStorage.setItem(cacheKey, serialized);
284  } catch (e) {
285    if (e.name === 'QuotaExceededError' || e.code === 22) {
286      console.log('Cache storage full, clearing old archive caches...');
287      try {
288        clearArchiveCache();
289        localStorage.setItem(cacheKey, JSON.stringify(entry));
290      } catch {
291        console.warn('Cache disabled: browser storage is full.');
292      }
293    } else {
294      console.warn('Cache write error:', e);
295    }
296  }
297};
298
299// Clear all archive caches from both localStorage and sessionStorage
300export const clearArchiveCache = () => {
301  try {
302    for (const storage of [localStorage, sessionStorage]) {
303      const keys = Object.keys(storage);
304      const archiveKeys = keys.filter(key => key.startsWith('archive_json_') || key.startsWith('archive_csv_'));
305      archiveKeys.forEach(key => storage.removeItem(key));
306    }
307    console.log('Archive caches cleared');
308  } catch (e) {
309    console.warn('Error clearing cache:', e);
310  }
311  // Also drop the IndexedDB core-data store. Fire-and-forget: idbClear never
312  // rejects (it resolves false on failure), and the version-namespaced key
313  // (coreIdbKey) already prevents a stale read, so callers needn't await this.
314  idbClear();
315};
316
317// IndexedDB cache key for the core payload. Namespaced by the manual
318// CACHE_VERSION knob and the deploy version (version.json, stored by
319// checkVersion), so a deploy or a cache-version bump addresses a fresh key —
320// a stale blob is never read, by construction, without relying on a racy
321// async clear. clearArchiveCache additionally wipes the store to bound growth.
322const readDeployVersion = () => {
323  try {
324    return localStorage.getItem('jrda_deploy_version') || 'novers';
325  } catch {
326    return 'novers';
327  }
328};
329const coreIdbKey = () => `archive-core::${CACHE_VERSION}::${readDeployVersion()}`;
330const makeCoreEntry = (data) => ({ data, timestamp: Date.now(), version: CACHE_VERSION });
331const isFreshEntry = (entry) =>
332  !!entry &&
333  entry.version === CACHE_VERSION &&
334  typeof entry.timestamp === 'number' &&
335  (Date.now() - entry.timestamp) <= CACHE_TTL_MS;
336
337/**
338 * Fetch core archive data (lightweight, for initial page load)
339 * This is the new optimized entry point - loads ~8MB instead of ~25MB
340 */
341export const fetchCoreData = async () => {
342  const dataUrl = DATA_CONFIG.archive_core;
343
344  // Check deploy version (clears caches if version changed), bounded so a
345  // slow version.json doesn't stall a load a good cache could satisfy.
346  await raceTimeout(checkVersion(), VERSION_CHECK_TIMEOUT_MS);
347
348  const idbKey = coreIdbKey();
349
350  // 1. IndexedDB cache — structured-clones the object on read, skipping the
351  // JSON.parse of the ~13 MB blob, and persists across tab close (the blob
352  // exceeds localStorage's ~5 MB cap, so the Web Storage fallback below lands
353  // in sessionStorage, which a fresh tab never sees). See idbCache.js (#275).
354  const idbEntry = await idbGet(idbKey);
355  if (isFreshEntry(idbEntry)) {
356    debug('Using IndexedDB-cached core data');
357    return idbEntry.data;
358  }
359
360  // 2. Web Storage cache — covers browsers where IndexedDB is blocked (Safari
361  // Private, Firefox strict tracking protection). Promote a hit into
362  // IndexedDB, best-effort and unawaited, so the next read skips the parse.
363  const cached = getCachedData(dataUrl);
364  if (cached) {
365    debug('Using Web Storage-cached core data');
366    idbSet(idbKey, makeCoreEntry(cached));
367    return cached;
368  }
369
370  debug('Fetching core data from:', dataUrl);
371
372  try {
373    const response = await fetch(dataUrl);
374    if (!response.ok) {
375      throw new Error(`HTTP error! status: ${response.status}`);
376    }
377
378    const data = await response.json();
379
380    // Inject dissertation record if not present
381    if (!data.records.find(r => r.id === 'dissertation-1986')) {
382      data.records.push({
383        id: DISSERTATION_RECORD.id,
384        title: DISSERTATION_RECORD.title,
385        date: DISSERTATION_RECORD.date,
386        year: DISSERTATION_RECORD.year,
387        era: DISSERTATION_RECORD.era,
388        pub: DISSERTATION_RECORD.pub,
389        categories: DISSERTATION_RECORD.categories,
390        type: DISSERTATION_RECORD.type,
391        verified: DISSERTATION_RECORD.verified,
392        summaryPreview: DISSERTATION_RECORD.summary.substring(0, 180) + '...'
393      });
394
395      // Also add dissertation facets if missing
396      DISSERTATION_RECORD.categories.forEach(c => {
397        if (!data.facets.categories.includes(c)) {
398          data.facets.categories.push(c);
399        }
400      });
401      data.facets.categories.sort();
402    }
403
404    // Cache the result. IndexedDB first; fall back to Web Storage only when
405    // IndexedDB is unavailable, so capable browsers skip the redundant ~13 MB
406    // sessionStorage write (and its JSON.stringify) entirely.
407    const storedInIdb = await idbSet(idbKey, makeCoreEntry(data));
408    if (!storedInIdb) {
409      setCachedData(dataUrl, data);
410    }
411
412    return data;
413  } catch (error) {
414    // Propagate the failure. Returning a DISSERTATION_RECORD-only fallback
415    // here would mask a real outage (archive-core.json 404/503/parse error,
416    // partial deploy) as a successful load of a 1-record archive — visitors
417    // and monitors could not tell it apart from the real one. App.js's
418    // .catch renders the explicit "Unable to load archive" error state
419    // instead.
420    console.error('Error fetching core data:', error);
421    throw error;
422  }
423};
424
425/**
426 * Fetch record details (on-demand, when modal opens)
427 * Returns full summary, quote, concepts, tags, url, author, relatedIds
428 */
429export const fetchRecordDetails = async (recordId) => {
430  // Return dissertation details from constant
431  if (recordId === 'dissertation-1986') {
432    return {
433      summary: DISSERTATION_RECORD.summary,
434      quote: DISSERTATION_RECORD.quote,
435      concepts: DISSERTATION_RECORD.concepts,
436      tags: DISSERTATION_RECORD.tags,
437      url: DISSERTATION_RECORD.url,
438      author: DISSERTATION_RECORD.author,
439      relatedIds: []
440    };
441  }
442
443  // Load details cache if not already loaded
444  if (!detailsCache) {
445    await loadDetailsCache();
446  }
447
448  return detailsCache?.[recordId] || null;
449};
450
451/**
452 * Load the full details cache (called once when first modal opens)
453 */
454const loadDetailsCache = async () => {
455  // Prevent multiple simultaneous loads
456  if (detailsLoading) {
457    return detailsLoadPromise;
458  }
459
460  detailsLoading = true;
461  detailsLoadPromise = (async () => {
462    const dataUrl = DATA_CONFIG.archive_details;
463
464    // Check cache first
465    const cached = getCachedData(dataUrl);
466    if (cached) {
467      debug('Using cached details data');
468      detailsCache = cached.details;
469      return;
470    }
471
472    debug('Fetching details data from:', dataUrl);
473
474    try {
475      const response = await fetch(dataUrl);
476      if (!response.ok) {
477        throw new Error(`HTTP error! status: ${response.status}`);
478      }
479
480      const data = await response.json();
481      detailsCache = data.details;
482
483      // Cache the result
484      setCachedData(dataUrl, data);
485    } catch (error) {
486      console.error('Error fetching details data:', error);
487      // Keep the cache absent so a later explicit retry performs a fresh
488      // request. Propagate the failure so the record reader can distinguish a
489      // network/parse outage from a successful payload that lacks this record.
490      detailsCache = null;
491      throw error;
492    } finally {
493      detailsLoading = false;
494      detailsLoadPromise = null;
495    }
496  })();
497
498  return detailsLoadPromise;
499};
500
501/**
502 * Project an entity payload into the records list buildEntityMaps expects.
503 * Prefer an explicit `records` array; otherwise derive it from
504 * `recordEntityMap`. The `|| {}` guard keeps a payload missing
505 * recordEntityMap from throwing. Used by both the cache-hit and network
506 * branches of fetchEntitiesData so they shape the input identically.
507 * @param {{ records?: Array, recordEntityMap?: Object }} payload
508 */
509export const toRecords = (payload) =>
510  payload.records || Object.entries(payload.recordEntityMap || {}).map(([id, relatedIds]) => ({
511    id,
512    relatedIds
513  }));
514
515/**
516 * True when a parsed entities payload has the shape fetchEntitiesData and
517 * buildEntityMaps expect. Both tolerate drift silently by design —
518 * buildEntityMaps guards each field with Array.isArray and just skips it,
519 * and toRecords falls back to `{}` for a missing recordEntityMap — so a
520 * renamed or retyped field would otherwise build an empty entity index with
521 * no error instead of failing loud (#503). `entities` must always be an
522 * array. `records`, when present, must be an array too (so a malformed
523 * `records` cannot silently take precedence over a valid `recordEntityMap`
524 * and then get skipped by buildEntityMaps); when `records` is absent,
525 * `recordEntityMap` must be a real, non-array object.
526 * @param {unknown} payload
527 * @returns {boolean}
528 */
529export const isValidEntitiesPayload = (payload) => {
530  if (!payload || typeof payload !== 'object' || Array.isArray(payload)) return false;
531  if (!Array.isArray(payload.entities)) return false;
532
533  const { records, recordEntityMap } = payload;
534  if (records !== undefined && records !== null) {
535    return Array.isArray(records);
536  }
537  return (
538    recordEntityMap !== undefined &&
539    recordEntityMap !== null &&
540    typeof recordEntityMap === 'object' &&
541    !Array.isArray(recordEntityMap)
542  );
543};
544
545/**
546 * The shaped failure fetchEntitiesData returns on a fetch/parse error, or on
547 * a payload (fresh or cached) that fails isValidEntitiesPayload. One source
548 * for both call sites keeps them from drifting apart — the earlier duplicate
549 * literal is what let one copy go untested (#503).
550 * @returns {{entities: Array, recordEntityMap: Object, error: string}}
551 */
552const entityLoadFailure = () => ({
553  entities: [],
554  recordEntityMap: {},
555  error: 'The entity index could not load. Archive records remain available.',
556});
557
558/**
559 * Remove one cache entry from both Web Storage backends. getCachedData
560 * already does this inline for a TTL-expired entry; fetchEntitiesData reuses
561 * it to drop a version-matched entry whose payload shape has drifted, so a
562 * later call re-fetches instead of replaying the same bad entry (#503).
563 * @param {string} url
564 */
565const evictCachedData = (url) => {
566  const cacheKey = cacheKeyFor(url);
567  try { sessionStorage.removeItem(cacheKey); } catch {}
568  try { localStorage.removeItem(cacheKey); } catch {}
569};
570
571/**
572 * Fetch entities data (on-demand, when the entity browser opens)
573 */
574export const fetchEntitiesData = async () => {
575  // Return from cache if already loaded
576  if (entitiesCache) {
577    return entitiesCache;
578  }
579
580  // Prevent multiple simultaneous loads
581  if (entitiesLoading) {
582    return entitiesLoadPromise;
583  }
584
585  entitiesLoading = true;
586  entitiesLoadPromise = (async () => {
587    // The whole body runs under one try/catch/finally now, cache check
588    // included, so every exit path — cache hit, cache-shape-drift, network
589    // success, network failure — resets entitiesLoading exactly once. The
590    // shape-drift branch used to return early before this finally existed,
591    // which left entitiesLoading stuck true and wedged every later call
592    // behind the memoized failure (#503).
593    try {
594      const dataUrl = DATA_CONFIG.archive_entities;
595
596      // Check cache first
597      const cached = getCachedData(dataUrl);
598      if (cached) {
599        if (!isValidEntitiesPayload(cached)) {
600          // A cache entry written before this shape check shipped, or one
601          // corrupted in Web Storage, can carry the same drift a fresh fetch
602          // can. Evict it so a later call re-fetches instead of replaying
603          // the same drifted entry for the full CACHE_TTL_MS, then route
604          // this call through the identical shaped failure the network
605          // branch uses rather than building an empty entity index from it.
606          console.error('Cached entities data has an unexpected shape; treating as a load failure.');
607          evictCachedData(dataUrl);
608          return entityLoadFailure();
609        }
610        debug('Using cached entities data');
611        entitiesCache = cached;
612        buildEntityMaps({
613          entities: cached.entities,
614          records: toRecords(cached)
615        });
616        return cached;
617      }
618
619      debug('Fetching entities data from:', dataUrl);
620
621      const response = await fetch(dataUrl);
622      if (!response.ok) {
623        throw new Error(`HTTP error! status: ${response.status}`);
624      }
625
626      const data = await response.json();
627      if (!isValidEntitiesPayload(data)) {
628        throw new Error('Entity data has an unexpected shape (entities is not an array, or records/recordEntityMap is malformed)');
629      }
630      entitiesCache = data;
631
632      // Build entity maps for the entity browser
633      buildEntityMaps({
634        entities: data.entities,
635        records: toRecords(data)
636      });
637
638      // Cache the result
639      setCachedData(dataUrl, data);
640
641      return data;
642    } catch (error) {
643      console.error('Error fetching entities data:', error);
644      // Record details can still fall back to category-based relationships,
645      // so keep returning a shaped payload for that consumer. Carry the
646      // failure explicitly so EntityBrowser can distinguish an outage from a
647      // legitimate empty scope instead of presenting a silent zero-result UI.
648      return entityLoadFailure();
649    } finally {
650      entitiesLoading = false;
651    }
652  })();
653
654  return entitiesLoadPromise;
655};
656
657/**
658 * Check if entities are loaded
659 */
660export const areEntitiesLoaded = () => entitiesCache !== null;
661
662/**
663 * Preload details in background (optional optimization)
664 */
665export const preloadDetails = () => {
666  if (!detailsCache && !detailsLoading) {
667    // Background warmup is best effort. Interactive fetchRecordDetails callers
668    // still receive the rejection and render their retry state, while this
669    // unawaited optimization must not create an unhandled rejection.
670    loadDetailsCache().catch(() => {});
671  }
672};
673
674/**
675 * Fetch prebuilt analytics aggregates (~1KB).
676 *
677 * Powers the default Analytics dashboard view without loading the ~28MB source
678 * into in-browser SQLite. Throws on a failed fetch so the dashboard can render
679 * its explicit error state rather than a misleading empty view (mirrors the
680 * fetchCoreData throw contract, issue #290).
681 */
682export const fetchAnalytics = async () => {
683  const dataUrl = DATA_CONFIG.archive_analytics;
684
685  // Run the same deploy-version gate as core data. A cold #analytics deep-link
686  // skips fetchCoreData, so without this a stale cache could survive a deploy.
687  await raceTimeout(checkVersion(), VERSION_CHECK_TIMEOUT_MS);
688
689  const cached = getCachedData(dataUrl);
690  if (cached) {
691    debug('Using cached analytics data');
692    return cached;
693  }
694
695  debug('Fetching analytics data from:', dataUrl);
696
697  const response = await fetch(dataUrl);
698  if (!response.ok) {
699    throw new Error(`HTTP error! status: ${response.status}`);
700  }
701  const data = await response.json();
702  setCachedData(dataUrl, data);
703  return data;
704};
705
706/**
707 * Lazily load and construct the MiniSearch full-text indexes (#276, #669).
708 *
709 * The article and social artifacts stay separate, but both are loaded on demand
710 * alongside MiniSearch. A browse-only visit therefore downloads none of them.
711 * The loader is memoized so repeat or concurrent searches share loaded
712 * artifacts. Each successful index is cached independently; a missing sibling
713 * is retried without refetching the healthy one. The indexes are an additive
714 * recall boost, never the only search path.
715 */
716let searchIndexLoaderPromise = null;
717export const loadSearchIndex = async () => {
718  if (!searchIndexLoaderPromise) {
719    searchIndexLoaderPromise = (async () => {
720      const { default: MiniSearch } = await import('minisearch');
721      return createResilientSearchIndexLoader([
722        { url: DATA_CONFIG.search_index, options: searchIndexOptions() },
723        { url: DATA_CONFIG.social_search_index, options: socialSearchIndexOptions() },
724      ], {
725        loadJSON: (serialized, options) => loadSearchIndexArtifact(
726          serialized,
727          options,
728          MiniSearch.loadJS.bind(MiniSearch),
729        ),
730      });
731    })();
732    searchIndexLoaderPromise.catch(() => { searchIndexLoaderPromise = null; });
733  }
734
735  const loader = await searchIndexLoaderPromise;
736  const result = await loader.load();
737  for (const failure of result.failures) {
738    console.warn(`[search] full-text index unavailable (${failure.url}):`, failure.error.message);
739  }
740  return result;
741};
742
743/**
744 * Initialize SQLite database with full archive data (optional)
745 * Call this to enable SQL queries for advanced analytics
746 */
747// Memoise the in-flight init so the two query surfaces (the raw-SQL box and the
748// Query Builder) that may both trigger a first load don't each fetch and parse
749// the ~28MB source. Released on a failed/false result so a later retry can run.
750let sqliteInitPromise = null;
751export const initSqlite = async () => {
752  if (sqliteInitPromise) return sqliteInitPromise;
753
754  const pending = (async () => {
755    try {
756      // Initialize the database
757      await initDatabase();
758
759      // Load full data if we have it cached, otherwise fetch it
760      const fullDataUrl = DATA_CONFIG.archive_json;
761      let fullData = getCachedData(fullDataUrl);
762
763      if (!fullData) {
764        debug('[SQLite] Fetching full archive data for SQL database...');
765        const response = await fetch(fullDataUrl);
766        if (response.ok) {
767          fullData = await response.json();
768          setCachedData(fullDataUrl, fullData);
769        }
770      }
771
772      if (fullData) {
773        await loadSqliteData(fullData);
774        debug('[SQLite] Database ready for queries');
775        return true;
776      }
777
778      return false;
779    } catch (error) {
780      console.error('[SQLite] Failed to initialize:', error);
781      return false;
782    }
783  })();
784
785  sqliteInitPromise = pending;
786  // Keep the memo only for a successful load; drop it on false/throw so the
787  // next query attempt re-tries instead of resolving the cached failure.
788  pending.then(
789    (ok) => { if (!ok && sqliteInitPromise === pending) sqliteInitPromise = null; },
790    () => { if (sqliteInitPromise === pending) sqliteInitPromise = null; }
791  );
792  return pending;
793};
794
795/**
796 * Check if SQLite is ready for queries
797 */
798export { isSqliteReady };
799
800// Re-export SQL query functions for easy access
801export {
802  queryAsObjects,
803  getRecordCountByYear,
804  getRecordCountByCategory,
805  getRecordCountByEra,
806  getMostMentionedEntities,
807  getMostCommonConcepts,
808  getCategoryCoOccurrence,
809  sqlSearchRecords,
810  getSqliteStats
811};
812
813// ============================================
814// DATA EXPORT FUNCTIONS (Dave Winer Open Data)
815// ============================================
816
817/**
818 * Trigger browser download of data
819 */
820const downloadFile = (content, filename, mimeType) => {
821  const blob = new Blob([content], { type: mimeType });
822  const url = URL.createObjectURL(blob);
823  const link = document.createElement('a');
824  link.href = url;
825  link.download = filename;
826  document.body.appendChild(link);
827  link.click();
828  document.body.removeChild(link);
829  URL.revokeObjectURL(url);
830};
831
832/**
833 * Export records as JSON file
834 * @param {Array} records - Records to export (filtered or all)
835 * @param {string} filename - Output filename
836 */
837export const exportAsJSON = (records, filename = 'jay-rosen-archive.json') => {
838  const exportData = {
839    exported: new Date().toISOString(),
840    source: "Jay Rosen's Internet Archive",
841    url: 'https://pressthink.org/j/rosen-archive/',
842    license: 'CC BY 4.0',
843    recordCount: records.length,
844    records: records.map(r => ({
845      id: r.id,
846      title: r.title,
847      author: r.author || 'Jay Rosen',
848      date: r.date,
849      year: r.year,
850      era: r.era,
851      pub: r.pub,
852      url: r.url,
853      summary: r.summary || r.summaryPreview,
854      categories: r.categories,
855      type: r.type
856    }))
857  };
858
859  downloadFile(JSON.stringify(exportData, null, 2), filename, 'application/json');
860};
861
862/**
863 * Export records as CSV file
864 * @param {Array} records - Records to export
865 * @param {string} filename - Output filename
866 */
867export const exportAsCSV = (records, filename = 'jay-rosen-archive.csv') => {
868  const headers = ['id', 'title', 'author', 'date', 'year', 'era', 'pub', 'url', 'categories', 'type'];
869
870  const rows = records.map(r => [
871    r.id,
872    r.title,
873    r.author || 'Jay Rosen',
874    r.date,
875    r.year,
876    r.era,
877    r.pub,
878    r.url,
879    (r.categories || []).join('; '),
880    r.type || 'article'
881  ].map(escapeCsvCell).join(','));
882
883  const csv = [headers.join(','), ...rows].join('\n');
884  downloadFile(csv, filename, 'text/csv;charset=utf-8');
885};
886
887/**
888 * Get URLs for open data resources
889 */
890export const getOpenDataURLs = () => {
891  // Derive from the shared resolver so open-data links use the same canonical
892  // URL scheme as the rest of the app (#300). The old production branch
893  // hardcoded the WordPress upload root (/wp-content/rosen-archive), which only
894  // resolved via a brittle WP rewrite from the canonical /j/rosen-archive.
895  // BASE_PATH already encodes the github-pages prefix, so the non-local cases
896  // collapse to it; local keeps the relative '.' the static preview servers use.
897  const basePath = IS_LOCAL ? '.' : BASE_PATH;
898
899  return {
900    json: `${basePath}/data/archive-data.json`,
901    csv: `${basePath}/data/archive_records-public.csv`,
902    rss: `${basePath}/data/feeds/rss.xml`,
903    articlesRss: `${basePath}/data/feeds/articles.xml`,
904    opml: `${basePath}/data/feeds/archive.opml`,
905    subscriptions: `${basePath}/data/feeds/subscriptions.opml`,
906    schema: `${basePath}/data/schema.json`,
907    schemaDoc: `${basePath}/data/SCHEMA.md`
908  };
909};

Line numbers count LF bytes from the start of the resource, as the search results do. Vendor segments are library code the classifier recognised; they are stored but not indexed. Bytes are shown as Latin1 characters, one per byte.