1 2import { DATA_CONFIG } from '../constants.js?v=3.8.36'; 3import { 4 initDatabase, 5 loadArchiveData as loadSqliteData, 6 isReady as isSqliteReady, 7 queryAsObjects, 8 getRecordCountByYear, 9 getRecordCountByCategory, 10 getRecordCountByEra, 11 getMostMentionedEntities, 12 getMostCommonConcepts, 13 getCategoryCoOccurrence, 14 searchRecords as sqlSearchRecords, 15 getStats as getSqliteStats 16} from './sqliteService.js?v=3.8.36'; 17import { IS_LOCAL, BASE_PATH } from '../utils/pathResolver.js?v=3.8.36'; 18import { searchIndexOptions, socialSearchIndexOptions } from '../utils/searchConfig.js?v=3.8.36'; 19import { escapeCsvCell } from '../utils/csvSafety.js?v=3.8.36'; 20import { idbGet, idbSet, idbClear } from './idbCache.js?v=3.8.36'; 21import { CACHE_VERSION, CACHE_TTL_MS, MAX_LOCALSTORAGE_SIZE, cacheKeyFor } from './cacheConfig.js?v=3.8.36'; 22import { raceTimeout } from '../utils/raceTimeout.js?v=3.8.36'; 23import { createResilientSearchIndexLoader, loadSearchIndexArtifact } from './searchIndexLoader.js?v=3.8.36'; 24import { loadReleaseMetadata } from './releaseMetadata.js?v=3.8.36'; 25 26// Routine cache-hit / fetch-start logs are silent in production. Set 27// `localStorage.jrda_debug = '1'` in DevTools and reload to opt in (#170). 28// Wrapped in try/catch because localStorage access can throw SecurityError 29// in privacy modes or when storage is blocked by browser policy; a throw 30// here would prevent this module (and therefore the whole app) from loading. 31const DEBUG = (() => { 32 try { 33 return typeof localStorage !== 'undefined' && localStorage.getItem('jrda_debug') === '1'; 34 } catch { 35 return false; 36 } 37})(); 38const debug = DEBUG ? console.log.bind(console) : () => {}; 39 40// Simple hash function for UI color selection (djb1 variant) 41// Used by App.js to deterministically assign colors to categories 42export const hashString = (str) => { 43 let hash = 0; 44 for (let i = 0; i < str.length; i++) hash = (hash << 5) - hash + str.charCodeAt(i); 45 return Math.abs(hash); 46}; 47 48// ============================================ 49// LAZY LOADING STATE 50// ============================================ 51 52// Cache for loaded details (populated on demand) 53let detailsCache = null; 54let detailsLoading = false; 55let detailsLoadPromise = null; 56 57// Cache for loaded entities (populated on demand) 58let entitiesCache = null; 59let entitiesLoading = false; 60let entitiesLoadPromise = null; 61 62// Entity lookup maps - populated when data is loaded 63let entityById = new Map(); // entity_id -> entity object 64let entityToRecords = new Map(); // entity_id -> Set of record ids 65let recordToEntities = new Map(); // record_id -> Set of entity ids 66 67/** 68 * Build entity lookup maps from loaded archive data 69 * Called after archive data is fetched 70 */ 71export const buildEntityMaps = (data) => { 72 entityById.clear(); 73 entityToRecords.clear(); 74 recordToEntities.clear(); 75 76 // Build entity ID -> entity object map 77 if (data.entities && Array.isArray(data.entities)) { 78 data.entities.forEach(entity => { 79 entityById.set(entity.id, entity); 80 }); 81 } 82 83 // Build bidirectional record <-> entity maps 84 if (data.records && Array.isArray(data.records)) { 85 data.records.forEach(record => { 86 const entityIds = record.relatedIds || []; 87 recordToEntities.set(record.id, new Set(entityIds)); 88 89 entityIds.forEach(entityId => { 90 if (!entityToRecords.has(entityId)) { 91 entityToRecords.set(entityId, new Set()); 92 } 93 entityToRecords.get(entityId).add(record.id); 94 }); 95 }); 96 } 97 98 console.log(`Built entity maps: ${entityById.size} entities, ${recordToEntities.size} records`); 99}; 100 101/** 102 * Get entity by ID 103 */ 104export const getEntityById = (entityId) => entityById.get(entityId); 105 106/** 107 * Get all records that mention a specific entity 108 */ 109export const getRecordsByEntity = (entityId) => { 110 const recordIds = entityToRecords.get(entityId); 111 return recordIds ? Array.from(recordIds) : []; 112}; 113 114/** 115 * Get all entities mentioned in a record 116 */ 117export const getEntitiesByRecord = (recordId) => { 118 const entityIds = recordToEntities.get(recordId); 119 if (!entityIds) return []; 120 return Array.from(entityIds).map(id =>
120 entityById.get(id)).filter(Boolean); 121}; 122 123/** 124 * Find shared entities between two records 125 * @param {string} recordId1 - First record ID 126 * @param {string} recordId2 - Second record ID 127 * @param {string|null} entityTypeFilter - Optional entity type filter (Person, Organization, Concept, etc.) 128 * @returns {Array} Array of shared entity objects with prominence scores 129 */ 130export const findSharedEntities = (recordId1, recordId2, entityTypeFilter = null) => { 131 const entities1 = recordToEntities.get(recordId1); 132 const entities2 = recordToEntities.get(recordId2); 133 134 if (!entities1 || !entities2) return []; 135 136 const shared = []; 137 entities1.forEach(entityId => { 138 if (entities2.has(entityId)) { 139 const entity = entityById.get(entityId); 140 if (entity) { 141 // Apply type filter if specified 142 if (!entityTypeFilter || entity.type === entityTypeFilter) { 143 shared.push(entity); 144 } 145 } 146 } 147 }); 148 149 // Sort by prominence (higher first) 150 return shared.sort((a, b) => (b.prominence || 0) - (a.prominence || 0)); 151}; 152 153/** 154 * Calculate connection strength between two records based on shared entities 155 * @param {string} recordId1 - First record ID 156 * @param {string} recordId2 - Second record ID 157 * @param {string|null} entityTypeFilter - Optional entity type filter 158 * @returns {Object} { strength: number, sharedEntities: Array } 159 */ 160export const calculateEntityConnectionStrength = (recordId1, recordId2, entityTypeFilter = null) => { 161 const sharedEntities = findSharedEntities(recordId1, recordId2, entityTypeFilter); 162 163 if (sharedEntities.length === 0) { 164 return { strength: 0, sharedEntities: [], prominenceScore: 0 }; 165 } 166 167 // Calculate weighted strength based on prominence scores 168 const prominenceScore = sharedEntities.reduce((sum, e) => sum + (e.prominence || 1), 0); 169 170 return { 171 strength: sharedEntities.length, 172 sharedEntities, 173 prominenceScore 174 }; 175}; 176 177const DISSERTATION_RECORD = { 178 id: 'dissertation-1986', 179 title: 'The Impossible Press: American Journalism and the Decline of Public Life', 180 author: 'Jay Rosen', 181 date: '1986-01-01', 182 year: '1986', 183 era: 'Public Journalism (90s)', 184 pub: 'New York University (Ph.D. Dissertation)', 185 url: IS_LOCAL ? './dissertation/reader/' : `${BASE_PATH}/dissertation/reader/`, 186 summary: 'Rosen\'s doctoral dissertation traces the history of the idea that the function of the press is to inform the public. It argues that the rise of the mass circulation newspaper, while creating a technical ability to reach everyone, actually undermined the conditions necessary for a "universal town meeting." Drawing heavily on Walter Lippmann and John Dewey, it suggests that the professionalization of journalism ("objectivity") was a retreat from the problem of creating a genuine public life in a complex society. It contrasts news as "symptom" vs. news as "symbol" and explores how the press creates a "pseudo-environment" of public opinion.', 187 quote: 'An impossible press was born, one which sought to solve the whole problem of public life simply by controlling the conduct of journalists.', 188 categories: ['Journalism Theory & Practice', 'Politics & Democracy', 'Press & Media Criticism', 'Audience & Public Engagement'], 189 concepts: ['Public Sphere', 'Omnicompetent Citizen', 'Objectivity', 'Mass Society', 'Professionalism', 'Communication vs Community', 'Democracy and Distance'], 190 tags: ['Walter Lippmann', 'John Dewey', 'James Gordon Bennett', 'Joseph Pulitzer', 'Penny Press', 'Yellow Journalism', 'Robert Park', 'Tocqueville'], 191 verified: true, 192 type: 'Dissertation' 193}; 194 195// Cache configuration (CACHE_VERSION / CACHE_TTL_MS / MAX_LOCALSTORAGE_SIZE 196// and the cacheKeyFor hash) lives in cacheConfig.js rather than inline. 197 198/** 199 * Check version.json on the server. If the version has changed, 200 * clear all caches so users get fresh data after deploys. 201 */ 202// Memoise the in-flight Promise so concurrent callers all await the same 203// fetch instead of racing past a sync boolean (#171). Released on settle 204// so a hung or failed check doesn't poison every future call in the session. 205let versionCheckPromise = null; 206const checkVersion = () => { 207 if (versionCheckPromise) return versionCheckPromise; 208 const pending = (async () => { 209 try { 210 const { version } = await loadReleaseMetadata(); 211 const stored = localStorage.getItem('jrda_deploy_version'); 212 if (stored && stored !== version) { 213 console.log('[Cache] Deploy version changed, clearing caches'); 214 clearArchiveCache(); 215 } 216 localStorage.setItem('jrda_deploy_version', version); 217 } catch { /* version.json not available, skip */ } 218 })(); 219 versionCheckPromise = pending; 220 pending.finally(() => {
221 if (versionCheckPromise === pending) versionCheckPromise = null; 222 }); 223 return pending; 224}; 225 226// Bound the wait at the call site so a slow or hung version.json can't 227// stall a load a good cache could satisfy. The check stays in flight in 228// the background and clears the cache if it eventually returns. 229const VERSION_CHECK_TIMEOUT_MS = 4000; 230 231const getCachedData = (url) => { 232 try { 233 const cacheKey = cacheKeyFor(url); 234 // localStorage is the only writer now (#337); still read a legacy 235 // sessionStorage entry first so caches from older builds expire cleanly. 236 let cached = sessionStorage.getItem(cacheKey) || localStorage.getItem(cacheKey); 237 if (!cached) return null; 238 239 const entry = JSON.parse(cached); 240 const now = Date.now(); 241 242 if (entry.version !== CACHE_VERSION || (now - entry.timestamp) > CACHE_TTL_MS) { 243 try { sessionStorage.removeItem(cacheKey); } catch {} 244 try { localStorage.removeItem(cacheKey); } catch {} 245 return null; 246 } 247 248 return entry.data; 249 } catch (e) { 250 console.warn('Cache read error:', e); 251 return null; 252 } 253}; 254 255const setCachedData = (url, data) => { 256 const cacheKey = cacheKeyFor(url); 257 const entry = { 258 data, 259 timestamp: Date.now(), 260 version: CACHE_VERSION 261 }; 262 263 try { 264 const serialized = JSON.stringify(entry); 265 266 // Web Storage gives localStorage and sessionStorage a ~5 MB quota each, so a 267 // payload too large for localStorage will not fit sessionStorage either. The 268 // old overflow-to-sessionStorage attempt therefore threw QuotaExceededError on 269 // every load for the large data dumps (archive-core ~13 MB, archive-details 270 // ~13 MB, archive-data ~30 MB) and cached nothing. Skip it: archive-core also 271 // has an IndexedDB cache (fetchCoreData, far larger quota), and all of these 272 // payloads are served from the service-worker Cache Storage on refetch 273 // (sw.js stale-while-revalidate), so Web Storage is redundant for them. (#337) 274 // These files are not in Web Storage and only archive-core is in IndexedDB, 275 // so when the service worker is unavailable they refetch on every 276 // navigation. That residual gap is surfaced once at startup (the 277 // SW-registration block in index.html), not cached here (#428). 278 if (serialized.length > MAX_LOCALSTORAGE_SIZE) { 279 return; 280 } 281 282 // Small data goes to localStorage (persists across sessions) 283 localStorage.setItem(cacheKey, serialized); 284 } catch (e) { 285 if (e.name === 'QuotaExceededError' || e.code === 22) { 286 console.log('Cache storage full, clearing old archive caches...'); 287 try { 288 clearArchiveCache(); 289 localStorage.setItem(cacheKey, JSON.stringify(entry)); 290 } catch { 291 console.warn('Cache disabled: browser storage is full.'); 292 } 293 } else { 294 console.warn('Cache write error:', e); 295 } 296 } 297}; 298 299// Clear all archive caches from both localStorage and sessionStorage 300export const clearArchiveCache = () => { 301 try { 302 for (const storage of [localStorage, sessionStorage]) { 303 const keys = Object.keys(storage); 304 const archiveKeys = keys.filter(key => key.startsWith('archive_json_') || key.startsWith('archive_csv_')); 305 archiveKeys.forEach(key => storage.removeItem(key)); 306 } 307 console.log('Archive caches cleared'); 308 } catch (e) { 309 console.warn('Error clearing cache:', e); 310 } 311 // Also drop the IndexedDB core-data store. Fire-and-forget: idbClear never 312 // rejects (it resolves false on failure), and the version-namespaced key 313 // (coreIdbKey) already prevents a stale read, so callers needn't await this. 314 idbClear(); 315}; 316 317// IndexedDB cache key for the core payload. Namespaced by the manual 318// CACHE_VERSION knob and the deploy version (version.json, stored by 319// checkVersion), so a deploy or a cache-version bump addresses a fresh key â 320// a stale blob is never read, by construction, without relying on a racy 321// async clear. clearArchiveCache additionally wipes the store to bound growth. 322const readDeployVersion = () => { 323 try { 324 return localStorage.getItem('jrda_deploy_version') || 'novers'; 325 } catch { 326 return 'novers'; 327 } 328}; 329const coreIdbKey = () => `archive-core::${CACHE_VERSION}::${readDeployVersion()}`; 330const makeCoreEntry = (data) => ({ data, timestamp: Date.now(), version: CACHE_VERSION }); 331const isFreshEntry = (entry) => 332 !!entry && 333 entry.version === CACHE_VERSION && 334 typeof entry.timestamp === 'number' && 335 (Date.now() - entry.timestamp) <= CACHE_TTL_MS; 336 337/**
338 * Fetch core archive data (lightweight, for initial page load) 339 * This is the new optimized entry point - loads ~8MB instead of ~25MB 340 */ 341export const fetchCoreData = async () => { 342 const dataUrl = DATA_CONFIG.archive_core; 343 344 // Check deploy version (clears caches if version changed), bounded so a 345 // slow version.json doesn't stall a load a good cache could satisfy. 346 await raceTimeout(checkVersion(), VERSION_CHECK_TIMEOUT_MS); 347 348 const idbKey = coreIdbKey(); 349 350 // 1. IndexedDB cache â structured-clones the object on read, skipping the 351 // JSON.parse of the ~13 MB blob, and persists across tab close (the blob 352 // exceeds localStorage's ~5 MB cap, so the Web Storage fallback below lands 353 // in sessionStorage, which a fresh tab never sees). See idbCache.js (#275). 354 const idbEntry = await idbGet(idbKey); 355 if (isFreshEntry(idbEntry)) { 356 debug('Using IndexedDB-cached core data'); 357 return idbEntry.data; 358 } 359 360 // 2. Web Storage cache â covers browsers where IndexedDB is blocked (Safari 361 // Private, Firefox strict tracking protection). Promote a hit into 362 // IndexedDB, best-effort and unawaited, so the next read skips the parse. 363 const cached = getCachedData(dataUrl); 364 if (cached) { 365 debug('Using Web Storage-cached core data'); 366 idbSet(idbKey, makeCoreEntry(cached)); 367 return cached; 368 } 369 370 debug('Fetching core data from:', dataUrl); 371 372 try { 373 const response = await fetch(dataUrl); 374 if (!response.ok) { 375 throw new Error(`HTTP error! status: ${response.status}`); 376 } 377 378 const data = await response.json(); 379 380 // Inject dissertation record if not present 381 if (!data.records.find(r => r.id === 'dissertation-1986')) { 382 data.records.push({ 383 id: DISSERTATION_RECORD.id, 384 title: DISSERTATION_RECORD.title, 385 date: DISSERTATION_RECORD.date, 386 year: DISSERTATION_RECORD.year, 387 era: DISSERTATION_RECORD.era, 388 pub: DISSERTATION_RECORD.pub, 389 categories: DISSERTATION_RECORD.categories, 390 type: DISSERTATION_RECORD.type, 391 verified: DISSERTATION_RECORD.verified, 392 summaryPreview: DISSERTATION_RECORD.summary.substring(0, 180) + '...' 393 }); 394 395 // Also add dissertation facets if missing 396 DISSERTATION_RECORD.categories.forEach(c => { 397 if (!data.facets.categories.includes(c)) { 398 data.facets.categories.push(c); 399 } 400 }); 401 data.facets.categories.sort(); 402 } 403 404 // Cache the result. IndexedDB first; fall back to Web Storage only when 405 // IndexedDB is unavailable, so capable browsers skip the redundant ~13 MB 406 // sessionStorage write (and its JSON.stringify) entirely. 407 const storedInIdb = await idbSet(idbKey, makeCoreEntry(data)); 408 if (!storedInIdb) { 409 setCachedData(dataUrl, data); 410 } 411 412 return data; 413 } catch (error) { 414 // Propagate the failure. Returning a DISSERTATION_RECORD-only fallback 415 // here would mask a real outage (archive-core.json 404/503/parse error, 416 // partial deploy) as a successful load of a 1-record archive â visitors 417 // and monitors could not tell it apart from the real one. App.js's 418 // .catch renders the explicit "Unable to load archive" error state 419 // instead. 420 console.error('Error fetching core data:', error); 421 throw error; 422 } 423}; 424 425/** 426 * Fetch record details (on-demand, when modal opens) 427 * Returns full summary, quote, concepts, tags, url, author, relatedIds 428 */ 429export const fetchRecordDetails = async (recordId) => { 430 // Return dissertation details from constant 431 if (recordId === 'dissertation-1986') { 432 return { 433 summary: DISSERTATION_RECORD.summary, 434 quote: DISSERTATION_RECORD.quote, 435 concepts: DISSERTATION_RECORD.concepts, 436 tags: DISSERTATION_RECORD.tags, 437 url: DISSERTATION_RECORD.url, 438 author: DISSERTATION_RECORD.author, 439 relatedIds: [] 440 }; 441 } 442 443 // Load details cache if not already loaded 444 if (!detailsCache) { 445 await loadDetailsCache(); 446 } 447 448 return detailsCache?.[recordId] || null; 449}; 450 451/** 452 * Load the full details cache (called once when first modal opens) 453 */ 454const loadDetailsCache = async () => { 455 // Prevent multiple simultaneous loads 456 if (detailsLoading) { 457 return detailsLoadPromise; 458 } 459 460 detailsLoading = true; 461 detailsLoadPromise = (async () => { 462 const dataUrl = DATA_CONFIG.archive_details; 463 464 // Check cache first 465 const cached = getCachedData(dataUrl); 466 if (cached) { 467 debug('Using cached details data'); 468 detailsCache = cached.details; 469 return; 470 } 471 472 debug('Fetching details data from:', dataUrl); 473 474 try { 475 const response = await fetch(dataUrl); 476 if (!response.ok) { 477 throw new Error(`HTTP error! status: ${response.status}`); 478 } 479 480 const data = await response.json(); 481 detailsCache = data.details; 482 483 // Cache the result 484 setCachedData(dataUrl, data); 485 } catch (error) { 486 console.error('Error fetching details data:', error); 487 // Keep the cache absent so a later explicit retry performs a fresh 488 // request. Propagate the failure so the record reader can distinguish a 489 // network/parse outage from a successful payload that lacks this record. 490 detailsCache = null; 491 throw error; 492 } finally { 493 detailsLoading = false;
494 detailsLoadPromise = null; 495 } 496 })(); 497 498 return detailsLoadPromise; 499}; 500 501/** 502 * Project an entity payload into the records list buildEntityMaps expects. 503 * Prefer an explicit `records` array; otherwise derive it from 504 * `recordEntityMap`. The `|| {}` guard keeps a payload missing 505 * recordEntityMap from throwing. Used by both the cache-hit and network 506 * branches of fetchEntitiesData so they shape the input identically. 507 * @param {{ records?: Array, recordEntityMap?: Object }} payload 508 */ 509export const toRecords = (payload) => 510 payload.records || Object.entries(payload.recordEntityMap || {}).map(([id, relatedIds]) => ({ 511 id, 512 relatedIds 513 })); 514 515/** 516 * True when a parsed entities payload has the shape fetchEntitiesData and 517 * buildEntityMaps expect. Both tolerate drift silently by design â 518 * buildEntityMaps guards each field with Array.isArray and just skips it, 519 * and toRecords falls back to `{}` for a missing recordEntityMap â so a 520 * renamed or retyped field would otherwise build an empty entity index with 521 * no error instead of failing loud (#503). `entities` must always be an 522 * array. `records`, when present, must be an array too (so a malformed 523 * `records` cannot silently take precedence over a valid `recordEntityMap` 524 * and then get skipped by buildEntityMaps); when `records` is absent, 525 * `recordEntityMap` must be a real, non-array object. 526 * @param {unknown} payload 527 * @returns {boolean} 528 */ 529export const isValidEntitiesPayload = (payload) => { 530 if (!payload || typeof payload !== 'object' || Array.isArray(payload)) return false; 531 if (!Array.isArray(payload.entities)) return false; 532 533 const { records, recordEntityMap } = payload; 534 if (records !== undefined && records !== null) { 535 return Array.isArray(records); 536 } 537 return ( 538 recordEntityMap !== undefined && 539 recordEntityMap !== null && 540 typeof recordEntityMap === 'object' && 541 !Array.isArray(recordEntityMap) 542 ); 543}; 544 545/** 546 * The shaped failure fetchEntitiesData returns on a fetch/parse error, or on 547 * a payload (fresh or cached) that fails isValidEntitiesPayload. One source 548 * for both call sites keeps them from drifting apart â the earlier duplicate 549 * literal is what let one copy go untested (#503). 550 * @returns {{entities: Array, recordEntityMap: Object, error: string}} 551 */ 552const entityLoadFailure = () => ({ 553 entities: [], 554 recordEntityMap: {}, 555 error: 'The entity index could not load. Archive records remain available.', 556}); 557 558/** 559 * Remove one cache entry from both Web Storage backends. getCachedData 560 * already does this inline for a TTL-expired entry; fetchEntitiesData reuses 561 * it to drop a version-matched entry whose payload shape has drifted, so a 562 * later call re-fetches instead of replaying the same bad entry (#503). 563 * @param {string} url 564 */ 565const evictCachedData = (url) => { 566 const cacheKey = cacheKeyFor(url); 567 try { sessionStorage.removeItem(cacheKey); } catch {} 568 try { localStorage.removeItem(cacheKey); } catch {} 569}; 570 571/** 572 * Fetch entities data (on-demand, when the entity browser opens) 573 */ 574export const fetchEntitiesData = async () => { 575 // Return from cache if already loaded 576 if (entitiesCache) { 577 return entitiesCache; 578 } 579 580 // Prevent multiple simultaneous loads 581 if (entitiesLoading) { 582 return entitiesLoadPromise; 583 } 584 585 entitiesLoading = true; 586 entitiesLoadPromise = (async () => { 587 // The whole body runs under one try/catch/finally now, cache check 588 // included, so every exit path â cache hit, cache-shape-drift, network 589 // success, network failure â resets entitiesLoading exactly once. The 590 // shape-drift branch used to return early before this finally existed, 591 // which left entitiesLoading stuck true and wedged every later call 592 // behind the memoized failure (#503). 593 try { 594 const dataUrl = DATA_CONFIG.archive_entities; 595 596 // Check cache first 597 const cached = getCachedData(dataUrl); 598 if (cached) { 599 if (!isValidEntitiesPayload(cached)) { 600 // A cache entry written before this shape check shipped, or one 601 // corrupted in Web Storage, can carry the same drift a fresh fetch 602 // can. Evict it so a later call re-fetches instead of replaying 603 // the same drifted entry for the full CACHE_TTL_MS, then route 604 // this call through the identical shaped failure the network 605 // branch uses rather than building an empty entity index from it. 606 console.error('Cached entities data has an unexpected shape; treating as a load failure.'); 607 evictCachedData(dataUrl); 608 return entityLoadFailure(); 609 } 610 debug('Using cached entities data'); 611 entitiesCache = cached; 612 buildEntityMaps({ 613 entities: cached.entities, 614 records: toRecords(cached) 615 }); 616 return cached; 617 } 618 619 debug('Fetching entities data from:', dataUrl); 620 621 const response = await fetch(dataUrl); 622 if (!response.ok) { 623 throw new Error(`HTTP error! status: ${response.status}`); 624 } 625 626 const data = await response.json(); 627 if (!isValidEntitiesPayload(data)) { 628 throw new Error('Entity data has an unexpected shape (entities is not an array, or records/recordEntityMap is malformed)'); 629 } 630 entitiesCache = data; 631 632 // Build entity maps for the entity browser 633 buildEntityMaps({ 634 entities: data.entities, 635 records: toRecords(data) 636 }); 637 638 // Cache the result 639 setCachedData(dataUrl, data); 640 641 return data; 642 } catch (error) { 643 console.error('Error fetching entities data:', error); 644 // Record details can still fall back to category-based relationships, 645 // so keep returning a shaped payload for that consumer. Carry the 646 // failure explicitly so EntityBrowser can distinguish an outage from a 647 // legitimate empty scope instead of presenting a silent zero-result UI. 648 return entityLoadFailure(); 649 } finally { 650 entitiesLoading = false;
651 } 652 })(); 653 654 return entitiesLoadPromise; 655}; 656 657/** 658 * Check if entities are loaded 659 */ 660export const areEntitiesLoaded = () => entitiesCache !== null; 661 662/** 663 * Preload details in background (optional optimization) 664 */ 665export const preloadDetails = () => { 666 if (!detailsCache && !detailsLoading) { 667 // Background warmup is best effort. Interactive fetchRecordDetails callers 668 // still receive the rejection and render their retry state, while this 669 // unawaited optimization must not create an unhandled rejection. 670 loadDetailsCache().catch(() => {}); 671 } 672}; 673 674/** 675 * Fetch prebuilt analytics aggregates (~1KB). 676 * 677 * Powers the default Analytics dashboard view without loading the ~28MB source 678 * into in-browser SQLite. Throws on a failed fetch so the dashboard can render 679 * its explicit error state rather than a misleading empty view (mirrors the 680 * fetchCoreData throw contract, issue #290). 681 */ 682export const fetchAnalytics = async () => { 683 const dataUrl = DATA_CONFIG.archive_analytics; 684 685 // Run the same deploy-version gate as core data. A cold #analytics deep-link 686 // skips fetchCoreData, so without this a stale cache could survive a deploy. 687 await raceTimeout(checkVersion(), VERSION_CHECK_TIMEOUT_MS); 688 689 const cached = getCachedData(dataUrl); 690 if (cached) { 691 debug('Using cached analytics data'); 692 return cached; 693 } 694 695 debug('Fetching analytics data from:', dataUrl); 696 697 const response = await fetch(dataUrl); 698 if (!response.ok) { 699 throw new Error(`HTTP error! status: ${response.status}`); 700 } 701 const data = await response.json(); 702 setCachedData(dataUrl, data); 703 return data; 704}; 705 706/** 707 * Lazily load and construct the MiniSearch full-text indexes (#276, #669). 708 * 709 * The article and social artifacts stay separate, but both are loaded on demand 710 * alongside MiniSearch. A browse-only visit therefore downloads none of them. 711 * The loader is memoized so repeat or concurrent searches share loaded 712 * artifacts. Each successful index is cached independently; a missing sibling 713 * is retried without refetching the healthy one. The indexes are an additive 714 * recall boost, never the only search path. 715 */ 716let searchIndexLoaderPromise = null; 717export const loadSearchIndex = async () => { 718 if (!searchIndexLoaderPromise) { 719 searchIndexLoaderPromise = (async () => { 720 const { default: MiniSearch } = await import('minisearch'); 721 return createResilientSearchIndexLoader([ 722 { url: DATA_CONFIG.search_index, options: searchIndexOptions() }, 723 { url: DATA_CONFIG.social_search_index, options: socialSearchIndexOptions() }, 724 ], { 725 loadJSON: (serialized, options) => loadSearchIndexArtifact( 726 serialized, 727 options, 728 MiniSearch.loadJS.bind(MiniSearch), 729 ), 730 }); 731 })(); 732 searchIndexLoaderPromise.catch(() => { searchIndexLoaderPromise = null; }); 733 } 734 735 const loader = await searchIndexLoaderPromise; 736 const result = await loader.load(); 737 for (const failure of result.failures) { 738 console.warn(`[search] full-text index unavailable (${failure.url}):`, failure.error.message); 739 } 740 return result; 741}; 742 743/** 744 * Initialize SQLite database with full archive data (optional) 745 * Call this to enable SQL queries for advanced analytics 746 */ 747// Memoise the in-flight init so the two query surfaces (the raw-SQL box and the 748// Query Builder) that may both trigger a first load don't each fetch and parse 749// the ~28MB source. Released on a failed/false result so a later retry can run. 750let sqliteInitPromise = null; 751export const initSqlite = async () => { 752 if (sqliteInitPromise) return sqliteInitPromise; 753 754 const pending = (async () => { 755 try { 756 // Initialize the database 757 await initDatabase(); 758 759 // Load full data if we have it cached, otherwise fetch it 760 const fullDataUrl = DATA_CONFIG.archive_json; 761 let fullData = getCachedData(fullDataUrl); 762 763 if (!fullData) { 764 debug('[SQLite] Fetching full archive data for SQL database...'); 765 const response = await fetch(fullDataUrl); 766 if (response.ok) { 767 fullData = await response.json(); 768 setCachedData(fullDataUrl, fullData); 769 } 770 } 771 772 if (fullData) { 773 await loadSqliteData(fullData); 774 debug('[SQLite] Database ready for queries'); 775 return true; 776 } 777 778 return false;
779 } catch (error) { 780 console.error('[SQLite] Failed to initialize:', error); 781 return false; 782 } 783 })(); 784 785 sqliteInitPromise = pending; 786 // Keep the memo only for a successful load; drop it on false/throw so the 787 // next query attempt re-tries instead of resolving the cached failure. 788 pending.then( 789 (ok) => { if (!ok && sqliteInitPromise === pending) sqliteInitPromise = null; }, 790 () => { if (sqliteInitPromise === pending) sqliteInitPromise = null; } 791 ); 792 return pending; 793}; 794 795/** 796 * Check if SQLite is ready for queries 797 */ 798export { isSqliteReady }; 799 800// Re-export SQL query functions for easy access 801export { 802 queryAsObjects, 803 getRecordCountByYear, 804 getRecordCountByCategory, 805 getRecordCountByEra, 806 getMostMentionedEntities, 807 getMostCommonConcepts, 808 getCategoryCoOccurrence, 809 sqlSearchRecords, 810 getSqliteStats 811}; 812 813// ============================================ 814// DATA EXPORT FUNCTIONS (Dave Winer Open Data) 815// ============================================ 816 817/** 818 * Trigger browser download of data 819 */ 820const downloadFile = (content, filename, mimeType) => { 821 const blob = new Blob([content], { type: mimeType }); 822 const url = URL.createObjectURL(blob); 823 const link = document.createElement('a'); 824 link.href = url; 825 link.download = filename; 826 document.body.appendChild(link); 827 link.click(); 828 document.body.removeChild(link); 829 URL.revokeObjectURL(url); 830}; 831 832/** 833 * Export records as JSON file 834 * @param {Array} records - Records to export (filtered or all) 835 * @param {string} filename - Output filename 836 */ 837export const exportAsJSON = (records, filename = 'jay-rosen-archive.json') => { 838 const exportData = { 839 exported: new Date().toISOString(), 840 source: "Jay Rosen's Internet Archive", 841 url: 'https://pressthink.org/j/rosen-archive/', 842 license: 'CC BY 4.0', 843 recordCount: records.length, 844 records: records.map(r => ({ 845 id: r.id, 846 title: r.title, 847 author: r.author || 'Jay Rosen', 848 date: r.date, 849 year: r.year, 850 era: r.era, 851 pub: r.pub, 852 url: r.url, 853 summary: r.summary || r.summaryPreview, 854 categories: r.categories, 855 type: r.type 856 })) 857 }; 858 859 downloadFile(JSON.stringify(exportData, null, 2), filename, 'application/json'); 860}; 861 862/** 863 * Export records as CSV file 864 * @param {Array} records - Records to export 865 * @param {string} filename - Output filename 866 */ 867export const exportAsCSV = (records, filename = 'jay-rosen-archive.csv') => { 868 const headers = ['id', 'title', 'author', 'date', 'year', 'era', 'pub', 'url', 'categories', 'type']; 869 870 const rows = records.map(r => [ 871 r.id, 872 r.title, 873 r.author || 'Jay Rosen', 874 r.date,
875 r.year, 876 r.era, 877 r.pub, 878 r.url, 879 (r.categories || []).join('; '), 880 r.type || 'article' 881 ].map(escapeCsvCell).join(',')); 882 883 const csv = [headers.join(','), ...rows].join('\n'); 884 downloadFile(csv, filename, 'text/csv;charset=utf-8'); 885}; 886 887/** 888 * Get URLs for open data resources 889 */ 890export const getOpenDataURLs = () => { 891 // Derive from the shared resolver so open-data links use the same canonical 892 // URL scheme as the rest of the app (#300). The old production branch 893 // hardcoded the WordPress upload root (/wp-content/rosen-archive), which only 894 // resolved via a brittle WP rewrite from the canonical /j/rosen-archive. 895 // BASE_PATH already encodes the github-pages prefix, so the non-local cases 896 // collapse to it; local keeps the relative '.' the static preview servers use. 897 const basePath = IS_LOCAL ? '.' : BASE_PATH; 898 899 return { 900 json: `${basePath}/data/archive-data.json`, 901 csv: `${basePath}/data/archive_records-public.csv`, 902 rss: `${basePath}/data/feeds/rss.xml`, 903 articlesRss: `${basePath}/data/feeds/articles.xml`, 904 opml: `${basePath}/data/feeds/archive.opml`, 905 subscriptions: `${basePath}/data/feeds/subscriptions.opml`, 906 schema: `${basePath}/data/schema.json`, 907 schemaDoc: `${basePath}/data/SCHEMA.md` 908 }; 909};
Line numbers count LF bytes from the start of the resource, as the search results do. Vendor segments are library code the classifier recognised; they are stored but not indexed. Bytes are shown as Latin1 characters, one per byte.