1// Read-only WebMCP tools for the public archive. 2// 3// This module keeps the browser integration at the edge. The tool handlers use 4// the same archive loaders and search helpers as the visible application. A 5// browser without document.modelContext sees no change in behavior. 6 7import { 8 fetchCoreData, 9 fetchEntitiesData, 10 fetchRecordDetails, 11 loadSearchIndex, 12} from './archiveService.js?v=3.8.36'; 13import { sortRecords, RECORD_SORTS } from '../utils/recordSort.js?v=3.8.36'; 14import { 15 buildSearchText, 16 matchesParsedSearchText, 17 normalizeForSearch, 18 parseSearchQuery, 19 searchLoadedIndexes, 20} from '../utils/searchNormalize.js?v=3.8.36'; 21 22const DEFAULT_LIMIT = 10; 23const MAX_LIMIT = 20; 24const MAX_QUERY_LENGTH = 200; 25const MAX_FILTER_LENGTH = 200; 26const MAX_RECORD_ID_LENGTH = 80; 27const MAX_DETAIL_LENGTH = 6000; 28const MAX_QUOTE_LENGTH = 3000; 29const MAX_DETAIL_ITEMS = 50; 30const CONTENT_TYPES = ['article', 'twitter', 'bluesky']; 31const TWITTER_PUBLICATIONS = new Set(['twitter', 'twitter/x']); 32const ENTITY_TYPES = ['Person', 'Organization', 'Concept', 'Work', 'Event', 'Location']; 33 34function asInputObject(input) { 35 if (input === undefined || input === null) return {}; 36 if (typeof input !== 'object' || Array.isArray(input)) { 37 throw new TypeError('Tool input must be an object'); 38 } 39 return input; 40} 41 42function optionalString(value, name, maxLength = MAX_FILTER_LENGTH) { 43 if (value === undefined || value === null || value === '') return null; 44 if (typeof value !== 'string') throw new TypeError(`${name} must be a string`); 45 const trimmed = value.trim(); 46 if (!trimmed) return null; 47 if (trimmed.length > maxLength) { 48 throw new RangeError(`${name} must be ${maxLength} characters or fewer`); 49 } 50 return trimmed; 51} 52 53function requiredRecordId(value) { 54 const recordId = optionalString(value, 'record_id', MAX_RECORD_ID_LENGTH); 55 if (!recordId) throw new TypeError('record_id is required'); 56 if (!/^[A-Za-z0-9_.:-]+$/.test(recordId)) { 57 throw new TypeError('record_id contains unsupported characters'); 58 } 59 return recordId; 60} 61 62function boundedLimit(value) { 63 if (value === undefined || value === null) return DEFAULT_LIMIT; 64 if (!Number.isInteger(value) || value < 1 || value > MAX_LIMIT) { 65 throw new RangeError(`limit must be an integer from 1 through ${MAX_LIMIT}`); 66 } 67 return value; 68} 69 70function stringArray(value, name, maxItems = 6) { 71 if (value === undefined || value === null) return []; 72 if (!Array.isArray(value)) throw new TypeError(`${name} must be an array`); 73 if (value.length > maxItems) { 74 throw new RangeError(`${name} can contain at most ${maxItems} values`); 75 } 76 const values = value.map((item, index) => { 77 const parsed = optionalString(item, `${name}[${index}]`, 120); 78 if (!parsed) throw new TypeError(`${name}[${index}] must not be empty`); 79 return parsed; 80 }); 81 if (new Set(values).size !== values.length) { 82 throw new TypeError(`${name} must not contain duplicate values`); 83 } 84 return values; 85} 86 87function oneOf(value, name, allowed, fallback = null) { 88 const parsed = optionalString(value, name); 89 if (parsed === null) return fallback; 90 if (!allowed.includes(parsed)) { 91 throw new TypeError(`${name} must be one of: ${allowed.join(', ')}`); 92 } 93 return parsed; 94} 95 96function parseSearchInput(input) { 97 const source = asInputObject(input); 98 const query = optionalString(source.query, 'query', MAX_QUERY_LENGTH) || ''; 99 const year = optionalString(source.year, 'year', 4); 100 const era = optionalString(source.era, 'era'); 101 if (year !== null && !/^\d{4}$/.test(year)) { 102 throw new TypeError('year must contain four digits'); 103 } 104 if (year && era) { 105 throw new TypeError('year and era cannot be used together'); 106 } 107 return { 108 query, 109 categories: stringArray(source.categories, 'categories'), 110 era, 111 year, 112 publication: optionalString(source.publication, 'publication'), 113 content_type: oneOf(source.content_type, 'content_type', CONTENT_TYPES), 114 sort: oneOf(source.sort, 'sort', RECORD_SORTS, 'date-desc'),
115 limit: boundedLimit(source.limit), 116 }; 117} 118 119function valuesMatch(left, right) { 120 return normalizeForSearch(left) === normalizeForSearch(right); 121} 122 123function matchesContentType(record, contentType) { 124 if (!contentType) return true; 125 if (contentType === 'article') return record.type !== 'social'; 126 if (record.type !== 'social') return false; 127 const publication = normalizeForSearch(record.pub); 128 if (contentType === 'twitter') { 129 return TWITTER_PUBLICATIONS.has(publication); 130 } 131 return contentType === 'bluesky' && publication.includes('bluesky'); 132} 133 134function projectRecord(record, baseHref) { 135 return { 136 id: record.id, 137 title: record.title || '', 138 date: record.date || '', 139 year: record.year || '', 140 era: record.era || '', 141 publication: record.pub || '', 142 type: record.type || 'article', 143 categories: Array.isArray(record.categories) ? record.categories : [], 144 verified: record.verified === true, 145 needsReview: record.needsReview === true, 146 summaryPreview: record.summaryPreview || record.summary || '', 147 archiveUrl: buildRecordLink(record.id, baseHref), 148 }; 149} 150 151/** 152 * Build a stable archive link for one record without carrying unrelated state. 153 */ 154export function buildRecordLink(recordId, baseHref) { 155 const id = requiredRecordId(recordId); 156 const url = new URL(baseHref); 157 url.search = ''; 158 url.hash = ''; 159 url.searchParams.set('record', id); 160 return url.toString(); 161} 162 163/** 164 * Search core archive records with the visible archive's search semantics. 165 */ 166export function searchArchiveRecords(records, input = {}, options = {}) { 167 const parsedInput = parseSearchInput(input); 168 const sourceRecords = Array.isArray(records) ? records : []; 169 const parsedQuery = parseSearchQuery(parsedInput.query); 170 const normalizedQuery = normalizeForSearch(parsedInput.query); 171 const searchIndexes = Array.isArray(options.searchIndexes) ? options.searchIndexes : []; 172 const indexMatches = parsedInput.query && searchIndexes.length > 0 173 ? new Set(searchLoadedIndexes(searchIndexes, parsedInput.query).map(hit => hit.id)) 174 : null; 175 176 const matches = sourceRecords.filter((record) => { 177 if (!record || typeof record !== 'object') return false; 178 179 if (normalizedQuery) { 180 const searchText = buildSearchText(record); 181 const coreMatch = parsedQuery.phraseKeys.length > 0 182 ? matchesParsedSearchText(searchText, parsedQuery) 183 : searchText.includes(normalizedQuery); 184 if (!coreMatch && !indexMatches?.has(record.id)) return false; 185 } 186 187 if (parsedInput.categories.length > 0) { 188 const recordCategories = Array.isArray(record.categories) ? record.categories : []; 189 if (!parsedInput.categories.every(category => recordCategories.includes(category))) return false; 190 } 191 if (parsedInput.year && String(record.year || '') !== parsedInput.year) return false; 192 if (parsedInput.era && !valuesMatch(record.era, parsedInput.era)) return false; 193 if (parsedInput.publication && !valuesMatch(record.pub, parsedInput.publication)) return false; 194 if (!matchesContentType(record, parsedInput.content_type)) return false; 195 return true; 196 }); 197 198 const sorted = sortRecords(matches, parsedInput.sort); 199 const limited = sorted.slice(0, parsedInput.limit); 200 const baseHref = options.baseHref || 'https://pressthink.org/j/rosen-archive/'; 201 202 return { 203 filters: parsedInput, 204 totalMatches: sorted.length, 205 returned: limited.length, 206 records: limited.map(record => projectRecord(record, baseHref)), 207 }; 208} 209 210/** 211 * Return the entities linked to one record from the archive entity payload. 212 */ 213export function findRelatedEntities(payload, recordId, input = {}) { 214 const id = requiredRecordId(recordId); 215 const source = asInputObject(input); 216 const entityType = oneOf(source.entity_type, 'entity_type', ENTITY_TYPES); 217 const limit = boundedLimit(source.limit); 218 const entities = Array.isArray(payload?.entities) ? payload.entities : []; 219 const relatedIds = Array.isArray(payload?.recordEntityMap?.[id]) 220 ? payload.recordEntityMap[id] 221 : []; 222 const relatedSet = new Set(relatedIds.map(String)); 223 224 const matches = entities 225 .filter(entity => relatedSet.has(String(entity?.id))) 226 .filter(entity => !entityType || entity.type === entityType) 227 .sort((left, right) => ( 228 (Number(right.prominence) || 0) - (Number(left.prominence) || 0) 229 || String(left.name || '').localeCompare(String(right.name || '')) 230 )); 231 232 return { 233 recordId: id, 234 entityType, 235 totalMatches: matches.length, 236 returned: Math.min(matches.length, limit), 237 entities: matches.slice(0, limit).map(entity => ({ 238 id: entity.id, 239 name: entity.name || '', 240 type: entity.type || '', 241 role: entity.role || '', 242 affiliation: entity.affiliation || '', 243 prominence: Number(entity.prominence) || 0, 244 totalMentions: Number(entity.totalMentions) || 0, 245 })), 246 }; 247} 248 249function validateFacetSelections(input, facets = {}) { 250 const checks = [ 251 ['categories', input.categories, facets.categories || []], 252 ['era', input.era ? [input.era] : [], facets.eras || []], 253 ['publication', input.publication ? [input.publication] : [], facets.publications || []], 254 ]; 255 for (const [name, selected, allowed] of checks) { 256 const unknown = selected.filter(value => !allowed.includes(value)); 257 if (unknown.length > 0) { 258 throw new TypeError(`Unknown ${name} value: ${unknown.join(', ')}`); 259 } 260 } 261} 262 263function collectPublications(data = {}) { 264 const records = Array.isArray(data?.records) ? data.records : []; 265 const configured = Array.isArray(data?.facets?.publications) 266 ? data.facets.publications 267 : []; 268 return [...new Set([ 269 ...configured, 270 ...records.map(record => record?.pub), 271 ].filter(value => typeof value === 'string' && value.trim()).map(value => value.trim()))] 272 .sort((left, right) => left.localeCompare(right)); 273} 274 275function clipText(value, maxLength) { 276 const text = typeof value === 'string' ? value : ''; 277 return text.length <= maxLength ? text : `${text.slice(0, maxLength - 1)}â¦`; 278} 279 280function memoizeLoader(loader) { 281 let pending = null; 282 return () => { 283 if (!pending) { 284 pending = Promise.resolve().then(loader); 285 pending.catch(() => { pending = null; }); 286 } 287 return pending; 288 }; 289} 290 291const emptyInputSchema = { 292 type: 'object', 293 properties: {}, 294 additionalProperties: false, 295}; 296 297/** 298 * Create tool descriptors. Dependencies are injectable for deterministic tests. 299 */ 300export function createArchiveSiteTools(dependencies = {}) {
301 const loadCoreData = dependencies.loadCoreData || fetchCoreData; 302 const loadRecordDetails = dependencies.loadRecordDetails || fetchRecordDetails; 303 const loadEntitiesData = dependencies.loadEntitiesData || fetchEntitiesData; 304 const loadFullTextIndexes = dependencies.loadFullTextIndexes || loadSearchIndex; 305 const currentHref = dependencies.currentHref || (() => globalThis.location?.href 306 || 'https://pressthink.org/j/rosen-archive/'); 307 const getCoreData = memoizeLoader(loadCoreData); 308 const getEntitiesData = memoizeLoader(async () => { 309 const entityData = await loadEntitiesData(); 310 if (entityData?.error) throw new Error(entityData.error); 311 return entityData; 312 }); 313 314 return [ 315 { 316 name: 'get_archive_facets', 317 description: 'List the categories, eras, publications, and content types accepted by archive search.', 318 inputSchema: emptyInputSchema, 319 annotations: { readOnlyHint: true }, 320 execute: async () => { 321 const data = await getCoreData(); 322 return { 323 recordCount: Array.isArray(data?.records) ? data.records.length : 0, 324 categories: data?.facets?.categories || [], 325 eras: data?.facets?.eras || [], 326 publications: collectPublications(data), 327 contentTypes: CONTENT_TYPES, 328 }; 329 }, 330 }, 331 { 332 name: 'search_archive', 333 description: 'Search public archive records without changing the current page or archive data.', 334 inputSchema: { 335 type: 'object', 336 properties: { 337 query: { type: 'string', maxLength: MAX_QUERY_LENGTH }, 338 categories: { 339 type: 'array', 340 maxItems: 6, 341 uniqueItems: true, 342 items: { type: 'string', minLength: 1, maxLength: 120 }, 343 }, 344 era: { 345 type: 'string', 346 minLength: 1, 347 maxLength: MAX_FILTER_LENGTH, 348 description: 'An exact value from get_archive_facets. Do not use with year.', 349 }, 350 year: { 351 type: 'string', 352 pattern: '^\\d{4}$', 353 description: 'A four-digit year. Do not use with era.', 354 }, 355 publication: { type: 'string', minLength: 1, maxLength: MAX_FILTER_LENGTH }, 356 content_type: { type: 'string', enum: CONTENT_TYPES }, 357 sort: { type: 'string', enum: RECORD_SORTS }, 358 limit: { type: 'integer', minimum: 1, maximum: MAX_LIMIT, default: DEFAULT_LIMIT }, 359 }, 360 additionalProperties: false, 361 }, 362 annotations: { readOnlyHint: true }, 363 execute: async (input) => { 364 const parsedInput = parseSearchInput(input); 365 const data = await getCoreData(); 366 validateFacetSelections(parsedInput, { 367 ...(data?.facets || {}), 368 publications: collectPublications(data), 369 }); 370 371 let indexes = []; 372 let searchCoverage = parsedInput.query ? 'core fields only' : 'not applicable'; 373 let searchWarning = null; 374 if (parsedInput.query) { 375 try { 376 const result = await loadFullTextIndexes(); 377 indexes = result?.indexes || []; 378 searchCoverage = result?.complete ? 'complete full text' : 'partial full text'; 379 if (Array.isArray(result?.failures) && result.failures.length > 0) { 380 searchWarning = 'One full-text index was unavailable. Results include every available index and core record fields.'; 381 } 382 } catch { 383 searchWarning = 'Full-text indexes were unavailable. Results cover titles, summary previews, and categories.'; 384 } 385 } 386 387 return { 388 ...searchArchiveRecords(data?.records, parsedInput, { 389 baseHref: currentHref(), 390 searchIndexes: indexes, 391 }), 392 searchCoverage, 393 ...(searchWarning ? { warning: searchWarning } : {}), 394 }; 395 }, 396 }, 397 { 398 name: 'get_archive_record', 399 description: 'Read one public archive record by its record identifier without changing the page.', 400 inputSchema: { 401 type: 'object', 402 properties: { 403 record_id: { 404 type: 'string', 405 minLength: 1, 406 maxLength: MAX_RECORD_ID_LENGTH, 407 pattern: '^[A-Za-z0-9_.:-]+$', 408 }, 409 }, 410 required: ['record_id'], 411 additionalProperties: false, 412 }, 413 annotations: { readOnlyHint: true }, 414 execute: async (input) => { 415 const recordId = requiredRecordId(asInputObject(input).record_id); 416 const data = await getCoreData(); 417 const coreRecord = Array.isArray(data?.records) 418 ? data.records.find(record => record.id === recordId) 419 : null; 420 if (!coreRecord) return { found: false, recordId }; 421
422 const details = await loadRecordDetails(recordId) || {}; 423 const summary = details.summary || coreRecord.summaryPreview || ''; 424 const quote = details.quote || ''; 425 const relatedEntityIds = Array.isArray(details.relatedIds) ? details.relatedIds : []; 426 const truncatedFields = []; 427 if (summary.length > MAX_DETAIL_LENGTH) truncatedFields.push('summary'); 428 if (quote.length > MAX_QUOTE_LENGTH) truncatedFields.push('quote'); 429 if (relatedEntityIds.length > MAX_DETAIL_ITEMS) truncatedFields.push('relatedEntityIds'); 430 431 return { 432 found: true, 433 record: { 434 ...projectRecord(coreRecord, currentHref()), 435 author: details.author || 'Jay Rosen', 436 summary: clipText(summary, MAX_DETAIL_LENGTH), 437 quote: clipText(quote, MAX_QUOTE_LENGTH), 438 concepts: Array.isArray(details.concepts) 439 ? details.concepts.slice(0, MAX_DETAIL_ITEMS) 440 : [], 441 tags: Array.isArray(details.tags) ? details.tags.slice(0, MAX_DETAIL_ITEMS) : [], 442 sourceUrl: details.url || '', 443 relatedEntityIds: relatedEntityIds.slice(0, MAX_DETAIL_ITEMS), 444 truncatedFields, 445 }, 446 }; 447 }, 448 }, 449 { 450 name: 'find_related_entities', 451 description: 'Read the people, organizations, concepts, works, events, or locations linked to one archive record.', 452 inputSchema: { 453 type: 'object', 454 properties: { 455 record_id: { 456 type: 'string', 457 minLength: 1, 458 maxLength: MAX_RECORD_ID_LENGTH, 459 pattern: '^[A-Za-z0-9_.:-]+$', 460 }, 461 entity_type: { type: 'string', enum: ENTITY_TYPES }, 462 limit: { type: 'integer', minimum: 1, maximum: MAX_LIMIT, default: DEFAULT_LIMIT }, 463 }, 464 required: ['record_id'], 465 additionalProperties: false, 466 }, 467 annotations: { readOnlyHint: true }, 468 execute: async (input) => { 469 const source = asInputObject(input); 470 const recordId = requiredRecordId(source.record_id); 471 const data = await getCoreData(); 472 const recordExists = Array.isArray(data?.records) 473 && data.records.some(record => record.id === recordId); 474 if (!recordExists) return { found: false, recordId }; 475 476 const entityData = await getEntitiesData(); 477 return { 478 found: true, 479 ...findRelatedEntities(entityData, recordId, source), 480 }; 481 }, 482 }, 483 ]; 484} 485 486/** 487 * Register tools only when the current browser implements WebMCP. 488 */ 489export async function registerArchiveSiteTools(options = {}) {
490 const { documentObject = globalThis.document, ...dependencies } = options; 491 const modelContext = documentObject?.modelContext; 492 if (typeof modelContext?.registerTool !== 'function') return false; 493 494 for (const tool of createArchiveSiteTools(dependencies)) { 495 await modelContext.registerTool(tool); 496 } 497 return true; 498}
Line numbers count LF bytes from the start of the resource, as the search results do. Vendor segments are library code the classifier recognised; they are stored but not indexed. Bytes are shown as Latin1 characters, one per byte.