PageSourceSearch

https://pressthink.org/j/rosen-archive/frontend/services/siteTools.js?v=3.8.36

js pressthink.org collected 2026-10-02 04:25:55 UTC 18,329 bytes, 498 lines download raw bytes

1// Read-only WebMCP tools for the public archive.
2//
3// This module keeps the browser integration at the edge. The tool handlers use
4// the same archive loaders and search helpers as the visible application. A
5// browser without document.modelContext sees no change in behavior.
6
7import {
8  fetchCoreData,
9  fetchEntitiesData,
10  fetchRecordDetails,
11  loadSearchIndex,
12} from './archiveService.js?v=3.8.36';
13import { sortRecords, RECORD_SORTS } from '../utils/recordSort.js?v=3.8.36';
14import {
15  buildSearchText,
16  matchesParsedSearchText,
17  normalizeForSearch,
18  parseSearchQuery,
19  searchLoadedIndexes,
20} from '../utils/searchNormalize.js?v=3.8.36';
21
22const DEFAULT_LIMIT = 10;
23const MAX_LIMIT = 20;
24const MAX_QUERY_LENGTH = 200;
25const MAX_FILTER_LENGTH = 200;
26const MAX_RECORD_ID_LENGTH = 80;
27const MAX_DETAIL_LENGTH = 6000;
28const MAX_QUOTE_LENGTH = 3000;
29const MAX_DETAIL_ITEMS = 50;
30const CONTENT_TYPES = ['article', 'twitter', 'bluesky'];
31const TWITTER_PUBLICATIONS = new Set(['twitter', 'twitter/x']);
32const ENTITY_TYPES = ['Person', 'Organization', 'Concept', 'Work', 'Event', 'Location'];
33
34function asInputObject(input) {
35  if (input === undefined || input === null) return {};
36  if (typeof input !== 'object' || Array.isArray(input)) {
37    throw new TypeError('Tool input must be an object');
38  }
39  return input;
40}
41
42function optionalString(value, name, maxLength = MAX_FILTER_LENGTH) {
43  if (value === undefined || value === null || value === '') return null;
44  if (typeof value !== 'string') throw new TypeError(`${name} must be a string`);
45  const trimmed = value.trim();
46  if (!trimmed) return null;
47  if (trimmed.length > maxLength) {
48    throw new RangeError(`${name} must be ${maxLength} characters or fewer`);
49  }
50  return trimmed;
51}
52
53function requiredRecordId(value) {
54  const recordId = optionalString(value, 'record_id', MAX_RECORD_ID_LENGTH);
55  if (!recordId) throw new TypeError('record_id is required');
56  if (!/^[A-Za-z0-9_.:-]+$/.test(recordId)) {
57    throw new TypeError('record_id contains unsupported characters');
58  }
59  return recordId;
60}
61
62function boundedLimit(value) {
63  if (value === undefined || value === null) return DEFAULT_LIMIT;
64  if (!Number.isInteger(value) || value < 1 || value > MAX_LIMIT) {
65    throw new RangeError(`limit must be an integer from 1 through ${MAX_LIMIT}`);
66  }
67  return value;
68}
69
70function stringArray(value, name, maxItems = 6) {
71  if (value === undefined || value === null) return [];
72  if (!Array.isArray(value)) throw new TypeError(`${name} must be an array`);
73  if (value.length > maxItems) {
74    throw new RangeError(`${name} can contain at most ${maxItems} values`);
75  }
76  const values = value.map((item, index) => {
77    const parsed = optionalString(item, `${name}[${index}]`, 120);
78    if (!parsed) throw new TypeError(`${name}[${index}] must not be empty`);
79    return parsed;
80  });
81  if (new Set(values).size !== values.length) {
82    throw new TypeError(`${name} must not contain duplicate values`);
83  }
84  return values;
85}
86
87function oneOf(value, name, allowed, fallback = null) {
88  const parsed = optionalString(value, name);
89  if (parsed === null) return fallback;
90  if (!allowed.includes(parsed)) {
91    throw new TypeError(`${name} must be one of: ${allowed.join(', ')}`);
92  }
93  return parsed;
94}
95
96function parseSearchInput(input) {
97  const source = asInputObject(input);
98  const query = optionalString(source.query, 'query', MAX_QUERY_LENGTH) || '';
99  const year = optionalString(source.year, 'year', 4);
100  const era = optionalString(source.era, 'era');
101  if (year !== null && !/^\d{4}$/.test(year)) {
102    throw new TypeError('year must contain four digits');
103  }
104  if (year && era) {
105    throw new TypeError('year and era cannot be used together');
106  }
107  return {
108    query,
109    categories: stringArray(source.categories, 'categories'),
110    era,
111    year,
112    publication: optionalString(source.publication, 'publication'),
113    content_type: oneOf(source.content_type, 'content_type', CONTENT_TYPES),
114    sort: oneOf(source.sort, 'sort', RECORD_SORTS, 'date-desc'),
115    limit: boundedLimit(source.limit),
116  };
117}
118
119function valuesMatch(left, right) {
120  return normalizeForSearch(left) === normalizeForSearch(right);
121}
122
123function matchesContentType(record, contentType) {
124  if (!contentType) return true;
125  if (contentType === 'article') return record.type !== 'social';
126  if (record.type !== 'social') return false;
127  const publication = normalizeForSearch(record.pub);
128  if (contentType === 'twitter') {
129    return TWITTER_PUBLICATIONS.has(publication);
130  }
131  return contentType === 'bluesky' && publication.includes('bluesky');
132}
133
134function projectRecord(record, baseHref) {
135  return {
136    id: record.id,
137    title: record.title || '',
138    date: record.date || '',
139    year: record.year || '',
140    era: record.era || '',
141    publication: record.pub || '',
142    type: record.type || 'article',
143    categories: Array.isArray(record.categories) ? record.categories : [],
144    verified: record.verified === true,
145    needsReview: record.needsReview === true,
146    summaryPreview: record.summaryPreview || record.summary || '',
147    archiveUrl: buildRecordLink(record.id, baseHref),
148  };
149}
150
151/**
152 * Build a stable archive link for one record without carrying unrelated state.
153 */
154export function buildRecordLink(recordId, baseHref) {
155  const id = requiredRecordId(recordId);
156  const url = new URL(baseHref);
157  url.search = '';
158  url.hash = '';
159  url.searchParams.set('record', id);
160  return url.toString();
161}
162
163/**
164 * Search core archive records with the visible archive's search semantics.
165 */
166export function searchArchiveRecords(records, input = {}, options = {}) {
167  const parsedInput = parseSearchInput(input);
168  const sourceRecords = Array.isArray(records) ? records : [];
169  const parsedQuery = parseSearchQuery(parsedInput.query);
170  const normalizedQuery = normalizeForSearch(parsedInput.query);
171  const searchIndexes = Array.isArray(options.searchIndexes) ? options.searchIndexes : [];
172  const indexMatches = parsedInput.query && searchIndexes.length > 0
173    ? new Set(searchLoadedIndexes(searchIndexes, parsedInput.query).map(hit => hit.id))
174    : null;
175
176  const matches = sourceRecords.filter((record) => {
177    if (!record || typeof record !== 'object') return false;
178
179    if (normalizedQuery) {
180      const searchText = buildSearchText(record);
181      const coreMatch = parsedQuery.phraseKeys.length > 0
182        ? matchesParsedSearchText(searchText, parsedQuery)
183        : searchText.includes(normalizedQuery);
184      if (!coreMatch && !indexMatches?.has(record.id)) return false;
185    }
186
187    if (parsedInput.categories.length > 0) {
188      const recordCategories = Array.isArray(record.categories) ? record.categories : [];
189      if (!parsedInput.categories.every(category => recordCategories.includes(category))) return false;
190    }
191    if (parsedInput.year && String(record.year || '') !== parsedInput.year) return false;
192    if (parsedInput.era && !valuesMatch(record.era, parsedInput.era)) return false;
193    if (parsedInput.publication && !valuesMatch(record.pub, parsedInput.publication)) return false;
194    if (!matchesContentType(record, parsedInput.content_type)) return false;
195    return true;
196  });
197
198  const sorted = sortRecords(matches, parsedInput.sort);
199  const limited = sorted.slice(0, parsedInput.limit);
200  const baseHref = options.baseHref || 'https://pressthink.org/j/rosen-archive/';
201
202  return {
203    filters: parsedInput,
204    totalMatches: sorted.length,
205    returned: limited.length,
206    records: limited.map(record => projectRecord(record, baseHref)),
207  };
208}
209
210/**
211 * Return the entities linked to one record from the archive entity payload.
212 */
213export function findRelatedEntities(payload, recordId, input = {}) {
214  const id = requiredRecordId(recordId);
215  const source = asInputObject(input);
216  const entityType = oneOf(source.entity_type, 'entity_type', ENTITY_TYPES);
217  const limit = boundedLimit(source.limit);
218  const entities = Array.isArray(payload?.entities) ? payload.entities : [];
219  const relatedIds = Array.isArray(payload?.recordEntityMap?.[id])
220    ? payload.recordEntityMap[id]
221    : [];
222  const relatedSet = new Set(relatedIds.map(String));
223
224  const matches = entities
225    .filter(entity => relatedSet.has(String(entity?.id)))
226    .filter(entity => !entityType || entity.type === entityType)
227    .sort((left, right) => (
228      (Number(right.prominence) || 0) - (Number(left.prominence) || 0)
229      || String(left.name || '').localeCompare(String(right.name || ''))
230    ));
231
232  return {
233    recordId: id,
234    entityType,
235    totalMatches: matches.length,
236    returned: Math.min(matches.length, limit),
237    entities: matches.slice(0, limit).map(entity => ({
238      id: entity.id,
239      name: entity.name || '',
240      type: entity.type || '',
241      role: entity.role || '',
242      affiliation: entity.affiliation || '',
243      prominence: Number(entity.prominence) || 0,
244      totalMentions: Number(entity.totalMentions) || 0,
245    })),
246  };
247}
248
249function validateFacetSelections(input, facets = {}) {
250  const checks = [
251    ['categories', input.categories, facets.categories || []],
252    ['era', input.era ? [input.era] : [], facets.eras || []],
253    ['publication', input.publication ? [input.publication] : [], facets.publications || []],
254  ];
255  for (const [name, selected, allowed] of checks) {
256    const unknown = selected.filter(value => !allowed.includes(value));
257    if (unknown.length > 0) {
258      throw new TypeError(`Unknown ${name} value: ${unknown.join(', ')}`);
259    }
260  }
261}
262
263function collectPublications(data = {}) {
264  const records = Array.isArray(data?.records) ? data.records : [];
265  const configured = Array.isArray(data?.facets?.publications)
266    ? data.facets.publications
267    : [];
268  return [...new Set([
269    ...configured,
270    ...records.map(record => record?.pub),
271  ].filter(value => typeof value === 'string' && value.trim()).map(value => value.trim()))]
272    .sort((left, right) => left.localeCompare(right));
273}
274
275function clipText(value, maxLength) {
276  const text = typeof value === 'string' ? value : '';
277  return text.length <= maxLength ? text : `${text.slice(0, maxLength - 1)}…`;
278}
279
280function memoizeLoader(loader) {
281  let pending = null;
282  return () => {
283    if (!pending) {
284      pending = Promise.resolve().then(loader);
285      pending.catch(() => { pending = null; });
286    }
287    return pending;
288  };
289}
290
291const emptyInputSchema = {
292  type: 'object',
293  properties: {},
294  additionalProperties: false,
295};
296
297/**
298 * Create tool descriptors. Dependencies are injectable for deterministic tests.
299 */
300export function createArchiveSiteTools(dependencies = {}) {
301  const loadCoreData = dependencies.loadCoreData || fetchCoreData;
302  const loadRecordDetails = dependencies.loadRecordDetails || fetchRecordDetails;
303  const loadEntitiesData = dependencies.loadEntitiesData || fetchEntitiesData;
304  const loadFullTextIndexes = dependencies.loadFullTextIndexes || loadSearchIndex;
305  const currentHref = dependencies.currentHref || (() => globalThis.location?.href
306    || 'https://pressthink.org/j/rosen-archive/');
307  const getCoreData = memoizeLoader(loadCoreData);
308  const getEntitiesData = memoizeLoader(async () => {
309    const entityData = await loadEntitiesData();
310    if (entityData?.error) throw new Error(entityData.error);
311    return entityData;
312  });
313
314  return [
315    {
316      name: 'get_archive_facets',
317      description: 'List the categories, eras, publications, and content types accepted by archive search.',
318      inputSchema: emptyInputSchema,
319      annotations: { readOnlyHint: true },
320      execute: async () => {
321        const data = await getCoreData();
322        return {
323          recordCount: Array.isArray(data?.records) ? data.records.length : 0,
324          categories: data?.facets?.categories || [],
325          eras: data?.facets?.eras || [],
326          publications: collectPublications(data),
327          contentTypes: CONTENT_TYPES,
328        };
329      },
330    },
331    {
332      name: 'search_archive',
333      description: 'Search public archive records without changing the current page or archive data.',
334      inputSchema: {
335        type: 'object',
336        properties: {
337          query: { type: 'string', maxLength: MAX_QUERY_LENGTH },
338          categories: {
339            type: 'array',
340            maxItems: 6,
341            uniqueItems: true,
342            items: { type: 'string', minLength: 1, maxLength: 120 },
343          },
344          era: {
345            type: 'string',
346            minLength: 1,
347            maxLength: MAX_FILTER_LENGTH,
348            description: 'An exact value from get_archive_facets. Do not use with year.',
349          },
350          year: {
351            type: 'string',
352            pattern: '^\\d{4}$',
353            description: 'A four-digit year. Do not use with era.',
354          },
355          publication: { type: 'string', minLength: 1, maxLength: MAX_FILTER_LENGTH },
356          content_type: { type: 'string', enum: CONTENT_TYPES },
357          sort: { type: 'string', enum: RECORD_SORTS },
358          limit: { type: 'integer', minimum: 1, maximum: MAX_LIMIT, default: DEFAULT_LIMIT },
359        },
360        additionalProperties: false,
361      },
362      annotations: { readOnlyHint: true },
363      execute: async (input) => {
364        const parsedInput = parseSearchInput(input);
365        const data = await getCoreData();
366        validateFacetSelections(parsedInput, {
367          ...(data?.facets || {}),
368          publications: collectPublications(data),
369        });
370
371        let indexes = [];
372        let searchCoverage = parsedInput.query ? 'core fields only' : 'not applicable';
373        let searchWarning = null;
374        if (parsedInput.query) {
375          try {
376            const result = await loadFullTextIndexes();
377            indexes = result?.indexes || [];
378            searchCoverage = result?.complete ? 'complete full text' : 'partial full text';
379            if (Array.isArray(result?.failures) && result.failures.length > 0) {
380              searchWarning = 'One full-text index was unavailable. Results include every available index and core record fields.';
381            }
382          } catch {
383            searchWarning = 'Full-text indexes were unavailable. Results cover titles, summary previews, and categories.';
384          }
385        }
386
387        return {
388          ...searchArchiveRecords(data?.records, parsedInput, {
389            baseHref: currentHref(),
390            searchIndexes: indexes,
391          }),
392          searchCoverage,
393          ...(searchWarning ? { warning: searchWarning } : {}),
394        };
395      },
396    },
397    {
398      name: 'get_archive_record',
399      description: 'Read one public archive record by its record identifier without changing the page.',
400      inputSchema: {
401        type: 'object',
402        properties: {
403          record_id: {
404            type: 'string',
405            minLength: 1,
406            maxLength: MAX_RECORD_ID_LENGTH,
407            pattern: '^[A-Za-z0-9_.:-]+$',
408          },
409        },
410        required: ['record_id'],
411        additionalProperties: false,
412      },
413      annotations: { readOnlyHint: true },
414      execute: async (input) => {
415        const recordId = requiredRecordId(asInputObject(input).record_id);
416        const data = await getCoreData();
417        const coreRecord = Array.isArray(data?.records)
418          ? data.records.find(record => record.id === recordId)
419          : null;
420        if (!coreRecord) return { found: false, recordId };
421
422        const details = await loadRecordDetails(recordId) || {};
423        const summary = details.summary || coreRecord.summaryPreview || '';
424        const quote = details.quote || '';
425        const relatedEntityIds = Array.isArray(details.relatedIds) ? details.relatedIds : [];
426        const truncatedFields = [];
427        if (summary.length > MAX_DETAIL_LENGTH) truncatedFields.push('summary');
428        if (quote.length > MAX_QUOTE_LENGTH) truncatedFields.push('quote');
429        if (relatedEntityIds.length > MAX_DETAIL_ITEMS) truncatedFields.push('relatedEntityIds');
430
431        return {
432          found: true,
433          record: {
434            ...projectRecord(coreRecord, currentHref()),
435            author: details.author || 'Jay Rosen',
436            summary: clipText(summary, MAX_DETAIL_LENGTH),
437            quote: clipText(quote, MAX_QUOTE_LENGTH),
438            concepts: Array.isArray(details.concepts)
439              ? details.concepts.slice(0, MAX_DETAIL_ITEMS)
440              : [],
441            tags: Array.isArray(details.tags) ? details.tags.slice(0, MAX_DETAIL_ITEMS) : [],
442            sourceUrl: details.url || '',
443            relatedEntityIds: relatedEntityIds.slice(0, MAX_DETAIL_ITEMS),
444            truncatedFields,
445          },
446        };
447      },
448    },
449    {
450      name: 'find_related_entities',
451      description: 'Read the people, organizations, concepts, works, events, or locations linked to one archive record.',
452      inputSchema: {
453        type: 'object',
454        properties: {
455          record_id: {
456            type: 'string',
457            minLength: 1,
458            maxLength: MAX_RECORD_ID_LENGTH,
459            pattern: '^[A-Za-z0-9_.:-]+$',
460          },
461          entity_type: { type: 'string', enum: ENTITY_TYPES },
462          limit: { type: 'integer', minimum: 1, maximum: MAX_LIMIT, default: DEFAULT_LIMIT },
463        },
464        required: ['record_id'],
465        additionalProperties: false,
466      },
467      annotations: { readOnlyHint: true },
468      execute: async (input) => {
469        const source = asInputObject(input);
470        const recordId = requiredRecordId(source.record_id);
471        const data = await getCoreData();
472        const recordExists = Array.isArray(data?.records)
473          && data.records.some(record => record.id === recordId);
474        if (!recordExists) return { found: false, recordId };
475
476        const entityData = await getEntitiesData();
477        return {
478          found: true,
479          ...findRelatedEntities(entityData, recordId, source),
480        };
481      },
482    },
483  ];
484}
485
486/**
487 * Register tools only when the current browser implements WebMCP.
488 */
489export async function registerArchiveSiteTools(options = {}) {
490  const { documentObject = globalThis.document, ...dependencies } = options;
491  const modelContext = documentObject?.modelContext;
492  if (typeof modelContext?.registerTool !== 'function') return false;
493
494  for (const tool of createArchiveSiteTools(dependencies)) {
495    await modelContext.registerTool(tool);
496  }
497  return true;
498}

Line numbers count LF bytes from the start of the resource, as the search results do. Vendor segments are library code the classifier recognised; they are stored but not indexed. Bytes are shown as Latin1 characters, one per byte.