PageSourceSearch

https://scribeocr.com/scribe.js/js/recognizeConvert.js

js scribeocr.com collected 2026-09-25 20:05:11 UTC 63,401 bytes, 1,393 lines download raw bytes

1import { opt } from './containers/app.js';
2import { scribeDocDefaults } from './containers/scribeDocDefaults.js';
3import { loadBuiltInFontsRaw, loadChiSimFont } from './fontContainerMain.js';
4import { calcCharMetricsFromPages } from './fontStatistics.js';
5import { gs } from './generalWorkerMain.js';
6import { ImageWrapper } from './objects/imageObjects.js';
7import { LayoutDataTablePage, LayoutPage, addCircularRefsDataTables } from './objects/layoutObjects.js';
8import ocr, { addCircularRefsOcr, OcrPage } from './objects/ocrObjects.js';
9import { PageMetrics } from './objects/pageMetricsObjects.js';
10import { selectOcrPages } from './pdf/ocrPageSelection.js';
11import { clearObjectProperties } from './utils/miscUtils.js';
12
13/** @typedef {import('./containers/scribeDoc.js').ScribeDoc} ScribeDoc */
14
15// Data-derived thresholds for the post-OCR keep/discard gate (ocrAddsNewText, below).
16const OCR_NEW_CONF_MIN = 85; // min word confidence (0-100) to count an OCR word as legitimate
17const OCR_NEW_LINE_WORDS = 3; // new word-like words in one OCR line => a coherent new-text line
18const OCR_NEW_LINES_MIN = 2; // coherent new-text lines that keep the OCR
19const OCR_NEW_NUMS_MIN = 10; // new multi-digit numbers that keep the OCR (a baked data table)
20const OCR_NEW_CHARS_MIN = 100; // new word-like characters that keep the OCR
21
22/**
23 * Decide whether a page's pure OCR adds legitimate text its native layer lacks (the keep/discard gate).
24 * An OCR word counts as "new" when its normalized token is not a substring of the native token stream.
25 * The check is substring membership, not equality, because native and OCR segment words differently.
26 * @param {OcrPage|null|undefined} nativePage - The native (PDF) text layer.
27 *   A null layer (a scan or broken-encoding page) makes all OCR new, so the OCR is kept.
28 * @param {OcrPage} ocrPage - The page's pure OCR result.
29 * @returns {boolean} True to keep the OCR, false to discard it in favor of the native text.
30 */
31export function ocrAddsNewText(nativePage, ocrPage) {
32  if (!nativePage) return true;
33  const normTok = (/** @type {string} */ text) => ocr.replaceLigatures(text)
34    .toLowerCase()
35    .normalize('NFKD')
36    .replace(/[̀-ͯ]/g, '')
37    .replace(/[^a-z0-9]/g, '');
38  const nativeStream = nativePage.lines.flatMap((l) => l.words).map((w) => normTok(w.text)).filter(Boolean).join(' ');
39  let newChars = 0;
40  let newNums = 0;
41  let newTextLines = 0;
42  for (const line of ocrPage.lines) {
43    let lineNewWords = 0;
44    for (const word of line.words) {
45      const tok = normTok(word.text);
46      if (tok.length < 2 || word.conf < OCR_NEW_CONF_MIN || nativeStream.includes(tok)) continue;
47      if (/^[a-z]{3,}$/.test(tok) && /[aeiouy]/.test(tok)) {
48        newChars += tok.length;
49        lineNewWords += 1;
50      } else if (/^[0-9]{2,}$/.test(tok)) {
51        newNums += 1;
52      }
53    }
54    if (lineNewWords >= OCR_NEW_LINE_WORDS) newTextLines += 1;
55  }
56  return newTextLines >= OCR_NEW_LINES_MIN || newNums >= OCR_NEW_NUMS_MIN || newChars >= OCR_NEW_CHARS_MIN;
57}
58
59/**
60 * Build the canonical 'Combined' layer for a partial page selection and point `active` at it.
61 * Per page it keeps the engine's OCR, or falls back to native (PDF) text when the keep/discard gate finds the OCR adds nothing the native layer lacks.
62 * Every page is cloned, so editing 'Combined' cannot corrupt the source layers it draws from.
63 * A no-op when OCR ran on every page, when the document carries user-uploaded OCR, or when no page was OCR'd.
64 * @param {ScribeDoc} doc
65 * @param {OcrPage[]} source - The engine's full-document OCR layer ('Tesseract Combined' or a custom model's).
66 * @param {boolean[]} ocrPageMask - Which pages were sent to OCR.
67 * @param {boolean} gateApplies - Whether the keep/discard gate runs (the `auto*` ocrPages modes only).
68 * @param {boolean} fullOcr - True when every page was OCR'd, in which case `active` already names the engine layer.
69 */
70function buildCombinedLayer(doc, source, ocrPageMask, gateApplies, fullOcr) {
71  if (fullOcr || doc.ocr['User Upload'] || !ocrPageMask.some(Boolean)) return;
72  const native = doc.ocr.pdf;
73  // Relocate the pure Legacy+LSTM combine from 'Combined' to 'Tesseract Combined' (unless an existing-OCR
74  // run already put it there) so 'Combined' can hold the canonical result.
75  if (source === doc.ocr.Combined && !doc.ocr['Tesseract Combined']) doc.ocr['Tesseract Combined'] = source;
76  const combined = Array(doc.inputData.pageCount);
77  for (let i = 0;
77 i < combined.length; i++) {
78    const nat = native && native[i];
79    const ocrPage = source[i];
80    let chosen;
81    if (ocrPageMask[i] && ocrPage) {
82      chosen = (gateApplies && nat && !ocrAddsNewText(nat, ocrPage)) ? nat : ocrPage;
83    } else {
84      chosen = nat || ocrPage;
85    }
86    if (chosen) {
87      combined[i] = ocr.clonePage(chosen);
88      combined[i].angle = chosen.angle;
89    } else {
90      combined[i] = new OcrPage(i, doc.pageMetrics[i].dims);
91    }
92  }
93  doc.ocr.Combined = combined;
94  doc.ocr.active = doc.ocr.Combined;
95}
96
97/**
98 * Display warning/error message to user if missing character-level data.
99 *
100 * @param {ScribeDoc} doc
101 * @param {Array<Object.<string, string>>} warnArr - Array of objects containing warning/error messages from convertPage
102 */
103export function checkCharWarn(doc, warnArr) {
104  // TODO: Figure out what happens if there is one blank page with no identified characters (as that would presumably trigger an error and/or warning on the page level).
105  // Make sure the program still works in that case for both Tesseract and Abbyy.
106
107  const charErrorCt = warnArr.filter((x) => x?.char === 'char_error').length;
108  const charWarnCt = warnArr.filter((x) => x?.char === 'char_warning').length;
109  const charGoodCt = warnArr.length - charErrorCt - charWarnCt;
110
111  // The UI warning/error messages cannot be thrown within this function,
112  // as that would make this file break when imported into contexts that do not have the main UI.
113  if (charGoodCt === 0 && charErrorCt > 0) {
114    if (typeof process === 'undefined') {
115      const errorHTML = `No character-level OCR data detected. Abbyy XML is only supported with character-level data.
116        <a href="https://docs.scribeocr.com/faq.html#is-character-level-ocr-data-required--why" target="_blank" class="alert-link">Learn more.</a>`;
117      doc.errorHandler({ message: errorHTML });
118    } else {
119      const errorText = `No character-level OCR data detected. Abbyy XML is only supported with character-level data.
120        See: https://docs.scribeocr.com/faq.html#is-character-level-ocr-data-required--why`;
121      doc.errorHandler({ message: errorText });
122    }
123  } if (charGoodCt === 0 && charWarnCt > 0 && typeof process === 'undefined') {
124    const warningHTML = `No character-level OCR data detected. Font optimization features will be disabled.
125      <a href="https://docs.scribeocr.com/faq.html#is-character-level-ocr-data-required--why" target="_blank" class="alert-link">Learn more.</a>`;
126    doc.warningHandler({ message: warningHTML });
127  }
128}
129
130/**
131 * Sum up evaluation statistics for all pages.
132 * @param {Array<EvalMetrics>} evalStatsArr
133 */
134export const calcEvalStatsDoc = (evalStatsArr) => {
135  const evalStatsDoc = {
136    total: 0,
137    correct: 0,
138    incorrect: 0,
139    missed: 0,
140    extra: 0,
141    correctLowConf: 0,
142    incorrectHighConf: 0,
143  };
144
145  for (let i = 0; i < evalStatsArr.length; i++) {
146    evalStatsDoc.total += evalStatsArr[i].total;
147    evalStatsDoc.correct += evalStatsArr[i].correct;
148    evalStatsDoc.incorrect += evalStatsArr[i].incorrect;
149    evalStatsDoc.missed += evalStatsArr[i].missed;
150    evalStatsDoc.extra += evalStatsArr[i].extra;
151    evalStatsDoc.correctLowConf += evalStatsArr[i].correctLowConf;
152    evalStatsDoc.incorrectHighConf += evalStatsArr[i].incorrectHighConf;
153  }
154  return evalStatsDoc;
155};
156
157/**
158 * Throw an AbortError if the given signal has fired.
159 * Matches the shape of fetch()/streams APIs so callers can `if (e.name === 'AbortError')`.
160 * @param {AbortSignal} [signal]
161 */
162const throwIfAborted = (signal) => {
163  if (!signal || !signal.aborted) return;
164  const reason = signal.reason instanceof Error ? signal.reason : undefined;
165  if (typeof DOMException !== 'undefined') {
166    throw new DOMException(reason ? reason.message : 'Recognition aborted', 'AbortError');
167  }
168  const err = new Error(reason ? reason.message : 'Recognition aborted');
169  err.name = 'AbortError';
170  throw err;
171};
172
173/**
174 * setTimeout that resolves early when `signal` fires. Never rejects.
175 * @param {number} ms
176 * @param {AbortSignal} [signal]
177 */
178const abortableDelay = (ms, signal) => new Promise((resolve) => {
179  if (signal && signal.aborted) { resolve(); return; }
180  const onAbort = () => {
181    clearTimeout(t);
182    if (signal) signal.removeEventListener('abort', onAbort);
183    resolve();
184  };
185  const t = setTimeout(() => {
186    if (signal) signal.removeEventListener('abort', onAbort);
187    resolve();
188  }, ms);
189  if (signal) signal.addEventListener('abort', onAbort, { once: true });
190});
191
192/**
193 * @param {ScribeDoc} doc
194 * @param {Object} params
195 * @param {OcrPage | OcrLine} params.page
196 * @param {?function} [params.func=null]
197 * @param {boolean}
197 [params.view=false] - Draw results on debugging canvases
198 */
199export async function evalOCRPage(doc, params) {
200  const n = 'page' in params.page ? params.page.page.n : params.page.n;
201  const binaryImage = await doc.images.getBinary(n);
202  const pageMetricsObj = doc.pageMetrics[n];
203  return gs.evalPageBase({
204    page: params.page, binaryImage, pageMetricsObj, func: params.func, view: params.view, docId: doc.id,
205  });
206}
207
208/**
209 * Compare two sets of OCR data.
210 * @param {ScribeDoc} doc
211 * @param {Array<OcrPage>} ocrA
212 * @param {Array<OcrPage>} ocrB
213 * @param  {Parameters<import('./worker/compareOCRModule.js').compareOCRPageImp>[0]['options']} [options]
214 * @param {?function} [progressCallback=null]
215 */
216export async function compareOCR(doc, ocrA, ocrB, options, progressCallback = null) {
217  /** @type {Parameters<import('./worker/compareOCRModule.js').compareOCRPageImp>[0]['options']} */
218  const compOptions = {
219    ignorePunct: scribeDocDefaults.ignorePunct,
220    ignoreCap: scribeDocDefaults.ignoreCap,
221    confThreshHigh: scribeDocDefaults.confThreshHigh,
222    confThreshMed: scribeDocDefaults.confThreshMed,
223  };
224
225  if (options) Object.assign(compOptions, options);
226
227  /** @type {Array<OcrPage>} */
228  const ocrArr = [];
229  /** @type {Array<?EvalMetrics>} */
230  const metricsArr = [];
231  /** @type {Array<Array<CompDebugBrowser | CompDebugNode>>} */
232  const debugImageArr = [];
233
234  const comparePageI = async (i) => {
235    const pageA = ocrA[i];
236    const pageB = ocrB[i];
237    // A per-page OCR subset leaves skipped pages empty, so skip any page missing from either input.
238    if (!pageA || !pageB) {
239      if (progressCallback) progressCallback();
240      return;
241    }
242    // Some option combinations need the page image and some do not.
243    // Skip it when unneeded for performance, and so accuracy benchmarks can run without an image.
244    const mode = compOptions.mode || 'stats';
245    const evalConflicts = compOptions.evalConflicts ?? true;
246    const supplementComp = compOptions.supplementComp ?? false;
247    const skipImage = (mode === 'stats' && !supplementComp) || (mode === 'comb' && !evalConflicts && !supplementComp);
248    const binaryImage = skipImage ? null : await doc.images.getBinary(pageA.n);
249    const res = await gs.compareOCRPageImp({
250      pageA,
251      pageB,
252      binaryImage,
253      pageMetricsObj: doc.pageMetrics[pageA.n],
254      options: compOptions,
255      docId: doc.id,
256    });
257
258    ocrArr[i] = res.page;
259
260    metricsArr[i] = res.metrics;
261
262    if (res.debugImg) debugImageArr[i] = res.debugImg;
263    if (progressCallback) progressCallback();
264  };
265
266  const indices = [...Array(ocrA.length).keys()];
267  const compPromises = indices.map(async (i) => comparePageI(i));
268  await Promise.allSettled(compPromises);
269
270  return { ocr: ocrArr, metrics: metricsArr, debug: debugImageArr };
271}
272
273/**
274 *  Calculate what arguments to use with Tesseract `recognize` function relating to rotation.
275 * @param {ScribeDoc} doc
276 * @param {number} n - Page number to recognize.
277 * @param {boolean} areaMode
278 */
279async function calcRecognizeRotateArgs(doc, n, areaMode) {
280  // Whether the binary image should be rotated internally by Tesseract
281  // This should always be true (Tesseract results are horrible without auto-rotate) but kept as a variable for debugging purposes.
282  const rotate = true;
283
284  // Whether the rotated images should be saved, overwriting any non-rotated images.
285  const autoRotate = true;
286
287  // Threshold (in radians) under which page angle is considered to be effectively 0.
288  const angleThresh = 0.0008726646;
289
290  const angle = doc.pageMetrics[n]?.angle;
291
292  // Whether the page angle is already known (or needs to be detected)
293  const angleKnown = typeof (angle) === 'number';
294
295  const nativeN = await doc.images.getNative(n);
296
297  // Calculate additional rotation to apply to page.  Rotation should not be applied if page has already been rotated.
298  const rotateDegrees = rotate && angle && Math.abs(angle || 0) > 0.05 && !nativeN.rotated ? angle * -1 : 0;
299  const rotateRadians = rotateDegrees * (Math.PI / 180);
300
301  let saveNativeImage = false;
302  let saveBinaryImageArg = false;
303
304  // Images are not saved when using "recognize area" as these intermediate images are cropped.
305  if (!areaMode) {
306    const binaryN = await doc.images.binary[n];
307    // Images are saved if either (1) we do not have any such image at present or (2) the current version is not rotated but the user has the "auto rotate" option enabled.
308    if (autoRotate && !nativeN.rotated[n] && (!angleKnown || Math.abs(rotateRadians) > angleThresh)) saveNativeImage = true;
309    if (!binaryN || autoRotate && !binaryN.rotated && (!angleKnown || Math.abs(rotateRadians) > angleThresh)) saveBinaryImageArg = true;
310  }
311
312  return {
313    angleThresh,
314    angleKnown,
315    rotateRadians,
316    saveNativeImage,
317    saveBinaryImageArg,
318  };
319}
320
321/**
322 * Lower-level function to run OCR for a single page.
323 * Requires additional code to handle the results; for advanced users only.
324 * Most users should use `recognize` instead to recognize all pages in a document.
325 *
326 * @param {ScribeDoc} doc
327 * @param {number} n - Page number to recognize.
328 * @param {boolean} legacy -
329 * @param {boolean} lstm -
330 * @param {boolean} areaMode -
331 * @param {Object<string, string>} tessOptions - Options to pass to Tesseract.js.
332 * @param {boolean} [debugVis=false] - Generate instructions for debugging visualizations.
333 * @param {?Array<string>} [langs=null] - Languages for this job. When set, the worker ensures its
334 *    engine matches before recognizing, so concurrent documents in different languages stay isolated.
335 * @param {boolean} [vanillaMode=false] - Use the vanilla Tesseract.js model.
336 */
337export async function recognizePageImp(doc, n, legacy, lstm, areaMode, tessOptions = {}, debugVis = false, langs = null, vanillaMode = false) {
338  const {
339    angleThresh, angleKnown, rotateRadians, saveNativeImage, saveBinaryImageArg,
340  } = await calcRecognizeRotateArgs(doc, n, areaMode);
341
342  const nativeN = await doc.images.getNative(n);
343
344  if (!nativeN) throw new Error(`No image source found for page ${n}`);
345
346  const config = {
347    ...{
348      rotateRadians, rotateAuto: !angleKnown, legacy, lstm,
349    },
350    ...tessOptions,
351  };
352
353  const pageDims = doc.pageMetrics[n].dims;
354
355  // If `legacy` and `lstm` are both `false`, recognition is not run, but layout analysis is.
356  // This combination of options would be set for debug mode, where the point of running Tesseract
357  // is to get debugging images for layout analysis rather than get text.
358  const runRecognition = legacy || lstm;
359
360  const resArr = await gs.recognizeAndConvert2({
361    image: nativeN.src,
362    options: config,
363    output: {
364      // text, blocks, hocr, and tsv must all be `false` to disable recognition
365      text: runRecognition,
366      blocks: runRecognition,
367      hocr: runRecognition,
368      tsv: runRecognition,
369      layoutBlocks: !runRecognition,
370      imageBinary: saveBinaryImageArg,
371      imageColor: saveNativeImage,
372      debug: true,
373      debugVis,
374    },
375    n,
376    knownAngle: doc.pageMetrics[n].angle,
377    pageDims,
378    langs,
379    vanillaMode,
380  });
381
382  const res0 = await resArr[0];
383
384  const elapsedSec = res0.recognitionTime / 1000;
385  if (scribeDocDefaults.printRecognitionTime === true || (typeof scribeDocDefaults.printRecognitionTime === 'number' && elapsedSec > scribeDocDefaults.printRecognitionTime)) {
386    console.log(`Page ${n} recognition time: ${elapsedSec.toFixed(2)}s`);
387  }
388
389  if (!angleKnown) doc.pageMetrics[n].angle = (res0.recognize.rotateRadians || 0) * (180 / Math.PI) * -1;
390
391  // An image is rotated if either the source was rotated or rotation was applied by Tesseract.
392  const isRotated = Boolean(res0.recognize.rotateRadians || 0) || nativeN.rotated;
393
394  // Images from Tesseract should not overwrite the existing images in the case where rotateAuto is true,
395  // but no significant rotation was actually detected.
396  const significantRotation = Math.abs(res0.recognize.rotateRadians || 0) > angleThresh;
397
398  const upscale = res0.recognize.upscale || false;
399  if (saveBinaryImageArg && res0.recognize.imageBinary && (significantRotation || !doc.images.binary[n])) {
400    doc.images.binaryProps[n] = { rotated: isRotated, upscaled: upscale, colorMode: 'binary' };
401    doc.images.binary[n] = new ImageWrapper(n, res0.recognize.imageBinary, 'binary', isRotated, upscale);
402  }
403
404  if (saveNativeImage && res0.recognize.imageColor && significantRotation) {
405    doc.images.nativeProps[n] = { rotated: isRotated, upscaled: upscale, colorMode: scribeDocDefaults.colorMode };
406    doc.images.native[n] = new ImageWrapper(n, res0.recognize.imageColor, 'native', isRotated, upscale);
407  }
408
409  return resArr;
410}
411
412/**
413 * Convert from raw OCR data to the internal hocr format used here
414 * Currently supports .hocr (used by Tesseract), Abbyy .xml, and stext (an intermediate data format used by mupdf).
415 *
416 * @param {string} ocrRaw - String containing raw OCR data for single page.
417 * @param {number} n - Page number
418 * @param {TextSource} format - Format of raw data.
419 * @param {boolean} [scribeMode=false] - Whether this is HOCR data from this program.
420 * @returns {Promise<Awaited<ReturnType<typeof import('./worker/generalWorker.js').recognizeAndConvert>>['convert']>}
421 */
422async function convertOCRPage(ocrRaw, n, format, scribeMode = false) {
423  await gs.getGeneralScheduler();
424  let res;
425  if (format === 'hocr') {
426    res = await gs.convertPageHocr({ ocrStr: ocrRaw, n, scribeMode });
427  } else if (format === 'abbyy') {
428    res = await gs.convertPageAbbyy({ ocrStr: ocrRaw, n });
429  } else if (format === 'alto') {
430    res = await gs.convertPageAlto({ ocrStr: ocrRaw, n });
431  } else if (format === 'textract') {
432    // res = await gs.convertPageTextract({ ocrStr: ocrRaw, n });
433  } else if (format === 'azure_doc_intel') {
434    // res = await gs.convertDocAzureDocIntel({ ocrStr: ocrRaw, });
435  } else if (format === 'google_doc_ai') {
436    // Document-level format, handled in convertOCR
437  } else if (format === 'google_vision') {
438    res = await gs.convertPageGoogleVision({ ocrStr: ocrRaw, n });
439  } else if (format === 'stext') {
440    res = await gs.convertPageStext({ ocrStr: ocrRaw, n });
441  } else if (format === 'text') {
442    res = await gs.convertPageText({ textStr: ocrRaw });
443  } else if (format === 'docx') {
444    console.error('format does not support page-level import.');
445    // res = await gs.convertDocDocx({ docxData: ocrRaw });
446  } else {
447    throw new Error(`Invalid format: ${format}`);
448  }
449
450  return res;
451}
452
453/**
454 * Install a parsed `OcrPage` into the doc at index `n`.
455 *
456 * @param {ScribeDoc} doc
457 * @param {number} n
458 * @param {OcrPage} page
459 * @param {object} options
460 * @param {string} options.engineName - Name of the OCR engine this page came from.
461 * @param {LayoutDataTablePage} [options.dataTables] - Per-page layout tables.
462 * @param {Object<string,string>} [options.warn] - Per-page conversion warning.
463 * @param {boolean} [options.mainData] - When true, this page's data drives pageMetrics and convertPageWarn for index `n`. Default true.
464 *   Set false when inserting a secondary engine's results into a doc that already has a primary OCR pass.
465 * @param {boolean} [options.setActive] - When true, point `doc.ocr.active` at this engine's array. Default true.
466 *   Set false to leave `doc.ocr.active` unchanged (used by recognition's per-page callback, which assigns active itself at the end of the run).
467 */
468export function insertParsedPage(doc, n, page, {
469  engineName, dataTables, warn = {}, mainData = true, setActive = true,
470}) {
471  addCircularRefsOcr([page]);
472
473  if (!doc.ocr[engineName]) doc.ocr[engineName] = Array(doc.inputData.pageCount);
474  doc.ocr[engineName][n] = page;
475  if (setActive) doc.ocr.active = doc.ocr[engineName];
476
477  if (mainData) {
478    doc.convertPageWarn[n] = warn;
479    if (page.dims && page.dims.height && page.dims.width) doc.pageMetrics[n] = new PageMetrics(page.dims);
480    doc.pageMetrics[n].angle = page.angle;
481  }
482
483  doc.inputData.xmlMode[n] = true;
484
485  if (dataTables && Object.keys(doc.layoutDataTables.pages[n].tables).length === 0) {
486    addCircularRefsDataTables([dataTables]);
487    doc.layoutDataTables.pages[n] = dataTables;
488  }
489
490  doc.progressHandler({ n, type: 'convert', info: { engineName } });
491}
492
493/**
494 * This function is called after running a `convertPage` (or `recognizeAndConvert`) function, updating this document with the results.
495 * This needs to be a separate function from `convertOCRPage`, given that sometimes recognition and conversion are combined by using `recognizeAndConvert`.
496 *
497 * @param {ScribeDoc} doc
498 * @param {Awaited<ReturnType<typeof import('./worker/generalWorker.js').recognizeAndConvert>>['convert']} params
499 * @param {number} n
500 * @param {boolean} mainData
501 * @param {string} engineName - Name of OCR engine.
502 */
503async function convertPageCallback(doc, {
504  pageObj, dataTables, warn, langSet,
505}, n, mainData, engineName) {
506  const fontPromiseArr = [];
507  if (langSet && langSet.has('chi_sim')) fontPromiseArr.push(loadChiSimFont());
508  if (langSet && (langSet.has('rus') || langSet.has('ukr') || langSet.has('ell'))) {
509    fontPromiseArr.push(loadBuiltInFontsRaw('all'));
510  } else {
511    fontPromiseArr.push(loadBuiltInFontsRaw());
512  }
513  await Promise.all(fontPromiseArr);
514
515  if (['Tesseract Legacy', 'Tesseract LSTM'].includes(engineName)) doc.ocr['Tesseract Latest'][n] = pageObj;
516
517  insertParsedPage(doc, n, pageObj, {
518    engineName, dataTables, warn, mainData, setActive: false,
519  });
520}
521
522/**
523 * Convert from raw OCR data to the internal hocr format used here
524 * Currently supports .hocr (used by Tesseract), Abbyy .xml, and stext (an intermediate data format used by mupdf).
525 *
526 * @param {ScribeDoc} doc
527 * @param {string[]} ocrRawArr - Array with raw OCR data, with an element for each page
528 * @param {boolean} mainData - Whether this is the "main" data that document metrics are calculated from.
529 *  For imports of user-provided data, the first data provided should be flagged as the "main" data.
530 *  For Tesseract.js recognition, the Tesseract Legacy results should be flagged as the "main" data.
531 * @param {TextSource} format - Format of raw data.
532 * @param {string} engineName - Name of OCR engine.
533 * @param {boolean} [scribeMode=false] - Whether this is HOCR data from this program.
534 * @param {?PageMetrics[]} [pageMetrics=null] - Page metrics to use for the pages (Textract only).
535 * @param {Object} [options]
536 * @param {'width' | 'sentence'} [options.docxLineSplitMode] - DOCX line-split mode.
537 *    Defaults to `scribeDocDefaults.docxLineSplitMode`. Ignored for non-docx formats.
538 */
539export async function convertOCR(doc, ocrRawArr, mainData, format, engineName, scribeMode, pageMetrics = null, options = {}) {
540  const docxLineSplitMode = options.docxLineSplitMode ?? scribeDocDefaults.docxLineSplitMode;
541  const promiseArr = [];
542  if (format === 'textract') {
543    if (!pageMetrics || !pageMetrics[0]?.dims) throw new Error('Page metrics must be provided for Textract data.');
544    const pageDims = pageMetrics.map((metrics) => (metrics.dims));
545
546    // When multiple Textract entries exist (per-page files), each file contains
547    // blocks with Page=1. Process each individually with the correct pageNum
548    // to avoid merging all pages into page 0.
549    if (ocrRawArr.length > 1) {
550      for (let i = 0; i < ocrRawArr.length; i++) {
551        const res = await gs.convertDocTextract({ ocrStr: [ocrRawArr[i]], pageDims: [pageDims[i]], pageNum: i });
552        if (res.length > 0) {
553          await convertPageCallback(doc, res[0], i, mainData, engineName);
554        }
555      }
556    } else {
557      const res = await gs.convertDocTextract({ ocrStr: ocrRawArr, pageDims });
558      for (let n = 0; n < res.length; n++) {
559        await convertPageCallback(doc, res[n], n, mainData, engineName);
560      }
561    }
562    return;
563  }
564
565  if (format === 'azure_doc_intel') {
566    if (!pageMetrics || !pageMetrics[0]?.dims) throw new Error('Page metrics must be provided for Azure Document Intelligence data.');
567    const pageDims = pageMetrics.map((metrics) => (metrics.dims));
568    const res = await gs.convertDocAzureDocIntel({ ocrStr: ocrRawArr, pageDims });
569    for (let n = 0; n < res.length; n++) {
570      await convertPageCallback(doc, res[n], n, mainData, engineName);
571    }
572    return;
573  }
574
575  if (format === 'google_doc_ai') {
576    if (!pageMetrics || !pageMetrics[0]?.dims) throw new Error('Page metrics must be provided for Google Document AI data.');
577    const pageDims = pageMetrics.map((metrics) => (metrics.dims));
578    const res = await gs.convertDocGoogleDocAI({ ocrStr: ocrRawArr, pageDims });
579    for (let n = 0; n < res.length; n++) {
580      await convertPageCallback(doc, res[n], n, mainData, engineName);
581    }
582    return;
583  }
584
585  if (format === 'google_vision' && pageMetrics && pageMetrics[0]?.dims) {
586    for (let n = 0; n < ocrRawArr.length; n++) {
587      const res = await gs.convertPageGoogleVision({ ocrStr: ocrRawArr[n], n, pageDims: pageMetrics[n].dims });
588      await convertPageCallback(doc, res, n, mainData, engineName);
589    }
590    return;
591  }
592
593  if (format === 'text') {
594    const res = await gs.convertPageText({ textStr: ocrRawArr[0] });
595
596    if (res.length > doc.inputData.pageCount) doc.inputData.pageCount = res.length;
597
598    for (let i = 0; i < res.length; i++) {
599      if (!doc.layoutRegions.pages[i]) doc.layoutRegions.pages[i] = new LayoutPage(i);
600    }
601
602    for (let i = 0; i < res.length; i++) {
603      if (!doc.layoutDataTables.pages[i]) doc.layoutDataTables.pages[i] = new LayoutDataTablePage(i);
604    }
605
606    for (let n = 0; n < res.length; n++) {
607      await convertPageCallback(doc, res[n], n, mainData, engineName);
608    }
609    return;
610  }
611
612  if (format === 'docx') {
613    const res = await gs.convertDocDocx({ docxData: ocrRawArr[0], lineSplitMode: docxLineSplitMode, docId: doc.id });
614
615    if (res.length > doc.inputData.pageCount) doc.inputData.pageCount = res.length;
616
617    for (let i = 0; i < res.length; i++) {
618      if (!doc.layoutRegions.pages[i]) doc.layoutRegions.pages[i] = new L
618ayoutPage(i);
619    }
620
621    for (let i = 0; i < res.length; i++) {
622      if (!doc.layoutDataTables.pages[i]) doc.layoutDataTables.pages[i] = new LayoutDataTablePage(i);
623    }
624
625    for (let n = 0; n < res.length; n++) {
626      await convertPageCallback(doc, res[n], n, mainData, engineName);
627    }
628    return;
629  }
630
631  for (let n = 0; n < ocrRawArr.length; n++) {
632    promiseArr.push(convertOCRPage(ocrRawArr[n], n, format, scribeMode)
633      .then((res) => convertPageCallback(doc, res, n, mainData, engineName)));
634  }
635  await Promise.all(promiseArr);
636}
637
638/**
639 * @param {ScribeDoc} doc
640 * @param {boolean} legacy
641 * @param {boolean} lstm
642 * @param {boolean} mainData
643 * @param {Array<string>} [langs=['eng']]
644 * @param {boolean} [vanillaMode=false]
645 * @param {Object<string, string>} [config={}]
646 * @param {?boolean[]} [ocrPageMask=null] - Per-page mask. When set, only `true` pages are recognized.
647 */
648async function recognizeAllPages(doc, legacy = true, lstm = true, mainData = false, langs = ['eng'], vanillaMode = false, config = {}, ocrPageMask = null) {
649  // Render all PDF pages to PNG if needed
650  // This step should not create binarized images as they will be created by Tesseract during recognition.
651  if (doc.inputData.pdfMode) await doc.images.preRenderRange({ min: 0, max: doc.images.pageCount - 1, binary: false });
652
653  if (legacy) {
654    const oemText = 'Tesseract Legacy';
655    if (!doc.ocr[oemText]) doc.ocr[oemText] = Array(doc.inputData.pageCount);
656    doc.ocr.active = doc.ocr[oemText];
657  }
658
659  if (lstm) {
660    const oemText = 'Tesseract LSTM';
661    if (!doc.ocr[oemText]) doc.ocr[oemText] = Array(doc.inputData.pageCount);
662    doc.ocr.active = doc.ocr[oemText];
663  }
664
665  // 'Tesseract Latest' includes the last version of Tesseract to run.
666  // It exists only so that data can be consistently displayed during recognition,
667  // should never be enabled after recognition is complete, and should never be editable by the user.
668  {
669    const oemText = 'Tesseract Latest';
670    if (!doc.ocr[oemText]) doc.ocr[oemText] = Array(doc.inputData.pageCount);
671    doc.ocr.active = doc.ocr[oemText];
672  }
673
674  await gs.initTesseract({
675    anyOk: false, vanillaMode, langs, config,
676  });
677
678  // If Legacy and LSTM are both requested, LSTM completion is tracked by a second array of promises (`promisesB`).
679  // In this case, `convertPageCallbackBrowser` can be run after the Legacy recognition is finished,
680  // however this function only returns after all recognition is completed.
681  // This provides no performance benefit in absolute terms, however halves the amount of time the user has to wait
682  // before seeing the initial recognition results.
683  // When a per-page selection mask is supplied, recognize only the selected pages.
684  // `resolvesA`/`resolvesB` are keyed by page index (sparse) so each result lands in its correct slot,
685  // while `promisesA`/`promisesB` stay dense for `Promise.all`.
686  const inputPages = ocrPageMask ? [...Array(doc.images.pageCount).keys()].filter((i) => ocrPageMask[i]) : [...Array(doc.images.pageCount).keys()];
687  const promisesA = [];
688  const resolvesA = [];
689  const promisesB = [];
690  const resolvesB = [];
691
692  for (const x of inputPages) {
693    promisesA.push(new Promise((resolve, reject) => {
694      resolvesA[x] = { resolve, reject };
695    }));
696    promisesB.push(new Promise((resolve, reject) => {
697      resolvesB[x] = { resolve, reject };
698    }));
699  }
700
701  // Upscaling is enabled only for image data, and only if the user has explicitly enabled it.
702  // For PDF data, if upscaling is desired, that should be handled by rendering the PDF at a higher resolution.
703  const upscale = doc.inputData.imageMode && scribeDocDefaults.enableUpscale;
704
705  const configPage = { upscale };
706
707  for (const x of inputPages) {
708    recognizePageImp(doc, x, legacy, lstm, false, configPage, scribeDocDefaults.debugVis, langs, vanillaMode).then(async (resArr) => {
709      const res0 = await resArr[0];
710
711      if (res0.recognize.debugVis) {
712        const { ScrollView } = await import('../scrollview-web/scrollview/ScrollView.js');
713        const sv = new ScrollView({
714          lightTheme: true,
715        });
716        await sv.processVisStr(res0.recognize.debugVis);
717        doc.vis[x] = await sv.getAll(true);
718      }
719
720      if (legacy) {
721        await convertPageCallback(doc, res0.convert.legacy, x, mainData, 'Tesseract Legacy');
722        resolvesA[x].resolve();
723      } else if (lstm) {
724        await convertPageCallback(doc, res0.convert.lstm, x, false, 'Tesseract LSTM');
725        resolvesA[x].resolve();
726      }
727
728      if (legacy && lstm) {
729        (async () => {
730          const res1 = await resArr[1];
731          await convertPageCallback(doc, res1.convert.lstm, x, false, 'Tesseract LSTM');
732          resolvesB[x].resolve();
733        })();
734      }
735    });
736  }
737
738  await Promise.all(promisesA);
739
740  if (mainData) {
741    await checkCharWarn(doc, doc.convertPageWarn);
742  }
743
744  if (legacy && lstm) await Promise.all(promisesB);
745
746  if (lstm) {
747    const oemText = 'Tesseract LSTM';
748    doc.ocr.active = doc.ocr[oemText];
749  } else {
750    const oemText = 'Tesseract Legacy';
751    doc.ocr.active = doc.ocr[oemText];
752  }
753}
754
755/**
756 * Convert a page of raw model output (HOCR string, Textract JSON string, etc.) into
757 * the internal OcrPage model and store it under the given engine name. Shared between
758 * the per-image and document-mode custom recognition paths.
759 * @param {ScribeDoc} doc
760 * @param {string} rawData
761 * @param {number} n - Page index.
762 * @param {RecognitionModel} model
763 */
764async function convertModelRawPage(doc, rawData, n, model) {
765  const engineName = model.config.name;
766  const outputFormat = model.config.outputFormat;
767  if (model.convertPage) {
768    const convertResult = await model.convertPage(rawData, n);
769    await convertPageCallback(doc, convertResult, n, true, engineName);
770  } else if (outputFormat === 'textract') {
771    const pageDims = [doc.pageMetrics[n].dims];
772    const res = await gs.convertDocTextract({ ocrStr: rawData, pageDims, pageNum: n });
773    for (let i = 0; i < res.length; i++) {
774      await convertPageCallback(doc, res[i], n + i, true, engineName);
775    }
776  } else if (outputFormat === 'azure_doc_intel') {
777    const pageDims = [doc.pageMetrics[n].dims];
778    const res = await gs.convertDocAzureDocIntel({ ocrStr: rawData, pageDims });
779    for (let i = 0; i < res.length; i++) {
780      await convertPageCallback(doc, res[i], n + i, true, engineName);
781    }
782  } else if (outputFormat === 'google_doc_ai') {
783    const pageDims = [doc.pageMetrics[n].dims];
784    const res = await gs.convertDocGoogleDocAI({ ocrStr: rawData, pageDims, pageNum: n });
785    for (let i = 0; i < res.length; i++) {
786      await convertPageCallback(doc, res[i], n + i, true, engineName);
787    }
788  } else if (outputFormat === 'google_vision') {
789    const res = await gs.convertPageGoogleVision({ ocrStr: rawData, n, pageDims: doc.pageMetrics[n].dims });
790    await convertPageCallback(doc, res, n, true, engineName);
791  } else {
792    const res = await convertOCRPage(rawData, n, /** @type {TextSource} */ (outputFormat));
793    await convertPageCallback(doc, res, n, true, engineName);
794  }
795}
796
797/**
798 * Document-mode recognition path: the model consumes the whole PDF at once (e.g. a
799 * server proxy that renders pages and runs OCR remotely) and streams back per-page raw
800 * results. Skips browser-side pre-rendering and the per-image dispatch loop entirely.
801 * @param {ScribeDoc} doc
802 * @param {Object} options
803 * @param {RecognitionModel} options.model
804 * @param {Object} [options.modelOptions]
805 * @param {AbortSignal} [options.signal]
806 * @param {?boolean[]} [ocrPageMask=null] - Per-page mask from `recognize`.
807 *   The whole-PDF API cannot select individual pages, so the mask is applied at document granularity.
808 *   No page selected skips OCR entirely. Any other selection OCRs the whole PDF, with a warning when the selection was partial.
809 */
810async function recognizeCustomModelDocumentMode(doc, options, ocrPageMask = null) {
811  const model = options.model;
812  const modelOptions = options.modelOptions || {};
813  const signal = options.signal;
814  const engineName = model.config.name;
815  // modelOptions passed to the model has `signal` merged in so network-backed models
816  // can cancel their in-flight HTTP requests. We keep the caller's modelOptions object
817  // untouched to avoid mutating a user-supplied reference.
818  const modelOptionsWithSignal = { ...modelOptions, signal };
819
820  if (!doc.ocr[engineName]) doc.ocr[engineName] = Array(doc.inputData.pageCount);
821  if (scribeDocDefaults.keepRawData && !doc.ocrRaw[engineName]) doc.ocrRaw[engineName] = Array(doc.inputData.pageCount);
822
823  if (ocrPageMask && !ocrPageMask.some(Boolean)) {
824    for (let n = 0; n < doc.inputData.pageCount; n++) {
825      doc.ocr[engineName][n] = (doc.ocr.pdf && doc.ocr.pdf[n]) || new OcrPage(n, doc.pageMetrics[n].dims);
826    }
827    doc.ocr.active = doc.ocr[engineName];
828    return doc.ocr.active;
829  }
830  if (ocrPageMask && !ocrPageMask.every(Boolean)) {
831    const ocrCount = ocrPageMask.filter(Boolean).length;
832    doc.warningHandler({ message: `Document-mode recognition OCRs the whole PDF; per-page OCR selection (${ocrCount}/${doc.inputData.pageCount} selected) cannot be applied and all pages will be sent.` });
833  }
834  // The whole PDF is OCR'd in document mode, so every page's active layer becomes OCR.
835  doc.inputData.ocrApplied = Array(doc.inputData.pageCount).fill(true);
836
837  throwIfAborted(signal);
838
839  const pageDims = doc.pageMetrics.map((m) => m.dims);
840  const pdfBytes = doc.inputData.pdfMode && doc.images.pdfData ? new Uint8Array(doc.images.pdfData) : null;
841
842  const stream = await model.recognizeDocument(
843    { pdfBytes, pageCount: doc.inputData.pageCount, pageDims },
844    modelOptionsWithSignal,
845  );
846
847  const failedPagesDoc = [];
848  let lastErrMsg = '';
849  let docAborted = false;
850  try {
851    for await (const entry of stream) {
852      if (signal && signal.aborted) { docAborted = true; break; }
853      if (!entry) continue;
854      if (entry.error) {
855        const errMsg = entry.error.message || String(entry.error);
856        failedPagesDoc.push(entry.pageNum);
857        lastErrMsg = errMsg;
858        doc.warningHandler({ message: `Recognition failed for page ${entry.pageNum}: ${errMsg}`, page: entry.pageNum });
859        doc.ocr[engineName][entry.pageNum] = new OcrPage(entry.pageNum, doc.pageMetrics[entry.pageNum].dims);
860        continue;
861      }
862      const { pageNum, rawData } = entry;
863      if (scribeDocDefaults.keepRawData) doc.ocrRaw[engineName][pageNum] = rawData;
864      doc.progressHandler({
865        n: pageNum, type: 'recognize', info: { status: 'received', engineName, timestamp: Date.now() },
866      });
867      await convertModelRawPage(doc, rawData, pageNum, model);
868    }
869  } finally {
870    // Best-effort: if the model's generator supports early termination via return(),
871    // give it a chance to clean up (e.g. cancel an in-flight fetch). Works for both
872    // native async generators (have .return) and hand-rolled async iterators.
873    if (docAborted && stream && typeof stream.return === 'function') {
874      try { await stream.return(); } catch (_) { /* ignore */ }
875    }
876  }
877
878  // Preserve partial results on abort — caller (and any resume layer) may want them.
879  doc.ocr.active = doc.ocr[engineName];
880  if (scribeDocDefaults.keepRawData) doc.ocrRaw.active = doc.ocrRaw[engineName];
881
882  // Always throw if the signal is aborted, regardless of whether the library or the
883  // model's own early-return terminated the for-await loop first.
884  throwIfAborted(signal);
885
886  if (failedPagesDoc.length === doc.inputData.pageCount) {
887    throw new Error(`Recognition failed for all pages. Last error message: ${lastErrMsg}`);
888  }
889  if (failedPagesDoc.length > 0) {
890    failedPagesDoc.sort((a, b) => a - b);
891    doc.warningHandler({ message: `Recognition failed for ${failedPagesDoc.length} page(s) (${failedPagesDoc.join(', ')}). These pages will have no OCR data.` });
892  }
893
894  return doc.ocr.active;
895}
896
897/**
898 * Recognize all pages using a custom (external) recognition model.
899 * Called by `recognize` when `options.model` is provided.
900 *
901 * @param {ScribeDoc} doc
902 * @param {Object} options - Options object from `recognize`, guaranteed non-null with `model` set.
903 * @param {RecognitionModel} options.model
904 * @param {Object} [options.modelOptions]
905 * @param {Array<string>} [options.langs]
906 * @param {AbortSignal} [options.signal] - Optional abort signal.
907 *   When aborted, recognition stops scheduling new pages, drains in-flight work,
908 *   preserves whatever pages completed, and throws an AbortError.
909 * @param {?boolean[]} [ocrPageMask=null] - Per-page mask from `recognize`.
910 *   When set, only `true` pages are sent to the model and skipped pages keep their native (PDF) text.
911 */
912async function recognizeCustomModel(doc, options, ocrPageMask = null) {
913  const model = options.model;
914  const modelOptions = options.modelOptions || {};
915  const signal = options.signal;
916  const engineName = model.config.name;
917  const outputFormat = model.config.outputFormat;
918  // modelOptions passed to the model has `signal` merged in so network-backed models
919  // can cancel their in-flight HTTP requests. We keep the caller's modelOptions object
920  // untouched to avoid mutating a user-supplied reference.
921  const modelOptionsWithSignal = { ...modelOptions, signal };
922
923  const knownFormats = ['hocr', 'abbyy', 'alto', 'textract', 'azure_doc_intel', 'google_doc_ai', 'google_vision', 'stext', 'text'];
924  if (!knownFormats.includes(outputFormat) && !model.convertPage) {
925    throw new Error(`Model output format '${outputFormat}' is not supported. Provide a convertPage method on the model.`);
926  }
927
928  await gs.getGeneralScheduler();
929
930  // Document-mode models OCR the whole PDF in a single call, so route them to their own path.
931  if (model.config.documentMode) return recognizeCustomModelDocumentMode(doc, options, ocrPageMask);
932
933  // Initialize array for custom model results
934  if (!doc.ocr[engineName]) doc.ocr[engineName] = Array(doc.inputData.pageCount);
935  if (scribeDocDefaults.keepRawData && !doc.ocrRaw[engineName]) doc.ocrRaw[engineName] = Array(doc.inputData.pageCount);
936
937  // No page selected: skip the model entirely, filling each page from its native (PDF) text.
938  if (ocrPageMask && !ocrPageMask.some(Boolean)) {
939    for (let n = 0; n < doc.inputData.pageCount; n++) {
940      doc.ocr[engineName][n] = (doc.ocr.pdf && doc.ocr.pdf[n]) || new OcrPage(n, doc.pageMetrics[n].dims);
941    }
942    doc.ocr.active = doc.ocr[engineName];
943    return doc.ocr.active;
944  }
945
946  // Different cloud providers implement usage quotas in different ways.
947  // AWS Textract (Sync) uses transactions per second (TPS).
948  // AWS Textract (Async) uses both transactions per second (TPS) and concurrent request limits.
949  // Google Vision (Sync) uses requests per minute (RPM).
950  // The core distinction is that TPS limits the number of requests SENT per second,
951  // rather than the number of live requests at any given time.
952  const configRateLimit = modelOptions.rateLimit ?? model.config.rateLimit ?? null;
953  const regionCount = Array.isArray(modelOptions?.region) ? modelOptions.region.length : 1;
954  const baseTps = configRateLimit?.tps ?? (configRateLimit?.rpm ? configRateLimit.rpm / 60 : null);
955  const tps = baseTps != null ? baseTps * regionCount : null;
956  let adaptiveTps = tps;
957  let lastRequestTime = 0;
958
959  let concurrency;
960  if (modelOptions.maxConcurrency != null) {
961    concurrency = modelOptions.maxConcurrency;
962  } else if (tps != null) {
963    // When tps is set, that is the primary means of limiting concurrency.
964    // This is set to a large number as a safeguard.
965    concurrency = 30;
966  } else if (opt.workerN) {
967    concurrency = opt.workerN;
968  } else if (typeof process === 'undefined') {
969    concurrency = Math.min(Math.round((globalThis.navigator.hardwareConcurrency || 8) / 2), 6);
970  } else {
971    const cpuN = Math.floor((await import('node:os')).cpus().length / 2);
972    concurrency = Math.max(Math.min(cpuN - 1, 8), 1);
973  }
974
975  // Process all selected pages with limited concurrency.
976  // Skipped pages keep their native (PDF) text so `doc.ocr[engineName]` (the active layer below) has no holes.
977  const pages = [...Array(doc.images.pageCount).keys()].filter((n) => !ocrPageMask || ocrPageMask[n]);
978  if (ocrPageMask) {
979    for (let n = 0; n < doc.images.pageCount; n++) {
980      if (!ocrPageMask[n]) doc.ocr[engineName][n] = (doc.ocr.pdf && doc.ocr.pdf[n]) || new OcrPage(n, doc.pageMetrics[n].dims);
981    }
982  }
983  const executing = new Set();
984
985  const maxConsecutiveFailures = 3;
986  let consecutiveFailures = 0;
987  let lastErrorMessage = '';
988  let quitEarly = false;
989  /** @type {number[]} */
990  const failedPages = [];
991
992  for (const n of pages) {
993    if (quitEarly) break;
994    if (signal && signal.aborted) break;
995    // eslint-disable-next-line no-loop-func
996    const p = (async () => {
997      if (quitEarly) return;
998      if (signal && signal.aborted) return;
999
1000      const nativeN = await doc.images.getNative(n);
1001      if (!nativeN) {
1002        doc.warningHandler({ message: `No image found for page ${n}, skipping.`, page: n });
1003        doc.ocr[engineName][n] = new OcrPage(n, doc.pageMetrics[n].dims);
1004        return;
1005      }
1006
1007      // Convert base64 data URL to Uint8Array for the model
1008      const base64Data = nativeN.src.split(',')[1];
1009      const binaryStr = atob(base64Data);
1010      const imageData = new Uint8Array(binaryStr.length);
1011      for (let i = 0; i < binaryStr.length; i++) {
1012        imageData[i] = binaryStr.charCodeAt(i);
1013      }
1014
1015      // Drop the cached render once its bytes are copied.
1016      // Without this every page's rendered image stays cached for the whole run,
1017      // and a large document exhausts the heap before recognition finishes.
1018      // TODO: This will delete images at the user's current position in the viewer.
1019      // Additionally, rendering 1k pages in the viewer will still cause a crash.
1020      // We should switch to a more robust system for clearing cache.
1021      // Gated to large documents for now since small docs are not a memory risk.
1022      if (doc.inputData.pdfMode && doc.inputData.pageCount > 100) delete doc.images.native[n];
1023
1024      const maxThrottleRetries = 3;
1025      /** @type {RecognitionResult} */
1026      let result = { success: false, format: '' };
1027
1028      // Attempt recognition with up to maxThrottleRetries retries for throttling errors.
1029      // Attempt 0 is the initial request; attempts 1–maxThrottleRetries are retries.
1030      for (let attempt = 0; attempt <= maxThrottleRetries; attempt++) {
1031        if (signal && signal.aborted) return;
1032        // TPS pacing: claim the next available dispatch slot before yielding.
1033        if (adaptiveTps != null && adaptiveTps > 0) {
1034          const now = Date.now();
1035          // Using a number slightly above 1 second to account for variation.
1036          const minInterval = 1050 / adaptiveTps;
1037          const targetTime = Math.max(now, lastRequestTime + minInterval);
1038          lastRequestTime = targetTime;
1039          const waitMs = targetTime - now;
1040          if (waitMs > 0) {
1041            await abortableDelay(waitMs, signal);
1042            if (signal && signal.aborted) return;
1043          }
1044        }
1045
1046        doc.progressHandler({ n, type: 'recognize', info: { status: 'sending', engineName, timestamp: Date.now() } });
1047        const recognizeStart = Date.now();
1048        result = await model.recognizeImage(imageData, modelOptionsWithSignal);
1049
1050        if (result.success) {
1051          const elapsedSec = (Date.now() - recognizeStart) / 1000;
1052          if (scribeDocDefaults.printRecognitionTime === true || (typeof scribeDocDefaults.printRecognitionTime === 'number' && elapsedSec > scribeDocDefaults.printRecognitionTime)) {
1053            console.log(`Page ${n} recognition time: ${elapsedSec.toFixed(2)}s`);
1054          }
1055          break;
1056        }
1057
1058        // Only throttling errors are retried.
1059        const isThrottle = model.isThrottlingError && result.error && model.isThrottlingError(result.error);
1060        if (!isThrottle) break;
1061
1062        if (attempt === maxThrottleRetries) {
1063          doc.warningHandler({ message: `Page ${n}: throttled ${maxThrottleRetries + 1} times, giving up.`, page: n });
1064          break;
1065        }
1066        const backoffMs = Math.min(1000 * (2 ** attempt), 16000);
1067        doc.warningHandler({ message: `Page ${n}: throttled by API, retrying in ${backoffMs}ms (attempt ${attempt + 1}/${maxThrottleRetries})`, page: n });
1068        if (adaptiveTps != null && adaptiveTps > 0.5) {
1069          adaptiveTps *= 0.9;
1070        }
1071        await abortableDelay(backoffMs, signal);
1072      }
1073
1074      if (signal && signal.aborted) return;
1075
1076      if (!result.success || !result.rawData) {
1077        const errMsg = result.error ? result.error.message : 'Unknown error';
1078        failedPages.push(n);
1079        doc.warningHandler({ message: `Recognition failed for page ${n}: ${errMsg}`, page: n });
1080        doc.ocr[engineName][n] = new OcrPage(n, doc.pageMetrics[n].dims);
1081        consecutiveFailures++;
1082        lastErrorMessage = errMsg;
1083        if (consecutiveFailures >= maxConsecutiveFailures) {
1084          quitEarly = true;
1085        }
1086        return;
1087      }
1088
1089      consecutiveFailures = 0;
1090
1091      const rawData = result.rawData;
1092      if (scribeDocDefaults.keepRawData) doc.ocrRaw[engineName][n] = rawData;
1093
1094      await convertModelRawPage(doc, rawData, n, model);
1095    })().then(() => executing.delete(p));
1096
1097    executing.add(p);
1098    if (executing.size >= concurrency) await Promise.race(executing);
1099  }
1100
1101  await Promise.allSettled(executing);
1102
1103  // On abort: preserve whatever pages completed and throw an AbortError.
1104  // A caller (e.g. the server proxy's resume-cache layer) may want the partial results.
1105  if (signal && signal.aborted) {
1106    doc.ocr.active = doc.ocr[engineName];
1107    if (scribeDocDefaults.keepRawData) doc.ocrRaw.active = doc.ocrRaw[engineName];
1108    throwIfAborted(signal);
1109  }
1110
1111  if (consecutiveFailures === doc.images.pageCount) {
1112    throw new Error(
1113      `Recognition failed for all pages. Last error message: ${lastErrorMessage}`,
1114    );
1115  }
1116
1117  if (quitEarly) {
1118    throw new Error(
1119      `Recognition aborted after ${consecutiveFailures} consecutive failures. Last error message: ${lastErrorMessage}`,
1120    );
1121  }
1122
1123  if (failedPages.length > 0) {
1124    failedPages.sort((a, b) => a - b);
1125    doc.warningHandler({ message: `Recognition failed for ${failedPages.length} page(s) (${failedPages.join(', ')}). These pages will have no OCR data.` });
1126  }
1127
1128  // Set active OCR to custom model results
1129  doc.ocr.active = doc.ocr[engineName];
1130  if (scribeDocDefaults.keepRawData) {
1131    doc.ocrRaw.active = doc.ocrRaw[engineName];
1132  }
1133  return doc.ocr.active;
1134}
1135
1136/**
1137 * Recognize all pages in this document.
1138 * Files for recognition should already be imported using `importFiles` before calling this function.
1139 * The results of recognition can be exported by calling `exportData` after this function.
1140 * @param {ScribeDoc} doc
1141 * @param {Object} options
1142 * @param {'speed'|'quality'} [options.mode='quality'] - Recognition mode.
1143 * @param {Array<string>} [options.langs=['eng']] - Language(s) in document.
1144 * @param {'lstm'|'legacy'|'combined'} [options.modeAdv='combined'] - Alternative method of setting recognition mode.
1145 * @param {'conf'|'data'|'none'} [options.combineMode='data'] - Method of combining OCR results. Used if OCR data already exists.
1146 * @param {('all'|'auto'|'autoShallow'|'autoDeep'|'none'|boolean[])} [options.ocrPages] - Which pages to OCR. Defaults to `scribeDocDefaults.ocrPages` (`'all'`).
1147 *    `'autoShallow'` decides per document: it skips a text-native document and OCRs an image-based one in full,
1148 *    re-OCRing one that already carries an OCR layer unless `usePDFText.ocr.main` trusts that layer.
1149 *     `'autoDeep'` (alias `'auto'`) is a strict superset of `'autoShallow'` that also OCRs the pages of an otherwise-skipped document that may hold baked-in text.
1150 *    `'all'`/`'none'` force every/no page; a boolean array (length === page count) selects pages explicitly.
1151 *    Image inputs, and any document with uploaded OCR, always OCR every page.
1152 * @param {typeof scribeDocDefaults.usePDFText} [options.usePDFText] - How to use a PDF's own extracted text, for this call.
1153 *    Defaults to `scribeDocDefaults.usePDFText`. For a document with an existing OCR layer, `ocr.main: true` trusts that
1154 *    layer as primary and skips OCR; `ocr.supp: true` merges it into a fresh OCR run; both false re-OCRs and discards it.
1155 * @param {boolean} [options.vanillaMode=false] - Whether to use the vanilla Tesseract.js model.
1156 * @param {Object<string, string>} [options.config={}] - Config params to pass to to Tesseract.js.
1157 * @param {RecognitionModel} [options.model] - Custom recognition model. See docs.
1158 * @param {Object} [options.modelOptions={}] - Options passed to the model's `recognizeImage` method.
1159 * @param {AbortSignal} [options.signal] - Optional abort signal for cancelling a custom-model
1160 *    recognition run. When aborted, scribe.js stops scheduling new pages, drains any in-flight
1161 *    page requests (so their network activity is not wasted), preserves the OCR data of pages
1162 *    that already completed, and throws an AbortError. Only applies when `options.model` is set.
1163 */
1164export async function recognize(doc, options = {}) {
1165  if (!doc.inputData.pdfMode && !doc.inputData.imageMode) throw new Error('No PDF or image data found to recognize.');
1166
1167  // Decide which pages require OCR based on document contents and options specified.
1168  const ocrPages = options.ocrPages ?? scribeDocDefaults.ocrPages;
1169  const usePDFText = options.usePDFText ?? scribeDocDefaults.usePDFText;
1170  const pageCount = doc.inputData.pageCount;
1171  const stats = doc.inputData.pageStats;
1172
1173  /** @type {boolean[]} */
1174  let ocrPageMask;
1175  if (Array.isArray(ocrPages) && !doc.ocr['User Upload']) {
1176    // An explicit per-page mask is used directly, independent of parse-time stats.
1177    if (ocrPages.length !== pageCount) {
1178      throw new Error(`ocrPages array length (${ocrPages.length}) must equal the page count (${pageCount}).`);
1179    }
1180    ocrPageMask = ocrPages.map(Boolean);
1181  } else if (doc.ocr['User Upload'] || !stats || stats.length !== pageCount) {
1182    // Uploaded OCR keeps the existing whole-document combine path (back-compat): OCR every page unless explicitly told `'none'`.
1183    // The same whole-document fallback applies when per-page stats are unavailable.
1184    ocrPageMask = Array(pageCount).fill(ocrPages !== 'none');
1185  } else {
1186    ocrPageMask = selectOcrPages(stats, doc.inputData.pdfType, /** @type {'all'|'none'|'auto'|'autoShallow'|'autoDeep'} */ (ocrPages), usePDFText);
1187  }
1188  doc.inputData.ocrApplied = ocrPageMask.slice();
1189  const fullOcr = ocrPageMask.every(Boolean);
1190  // The keep/discard gate runs only for the `auto*` ocrPages modes (not `all`, `none`, or an explicit mask).
1191  const gateApplies = ocrPages === 'autoDeep' || ocrPages === 'auto' || ocrPages === 'autoShallow';
1192
1193  // Custom recognition model path
1194  if (options.model) {
1195    await recognizeCustomModel(doc, /** @type {{ model: RecognitionModel }} */ (options), ocrPageMask);
1196    buildCombinedLayer(doc, doc.ocr.active, ocrPageMask, gateApplies, fullOcr);
1197    return doc.ocr.active;
1198  }
1199
1200  if (!ocrPageMask.some(Boolean)) {
1201    // No page needs OCR: keep the parsed native/existing text layer as the active layer and skip recognition.
1202    if (doc.ocr.pdf) doc.ocr.active = doc.ocr.pdf;
1203    return doc.ocr.active;
1204  }
1205
1206  await gs.getGeneralScheduler();
1207
1208  const combineMode = options && options.combineMode ? options.combineMode : 'data';
1209  const vanillaMode = options && options.vanillaMode !== undefined ? options.vanillaMode : false;
1210  const config = options && options.config ? options.config : {};
1211
1212  const langs = options && options.langs ? options.langs : ['eng'];
1213  let oemMode = 'combined';
1214  if (options && options.modeAdv) {
1215    oemMode = options.modeAdv;
1216  } else if (options && options.mode) {
1217    oemMode = options.mode === 'speed' ? 'lstm' : 'legacy';
1218  }
1219
1220  const fontPromiseArr = [];
1221  // Chinese requires loading a separate font.
1222  if (langs.includes('chi_sim')) fontPromiseArr.push(loadChiSimFont());
1223  // Greek and Cyrillic require loading a version of the base fonts that include these characters.
1224  if (langs.includes('rus') || langs.includes('ukr') || langs.includes('ell')) fontPromiseArr.push(loadBuiltInFontsRaw('all'));
1225  await Promise.all(fontPromiseArr);
1226
1227  let forceMainData = false;
1228  let existingOCR;
1229  if (doc.ocr['User Upload']) {
1230    existingOCR = doc.ocr['User Upload'];
1231  } else if (
1232    doc.ocr.pdf
1233    && ((doc.inputData.pdfType === 'text' && usePDFText.native.supp)
1234      || (doc.inputData.pdfType === 'ocr' && usePDFText.ocr.supp))
1235  ) {
1236    existingOCR = doc.ocr.pdf;
1237    // If the PDF text is not the active data, it is assumed to be for supplemental purposes only.
1238    forceMainData = doc.ocr.pdf !== doc.ocr.active;
1239  }
1240
1241  // A single Tesseract engine can be used (Legacy or LSTM) or the results from both can be used and combined.
1242  if (oemMode === 'legacy' || oemMode === 'lstm') {
1243    // Tesseract is used as the "main" data unless user-uploaded data exists and only the LSTM model is being run.
1244    // This is because Tesseract Legacy provides very strong metrics, and Abbyy often does not.
1245    await recognizeAllPages(doc, oemMode === 'legacy', oemMode === 'lstm', !existingOCR, langs, vanillaMode, config, ocrPageMask);
1246
1247    // Metrics from the LSTM model are so inaccurate they are not worth using.
1248    if (oemMode === 'legacy') {
1249      const charMetrics = calcCharMetricsFromPages(doc.ocr['Tesseract Legacy']);
1250      if (Object.keys(charMetrics).length > 0) {
1251        clearObjectProperties(doc.fonts.state.charMetrics);
1252        Object.assign(doc.fonts.state.charMetrics, charMetrics);
1253      }
1254      await doc.runOptimization(doc.ocr['Tesseract Legacy']);
1255    }
1256  } else if (oemMode === 'combined') {
1257    await recognizeAllPages(doc, true, true, !existingOCR, langs, vanillaMode, config, ocrPageMask);
1258
1259    const progressCb = () => doc.progressHandler({ type: 'recognize' });
1260
1261    if (scribeDocDefaults.saveDebugImages) {
1262      doc.debug.debugImg.Combined = new Array(doc.images.pageCount);
1263      for (let i = 0; i < doc.images.pageCount; i++) {
1264        doc.debug.debugImg.Combined[i] = [];
1265      }
1266    }
1267
1268    if (existingOCR) {
1269      const oemText = 'Tesseract Combined';
1270      if (!doc.ocr[oemText]) doc.ocr[oemText] = Array(doc.inputData.pageCount);
1271      doc.ocr.active = doc.ocr[oemText];
1272
1273      if (scribeDocDefaults.saveDebugImages) {
1274        doc.debug.debugImg['Tesseract Combined'] = new Array(doc.images.pageCount);
1275        for (let i = 0; i < doc.images.pageCount; i++) {
1276          doc.debug.debugImg['Tesseract Combined'][i] = [];
1277        }
1278      }
1279    }
1280
1281    // A new version of OCR data is created for font optimization and validation purposes.
1282    // This version has the bounding box and style data from the Legacy data, however uses the text from the LSTM data whenever conflicts occur.
1283    // Additionally, confidence is set to 0 when conflicts occur. Using this version benefits both font optimiztion and validation.
1284    // For optimization, using this version rather than Tesseract Legacy excludes data that conflicts with Tesseract LSTM and is therefore likely incorrect,
1285    // as low-confidence words are excluded when calculating overall character metrics.
1286    // For validation, this version is superior to both Legacy and LSTM, as it combines the more accurate bounding boxes/style data from Legacy
1287    // with the more accurate (on average) text data from LSTM.
1288    if (!doc.ocr['Tesseract Combined Temp']) doc.ocr['Tesseract Combined Temp'] = Array(doc.inputData.pageCount);
1289
1290    {
1291      /** @type {Parameters<typeof doc.compareOCR>[2]} */
1292      const compOptions = {
1293        mode: 'comb',
1294        evalConflicts: false,
1295        legacyLSTMComb: true,
1296      };
1297
1298      const res = await compareOCR(doc, doc.ocr['Tesseract Legacy'], doc.ocr['Tesseract LSTM'], compOptions, progressCb);
1299
1300      clearObjectProperties(doc.ocr['Tesseract Combined Temp']);
1301      Object.assign(doc.ocr['Tesseract Combined Temp'], res.ocr);
1302    }
1303
1304    // Evaluate default fonts using up to 5 pages.
1305    const pageNum = Math.min(doc.images.pageCount - 1, 5);
1306    await doc.images.preRenderRange({ min: 0, max: pageNum, binary: true });
1307    const charMetrics = calcCharMetricsFromPages(doc.ocr['Tesseract Combined Temp']);
1308    if (Object.keys(charMetrics).length > 0) {
1309      clearObjectProperties(doc.fonts.state.charMetrics);
1310      Object.assign(doc.fonts.state.charMetrics, charMetrics);
1311    }
1312    await doc.runOptimization(doc.ocr['Tesseract Combined Temp']);
1313
1314    const oemText = 'Combined';
1315    if (!doc.ocr[oemText]) doc.ocr[oemText] = Array(doc.inputData.pageCount);
1316    doc.ocr.active = doc.ocr[oemText];
1317
1318    {
1319      const tessCombinedLabel = existingOCR ? 'Tesseract Combined' : 'Combined';
1320
1321      /** @type {Parameters<import('./worker/compareOCRModule.js').compareOCRPageImp>[0]['options']} */
1322      const compOptions = {
1323        mode: 'comb',
1324        debugLabel: scribeDocDefaults.saveDebugImages ? tessCombinedLabel : undefined,
1325        ignoreCap: scribeDocDefaults.ignoreCap,
1326        ignorePunct: scribeDocDefaults.ignorePunct,
1327        confThreshHigh: scribeDocDefaults.confThreshHigh,
1328        confThreshMed: scribeDocDefaults.confThreshMed,
1329        legacyLSTMComb: true,
1330      };
1331
1332      const res = await compareOCR(doc, doc.ocr['Tesseract Legacy'], doc.ocr['Tesseract LSTM'], compOptions, progressCb);
1333
1334      if (doc.debug.debugImg[tessCombinedLabel]) doc.debug.debugImg[tessCombinedLabel] = res.debug;
1335
1336      clearObjectProperties(doc.ocr[tessCombinedLabel]);
1337      Object.assign(doc.ocr[tessCombinedLabel], res.ocr);
1338    }
1339
1340    // Compare the existing text layer against a secondary text layer word-by-word.
1341    // Runs for a whole-document OCR pass or for User-Upload data.
1342    if (existingOCR && (doc.ocr['User Upload'] || fullOcr)) {
1343      if (combineMode === 'conf') {
1344        /** @type {Parameters<import('./worker/compareOCRModule.js').compareOCRPageImp>[0]['options']} */
1345        const compOptions = {
1346          debugLabel: scribeDocDefaults.saveDebugImages ? 'Combined' : undefined,
1347          supplementComp: true,
1348          ignoreCap: scribeDocDefaults.ignoreCap,
1349          ignorePunct: scribeDocDefaults.ignorePunct,
1350          confThreshHigh: scribeDocDefaults.confThreshHigh,
1351          confThreshMed: scribeDocDefaults.confThreshMed,
1352          editConf: true,
1353        };
1354
1355        const res = await compareOCR(doc, existingOCR, doc.ocr['Tesseract Combined'], compOptions, progressCb);
1356
1357        if (doc.debug.debugImg.Combined) doc.debug.debugImg.Combined = res.debug;
1358
1359        clearObjectProperties(doc.ocr.Combined);
1360        Object.assign(doc.ocr.Combined, res.ocr);
1361      } else if (combineMode === 'data') {
1362        /** @type {Parameters<import('./worker/compareOCRModule.js').compareOCRPageImp>[0]['options']} */
1363        const compOptions = {
1364          mode: 'comb',
1365          debugLabel: 'Combined',
1366          ignoreCap: scribeDocDefaults.ignoreCap,
1367          ignorePunct: scribeDocDefaults.ignorePunct,
1368          confThreshHigh: scribeDocDefaults.confThreshHigh,
1369          confThreshMed: scribeDocDefaults.confThreshMed,
1370          // If the existing data was invisible OCR text extracted from a PDF, it is assumed to not have accurate bounding boxes.
1371          useBboxB: !forceMainData && existingOCR === doc.ocr.pdf && doc.inputData.pdfMode && !!doc.inputData.pdfType && ['image', 'ocr'].includes(doc.inputData.pdfType),
1372        };
1373
1374        let res;
1375        if (forceMainData) {
1376          res = await compareOCR(doc, doc.ocr['Tesseract Combined'], existingOCR, compOptions, progressCb);
1377        } else {
1378          res = await compareOCR(doc, existingOCR, doc.ocr['Tesseract Combined'], compOptions, progressCb);
1379        }
1380
1381        if (doc.debug.debugImg.Combined) doc.debug.debugImg.Combined = res.debug;
1382
1383        clearObjectProperties(doc.ocr.Combined);
1384        Object.assign(doc.ocr.Combined, res.ocr);
1385      }
1386    }
1387  }
1388
1389  // The engine's OCR layer to route: 'Tesseract Combined' for an existing-OCR run (where `active` points elsewhere), otherwise `active` itself.
1390  const tessSource = (existingOCR && doc.ocr['Tesseract Combined']) ? doc.ocr['Tesseract Combined'] : doc.ocr.active;
1391  buildCombinedLayer(doc, tessSource, ocrPageMask, gateApplies, fullOcr);
1392  return (doc.ocr.active);
1393}

Line numbers count LF bytes from the start of the resource, as the search results do. Vendor segments are library code the classifier recognised; they are stored but not indexed. Bytes are shown as Latin1 characters, one per byte.