1import { opt } from './containers/app.js'; 2import { scribeDocDefaults } from './containers/scribeDocDefaults.js'; 3import { loadBuiltInFontsRaw, loadChiSimFont } from './fontContainerMain.js'; 4import { calcCharMetricsFromPages } from './fontStatistics.js'; 5import { gs } from './generalWorkerMain.js'; 6import { ImageWrapper } from './objects/imageObjects.js'; 7import { LayoutDataTablePage, LayoutPage, addCircularRefsDataTables } from './objects/layoutObjects.js'; 8import ocr, { addCircularRefsOcr, OcrPage } from './objects/ocrObjects.js'; 9import { PageMetrics } from './objects/pageMetricsObjects.js'; 10import { selectOcrPages } from './pdf/ocrPageSelection.js'; 11import { clearObjectProperties } from './utils/miscUtils.js'; 12 13/** @typedef {import('./containers/scribeDoc.js').ScribeDoc} ScribeDoc */ 14 15// Data-derived thresholds for the post-OCR keep/discard gate (ocrAddsNewText, below). 16const OCR_NEW_CONF_MIN = 85; // min word confidence (0-100) to count an OCR word as legitimate 17const OCR_NEW_LINE_WORDS = 3; // new word-like words in one OCR line => a coherent new-text line 18const OCR_NEW_LINES_MIN = 2; // coherent new-text lines that keep the OCR 19const OCR_NEW_NUMS_MIN = 10; // new multi-digit numbers that keep the OCR (a baked data table) 20const OCR_NEW_CHARS_MIN = 100; // new word-like characters that keep the OCR 21 22/** 23 * Decide whether a page's pure OCR adds legitimate text its native layer lacks (the keep/discard gate). 24 * An OCR word counts as "new" when its normalized token is not a substring of the native token stream. 25 * The check is substring membership, not equality, because native and OCR segment words differently. 26 * @param {OcrPage|null|undefined} nativePage - The native (PDF) text layer. 27 * A null layer (a scan or broken-encoding page) makes all OCR new, so the OCR is kept. 28 * @param {OcrPage} ocrPage - The page's pure OCR result. 29 * @returns {boolean} True to keep the OCR, false to discard it in favor of the native text. 30 */ 31export function ocrAddsNewText(nativePage, ocrPage) { 32 if (!nativePage) return true; 33 const normTok = (/** @type {string} */ text) => ocr.replaceLigatures(text) 34 .toLowerCase() 35 .normalize('NFKD') 36 .replace(/[Ì-ͯ]/g, '') 37 .replace(/[^a-z0-9]/g, ''); 38 const nativeStream = nativePage.lines.flatMap((l) => l.words).map((w) => normTok(w.text)).filter(Boolean).join(' '); 39 let newChars = 0; 40 let newNums = 0; 41 let newTextLines = 0; 42 for (const line of ocrPage.lines) { 43 let lineNewWords = 0; 44 for (const word of line.words) { 45 const tok = normTok(word.text); 46 if (tok.length < 2 || word.conf < OCR_NEW_CONF_MIN || nativeStream.includes(tok)) continue; 47 if (/^[a-z]{3,}$/.test(tok) && /[aeiouy]/.test(tok)) { 48 newChars += tok.length; 49 lineNewWords += 1; 50 } else if (/^[0-9]{2,}$/.test(tok)) { 51 newNums += 1; 52 } 53 } 54 if (lineNewWords >= OCR_NEW_LINE_WORDS) newTextLines += 1; 55 } 56 return newTextLines >= OCR_NEW_LINES_MIN || newNums >= OCR_NEW_NUMS_MIN || newChars >= OCR_NEW_CHARS_MIN; 57} 58 59/** 60 * Build the canonical 'Combined' layer for a partial page selection and point `active` at it. 61 * Per page it keeps the engine's OCR, or falls back to native (PDF) text when the keep/discard gate finds the OCR adds nothing the native layer lacks. 62 * Every page is cloned, so editing 'Combined' cannot corrupt the source layers it draws from. 63 * A no-op when OCR ran on every page, when the document carries user-uploaded OCR, or when no page was OCR'd. 64 * @param {ScribeDoc} doc 65 * @param {OcrPage[]} source - The engine's full-document OCR layer ('Tesseract Combined' or a custom model's). 66 * @param {boolean[]} ocrPageMask - Which pages were sent to OCR. 67 * @param {boolean} gateApplies - Whether the keep/discard gate runs (the `auto*` ocrPages modes only). 68 * @param {boolean} fullOcr - True when every page was OCR'd, in which case `active` already names the engine layer. 69 */ 70function buildCombinedLayer(doc, source, ocrPageMask, gateApplies, fullOcr) { 71 if (fullOcr || doc.ocr['User Upload'] || !ocrPageMask.some(Boolean)) return; 72 const native = doc.ocr.pdf; 73 // Relocate the pure Legacy+LSTM combine from 'Combined' to 'Tesseract Combined' (unless an existing-OCR 74 // run already put it there) so 'Combined' can hold the canonical result. 75 if (source === doc.ocr.Combined && !doc.ocr['Tesseract Combined']) doc.ocr['Tesseract Combined'] = source; 76 const combined = Array(doc.inputData.pageCount); 77 for (let i = 0;
77 i < combined.length; i++) { 78 const nat = native && native[i]; 79 const ocrPage = source[i]; 80 let chosen; 81 if (ocrPageMask[i] && ocrPage) { 82 chosen = (gateApplies && nat && !ocrAddsNewText(nat, ocrPage)) ? nat : ocrPage; 83 } else { 84 chosen = nat || ocrPage; 85 } 86 if (chosen) { 87 combined[i] = ocr.clonePage(chosen); 88 combined[i].angle = chosen.angle; 89 } else { 90 combined[i] = new OcrPage(i, doc.pageMetrics[i].dims); 91 } 92 } 93 doc.ocr.Combined = combined; 94 doc.ocr.active = doc.ocr.Combined; 95} 96 97/** 98 * Display warning/error message to user if missing character-level data. 99 * 100 * @param {ScribeDoc} doc 101 * @param {Array<Object.<string, string>>} warnArr - Array of objects containing warning/error messages from convertPage 102 */ 103export function checkCharWarn(doc, warnArr) { 104 // TODO: Figure out what happens if there is one blank page with no identified characters (as that would presumably trigger an error and/or warning on the page level). 105 // Make sure the program still works in that case for both Tesseract and Abbyy. 106 107 const charErrorCt = warnArr.filter((x) => x?.char === 'char_error').length; 108 const charWarnCt = warnArr.filter((x) => x?.char === 'char_warning').length; 109 const charGoodCt = warnArr.length - charErrorCt - charWarnCt; 110 111 // The UI warning/error messages cannot be thrown within this function, 112 // as that would make this file break when imported into contexts that do not have the main UI. 113 if (charGoodCt === 0 && charErrorCt > 0) { 114 if (typeof process === 'undefined') { 115 const errorHTML = `No character-level OCR data detected. Abbyy XML is only supported with character-level data. 116 <a href="https://docs.scribeocr.com/faq.html#is-character-level-ocr-data-required--why" target="_blank" class="alert-link">Learn more.</a>`; 117 doc.errorHandler({ message: errorHTML }); 118 } else { 119 const errorText = `No character-level OCR data detected. Abbyy XML is only supported with character-level data. 120 See: https://docs.scribeocr.com/faq.html#is-character-level-ocr-data-required--why`; 121 doc.errorHandler({ message: errorText }); 122 } 123 } if (charGoodCt === 0 && charWarnCt > 0 && typeof process === 'undefined') { 124 const warningHTML = `No character-level OCR data detected. Font optimization features will be disabled. 125 <a href="https://docs.scribeocr.com/faq.html#is-character-level-ocr-data-required--why" target="_blank" class="alert-link">Learn more.</a>`; 126 doc.warningHandler({ message: warningHTML }); 127 } 128} 129 130/** 131 * Sum up evaluation statistics for all pages. 132 * @param {Array<EvalMetrics>} evalStatsArr 133 */ 134export const calcEvalStatsDoc = (evalStatsArr) => { 135 const evalStatsDoc = { 136 total: 0, 137 correct: 0, 138 incorrect: 0, 139 missed: 0, 140 extra: 0, 141 correctLowConf: 0, 142 incorrectHighConf: 0, 143 }; 144 145 for (let i = 0; i < evalStatsArr.length; i++) { 146 evalStatsDoc.total += evalStatsArr[i].total; 147 evalStatsDoc.correct += evalStatsArr[i].correct; 148 evalStatsDoc.incorrect += evalStatsArr[i].incorrect; 149 evalStatsDoc.missed += evalStatsArr[i].missed; 150 evalStatsDoc.extra += evalStatsArr[i].extra; 151 evalStatsDoc.correctLowConf += evalStatsArr[i].correctLowConf; 152 evalStatsDoc.incorrectHighConf += evalStatsArr[i].incorrectHighConf; 153 } 154 return evalStatsDoc; 155}; 156 157/** 158 * Throw an AbortError if the given signal has fired. 159 * Matches the shape of fetch()/streams APIs so callers can `if (e.name === 'AbortError')`. 160 * @param {AbortSignal} [signal] 161 */ 162const throwIfAborted = (signal) => { 163 if (!signal || !signal.aborted) return; 164 const reason = signal.reason instanceof Error ? signal.reason : undefined; 165 if (typeof DOMException !== 'undefined') { 166 throw new DOMException(reason ? reason.message : 'Recognition aborted', 'AbortError'); 167 } 168 const err = new Error(reason ? reason.message : 'Recognition aborted'); 169 err.name = 'AbortError'; 170 throw err; 171}; 172 173/** 174 * setTimeout that resolves early when `signal` fires. Never rejects. 175 * @param {number} ms 176 * @param {AbortSignal} [signal] 177 */ 178const abortableDelay = (ms, signal) => new Promise((resolve) => { 179 if (signal && signal.aborted) { resolve(); return; } 180 const onAbort = () => { 181 clearTimeout(t); 182 if (signal) signal.removeEventListener('abort', onAbort); 183 resolve(); 184 }; 185 const t = setTimeout(() => { 186 if (signal) signal.removeEventListener('abort', onAbort); 187 resolve(); 188 }, ms); 189 if (signal) signal.addEventListener('abort', onAbort, { once: true }); 190}); 191 192/** 193 * @param {ScribeDoc} doc 194 * @param {Object} params 195 * @param {OcrPage | OcrLine} params.page 196 * @param {?function} [params.func=null] 197 * @param {boolean}
197 [params.view=false] - Draw results on debugging canvases 198 */ 199export async function evalOCRPage(doc, params) { 200 const n = 'page' in params.page ? params.page.page.n : params.page.n; 201 const binaryImage = await doc.images.getBinary(n); 202 const pageMetricsObj = doc.pageMetrics[n]; 203 return gs.evalPageBase({ 204 page: params.page, binaryImage, pageMetricsObj, func: params.func, view: params.view, docId: doc.id, 205 }); 206} 207 208/** 209 * Compare two sets of OCR data. 210 * @param {ScribeDoc} doc 211 * @param {Array<OcrPage>} ocrA 212 * @param {Array<OcrPage>} ocrB 213 * @param {Parameters<import('./worker/compareOCRModule.js').compareOCRPageImp>[0]['options']} [options] 214 * @param {?function} [progressCallback=null] 215 */ 216export async function compareOCR(doc, ocrA, ocrB, options, progressCallback = null) { 217 /** @type {Parameters<import('./worker/compareOCRModule.js').compareOCRPageImp>[0]['options']} */ 218 const compOptions = { 219 ignorePunct: scribeDocDefaults.ignorePunct, 220 ignoreCap: scribeDocDefaults.ignoreCap, 221 confThreshHigh: scribeDocDefaults.confThreshHigh, 222 confThreshMed: scribeDocDefaults.confThreshMed, 223 }; 224 225 if (options) Object.assign(compOptions, options); 226 227 /** @type {Array<OcrPage>} */ 228 const ocrArr = []; 229 /** @type {Array<?EvalMetrics>} */ 230 const metricsArr = []; 231 /** @type {Array<Array<CompDebugBrowser | CompDebugNode>>} */ 232 const debugImageArr = []; 233 234 const comparePageI = async (i) => { 235 const pageA = ocrA[i]; 236 const pageB = ocrB[i]; 237 // A per-page OCR subset leaves skipped pages empty, so skip any page missing from either input. 238 if (!pageA || !pageB) { 239 if (progressCallback) progressCallback(); 240 return; 241 } 242 // Some option combinations need the page image and some do not. 243 // Skip it when unneeded for performance, and so accuracy benchmarks can run without an image. 244 const mode = compOptions.mode || 'stats'; 245 const evalConflicts = compOptions.evalConflicts ?? true; 246 const supplementComp = compOptions.supplementComp ?? false; 247 const skipImage = (mode === 'stats' && !supplementComp) || (mode === 'comb' && !evalConflicts && !supplementComp); 248 const binaryImage = skipImage ? null : await doc.images.getBinary(pageA.n); 249 const res = await gs.compareOCRPageImp({ 250 pageA, 251 pageB, 252 binaryImage, 253 pageMetricsObj: doc.pageMetrics[pageA.n], 254 options: compOptions, 255 docId: doc.id, 256 }); 257 258 ocrArr[i] = res.page; 259 260 metricsArr[i] = res.metrics; 261 262 if (res.debugImg) debugImageArr[i] = res.debugImg; 263 if (progressCallback) progressCallback(); 264 }; 265 266 const indices = [...Array(ocrA.length).keys()]; 267 const compPromises = indices.map(async (i) => comparePageI(i)); 268 await Promise.allSettled(compPromises); 269 270 return { ocr: ocrArr, metrics: metricsArr, debug: debugImageArr }; 271} 272 273/** 274 * Calculate what arguments to use with Tesseract `recognize` function relating to rotation. 275 * @param {ScribeDoc} doc 276 * @param {number} n - Page number to recognize. 277 * @param {boolean} areaMode 278 */ 279async function calcRecognizeRotateArgs(doc, n, areaMode) { 280 // Whether the binary image should be rotated internally by Tesseract 281 // This should always be true (Tesseract results are horrible without auto-rotate) but kept as a variable for debugging purposes. 282 const rotate = true; 283 284 // Whether the rotated images should be saved, overwriting any non-rotated images. 285 const autoRotate = true; 286 287 // Threshold (in radians) under which page angle is considered to be effectively 0. 288 const angleThresh = 0.0008726646; 289 290 const angle = doc.pageMetrics[n]?.angle; 291 292 // Whether the page angle is already known (or needs to be detected) 293 const angleKnown = typeof (angle) === 'number'; 294 295 const nativeN = await doc.images.getNative(n); 296 297 // Calculate additional rotation to apply to page. Rotation should not be applied if page has already been rotated. 298 const rotateDegrees = rotate && angle && Math.abs(angle || 0) > 0.05 && !nativeN.rotated ? angle * -1 : 0; 299 const rotateRadians = rotateDegrees * (Math.PI / 180); 300 301 let saveNativeImage = false; 302 let saveBinaryImageArg = false;
303 304 // Images are not saved when using "recognize area" as these intermediate images are cropped. 305 if (!areaMode) { 306 const binaryN = await doc.images.binary[n]; 307 // Images are saved if either (1) we do not have any such image at present or (2) the current version is not rotated but the user has the "auto rotate" option enabled. 308 if (autoRotate && !nativeN.rotated[n] && (!angleKnown || Math.abs(rotateRadians) > angleThresh)) saveNativeImage = true; 309 if (!binaryN || autoRotate && !binaryN.rotated && (!angleKnown || Math.abs(rotateRadians) > angleThresh)) saveBinaryImageArg = true; 310 } 311 312 return { 313 angleThresh, 314 angleKnown, 315 rotateRadians, 316 saveNativeImage, 317 saveBinaryImageArg, 318 }; 319} 320 321/** 322 * Lower-level function to run OCR for a single page. 323 * Requires additional code to handle the results; for advanced users only. 324 * Most users should use `recognize` instead to recognize all pages in a document. 325 * 326 * @param {ScribeDoc} doc 327 * @param {number} n - Page number to recognize. 328 * @param {boolean} legacy - 329 * @param {boolean} lstm - 330 * @param {boolean} areaMode - 331 * @param {Object<string, string>} tessOptions - Options to pass to Tesseract.js. 332 * @param {boolean} [debugVis=false] - Generate instructions for debugging visualizations. 333 * @param {?Array<string>} [langs=null] - Languages for this job. When set, the worker ensures its 334 * engine matches before recognizing, so concurrent documents in different languages stay isolated. 335 * @param {boolean} [vanillaMode=false] - Use the vanilla Tesseract.js model. 336 */ 337export async function recognizePageImp(doc, n, legacy, lstm, areaMode, tessOptions = {}, debugVis = false, langs = null, vanillaMode = false) { 338 const { 339 angleThresh, angleKnown, rotateRadians, saveNativeImage, saveBinaryImageArg, 340 } = await calcRecognizeRotateArgs(doc, n, areaMode); 341 342 const nativeN = await doc.images.getNative(n); 343 344 if (!nativeN) throw new Error(`No image source found for page ${n}`); 345 346 const config = { 347 ...{ 348 rotateRadians, rotateAuto: !angleKnown, legacy, lstm, 349 }, 350 ...tessOptions, 351 }; 352 353 const pageDims = doc.pageMetrics[n].dims; 354 355 // If `legacy` and `lstm` are both `false`, recognition is not run, but layout analysis is. 356 // This combination of options would be set for debug mode, where the point of running Tesseract 357 // is to get debugging images for layout analysis rather than get text. 358 const runRecognition = legacy || lstm; 359 360 const resArr = await gs.recognizeAndConvert2({ 361 image: nativeN.src, 362 options: config, 363 output: { 364 // text, blocks, hocr, and tsv must all be `false` to disable recognition 365 text: runRecognition, 366 blocks: runRecognition, 367 hocr: runRecognition, 368 tsv: runRecognition, 369 layoutBlocks: !runRecognition, 370 imageBinary: saveBinaryImageArg, 371 imageColor: saveNativeImage, 372 debug: true, 373 debugVis, 374 }, 375 n, 376 knownAngle: doc.pageMetrics[n].angle, 377 pageDims, 378 langs, 379 vanillaMode, 380 }); 381 382 const res0 = await resArr[0]; 383 384 const elapsedSec = res0.recognitionTime / 1000; 385 if (scribeDocDefaults.printRecognitionTime === true || (typeof scribeDocDefaults.printRecognitionTime === 'number' && elapsedSec > scribeDocDefaults.printRecognitionTime)) { 386 console.log(`Page ${n} recognition time: ${elapsedSec.toFixed(2)}s`); 387 } 388 389 if (!angleKnown) doc.pageMetrics[n].angle = (res0.recognize.rotateRadians || 0) * (180 / Math.PI) * -1; 390 391 // An image is rotated if either the source was rotated or rotation was applied by Tesseract. 392 const isRotated = Boolean(res0.recognize.rotateRadians || 0) || nativeN.rotated; 393 394 // Images from Tesseract should not overwrite the existing images in the case where rotateAuto is true, 395 // but no significant rotation was actually detected. 396 const significantRotation = Math.abs(res0.recognize.rotateRadians || 0) > angleThresh; 397 398 const upscale = res0.recognize.upscale || false; 399 if (saveBinaryImageArg && res0.recognize.imageBinary && (significantRotation || !doc.images.binary[n])) { 400 doc.images.binaryProps[n] = { rotated: isRotated, upscaled: upscale, colorMode: 'binary' }; 401 doc.images.binary[n] = new ImageWrapper(n, res0.recognize.imageBinary, 'binary', isRotated, upscale); 402 } 403 404 if (saveNativeImage && res0.recognize.imageColor && significantRotation) { 405 doc.images.nativeProps[n] = { rotated: isRotated, upscaled: upscale, colorMode: scribeDocDefaults.colorMode }; 406 doc.images.native[n] = new ImageWrapper(n, res0.recognize.imageColor, 'native', isRotated, upscale); 407 } 408 409 return resArr; 410} 411 412/** 413 * Convert from raw OCR data to the internal hocr format used here 414 * Currently supports .hocr (used by Tesseract), Abbyy .xml, and stext (an intermediate data format used by mupdf). 415 * 416 * @param {string} ocrRaw - String containing raw OCR data for single page. 417 * @param {number} n - Page number 418 * @param {TextSource} format - Format of raw data. 419 * @param {boolean} [scribeMode=false] - Whether this is HOCR data from this program. 420 * @returns {Promise<Awaited<ReturnType<typeof import('./worker/generalWorker.js').recognizeAndConvert>>['convert']>} 421 */ 422async function convertOCRPage(ocrRaw, n, format, scribeMode = false) { 423 await gs.getGeneralScheduler(); 424 let res; 425 if (format === 'hocr') { 426 res = await gs.convertPageHocr({ ocrStr: ocrRaw, n, scribeMode }); 427 } else if (format === 'abbyy') { 428 res = await gs.convertPageAbbyy({ ocrStr: ocrRaw, n }); 429 } else if (format === 'alto') { 430 res = await gs.convertPageAlto({ ocrStr: ocrRaw, n }); 431 } else if (format === 'textract') { 432 // res = await gs.convertPageTextract({ ocrStr: ocrRaw, n }); 433 } else if (format === 'azure_doc_intel') { 434 // res = await gs.convertDocAzureDocIntel({ ocrStr: ocrRaw, }); 435 } else if (format === 'google_doc_ai') { 436 // Document-level format, handled in convertOCR 437 } else if (format === 'google_vision') { 438 res = await gs.convertPageGoogleVision({ ocrStr: ocrRaw, n }); 439 } else if (format === 'stext') { 440 res = await gs.convertPageStext({ ocrStr: ocrRaw, n }); 441 } else if (format === 'text') { 442 res = await gs.convertPageText({ textStr: ocrRaw }); 443 } else if (format === 'docx') { 444 console.error('format does not support page-level import.'); 445 // res = await gs.convertDocDocx({ docxData: ocrRaw }); 446 } else { 447 throw new Error(`Invalid format: ${format}`); 448 } 449 450 return res; 451} 452 453/** 454 * Install a parsed `OcrPage` into the doc at index `n`. 455 * 456 * @param {ScribeDoc} doc 457 * @param {number} n 458 * @param {OcrPage} page 459 * @param {object} options 460 * @param {string} options.engineName - Name of the OCR engine this page came from. 461 * @param {LayoutDataTablePage} [options.dataTables] - Per-page layout tables. 462 * @param {Object<string,string>} [options.warn] - Per-page conversion warning. 463 * @param {boolean} [options.mainData] - When true, this page's data drives pageMetrics and convertPageWarn for index `n`. Default true. 464 * Set false when inserting a secondary engine's results into a doc that already has a primary OCR pass. 465 * @param {boolean} [options.setActive] - When true, point `doc.ocr.active` at this engine's array. Default true. 466 * Set false to leave `doc.ocr.active` unchanged (used by recognition's per-page callback, which assigns active itself at the end of the run). 467 */ 468export function insertParsedPage(doc, n, page, { 469 engineName, dataTables, warn = {}, mainData = true, setActive = true, 470}) { 471 addCircularRefsOcr([page]); 472 473 if (!doc.ocr[engineName]) doc.ocr[engineName] = Array(doc.inputData.pageCount); 474 doc.ocr[engineName][n] = page; 475 if (setActive) doc.ocr.active = doc.ocr[engineName]; 476 477 if (mainData) { 478 doc.convertPageWarn[n] = warn; 479 if (page.dims && page.dims.height && page.dims.width) doc.pageMetrics[n] = new PageMetrics(page.dims); 480 doc.pageMetrics[n].angle = page.angle; 481 } 482 483 doc.inputData.xmlMode[n] = true; 484 485 if (dataTables && Object.keys(doc.layoutDataTables.pages[n].tables).length === 0) { 486 addCircularRefsDataTables([dataTables]); 487 doc.layoutDataTables.pages[n] = dataTables; 488 } 489 490 doc.progressHandler({ n, type: 'convert', info: { engineName } }); 491} 492 493/** 494 * This function is called after running a `convertPage` (or `recognizeAndConvert`) function, updating this document with the results. 495 * This needs to be a separate function from `convertOCRPage`, given that sometimes recognition and conversion are combined by using `recognizeAndConvert`. 496 * 497 * @param {ScribeDoc} doc 498 * @param {Awaited<ReturnType<typeof import('./worker/generalWorker.js').recognizeAndConvert>>['convert']} params 499 * @param {number} n 500 * @param {boolean} mainData 501 * @param {string} engineName - Name of OCR engine. 502 */ 503async function convertPageCallback(doc, { 504 pageObj, dataTables, warn, langSet, 505}, n, mainData, engineName) { 506 const fontPromiseArr = []; 507 if (langSet && langSet.has('chi_sim')) fontPromiseArr.push(loadChiSimFont()); 508 if (langSet && (langSet.has('rus') || langSet.has('ukr') || langSet.has('ell'))) { 509 fontPromiseArr.push(loadBuiltInFontsRaw('all')); 510 } else { 511 fontPromiseArr.push(loadBuiltInFontsRaw()); 512 } 513 await Promise.all(fontPromiseArr); 514 515 if (['Tesseract Legacy', 'Tesseract LSTM'].includes(engineName)) doc.ocr['Tesseract Latest'][n] = pageObj; 516 517 insertParsedPage(doc, n, pageObj, { 518 engineName, dataTables, warn, mainData, setActive: false, 519 }); 520} 521 522/** 523 * Convert from raw OCR data to the internal hocr format used here 524 * Currently supports .hocr (used by Tesseract), Abbyy .xml, and stext (an intermediate data format used by mupdf). 525 * 526 * @param {ScribeDoc} doc 527 * @param {string[]} ocrRawArr - Array with raw OCR data, with an element for each page 528 * @param {boolean} mainData - Whether this is the "main" data that document metrics are calculated from.
529 * For imports of user-provided data, the first data provided should be flagged as the "main" data. 530 * For Tesseract.js recognition, the Tesseract Legacy results should be flagged as the "main" data. 531 * @param {TextSource} format - Format of raw data. 532 * @param {string} engineName - Name of OCR engine. 533 * @param {boolean} [scribeMode=false] - Whether this is HOCR data from this program. 534 * @param {?PageMetrics[]} [pageMetrics=null] - Page metrics to use for the pages (Textract only). 535 * @param {Object} [options] 536 * @param {'width' | 'sentence'} [options.docxLineSplitMode] - DOCX line-split mode. 537 * Defaults to `scribeDocDefaults.docxLineSplitMode`. Ignored for non-docx formats. 538 */ 539export async function convertOCR(doc, ocrRawArr, mainData, format, engineName, scribeMode, pageMetrics = null, options = {}) { 540 const docxLineSplitMode = options.docxLineSplitMode ?? scribeDocDefaults.docxLineSplitMode; 541 const promiseArr = []; 542 if (format === 'textract') { 543 if (!pageMetrics || !pageMetrics[0]?.dims) throw new Error('Page metrics must be provided for Textract data.'); 544 const pageDims = pageMetrics.map((metrics) => (metrics.dims)); 545 546 // When multiple Textract entries exist (per-page files), each file contains 547 // blocks with Page=1. Process each individually with the correct pageNum 548 // to avoid merging all pages into page 0. 549 if (ocrRawArr.length > 1) { 550 for (let i = 0; i < ocrRawArr.length; i++) { 551 const res = await gs.convertDocTextract({ ocrStr: [ocrRawArr[i]], pageDims: [pageDims[i]], pageNum: i }); 552 if (res.length > 0) { 553 await convertPageCallback(doc, res[0], i, mainData, engineName); 554 } 555 } 556 } else { 557 const res = await gs.convertDocTextract({ ocrStr: ocrRawArr, pageDims }); 558 for (let n = 0; n < res.length; n++) { 559 await convertPageCallback(doc, res[n], n, mainData, engineName); 560 } 561 } 562 return; 563 } 564 565 if (format === 'azure_doc_intel') { 566 if (!pageMetrics || !pageMetrics[0]?.dims) throw new Error('Page metrics must be provided for Azure Document Intelligence data.'); 567 const pageDims = pageMetrics.map((metrics) => (metrics.dims)); 568 const res = await gs.convertDocAzureDocIntel({ ocrStr: ocrRawArr, pageDims }); 569 for (let n = 0; n < res.length; n++) { 570 await convertPageCallback(doc, res[n], n, mainData, engineName); 571 } 572 return; 573 } 574 575 if (format === 'google_doc_ai') { 576 if (!pageMetrics || !pageMetrics[0]?.dims) throw new Error('Page metrics must be provided for Google Document AI data.'); 577 const pageDims = pageMetrics.map((metrics) => (metrics.dims)); 578 const res = await gs.convertDocGoogleDocAI({ ocrStr: ocrRawArr, pageDims }); 579 for (let n = 0; n < res.length; n++) { 580 await convertPageCallback(doc, res[n], n, mainData, engineName); 581 } 582 return; 583 } 584 585 if (format === 'google_vision' && pageMetrics && pageMetrics[0]?.dims) { 586 for (let n = 0; n < ocrRawArr.length; n++) { 587 const res = await gs.convertPageGoogleVision({ ocrStr: ocrRawArr[n], n, pageDims: pageMetrics[n].dims }); 588 await convertPageCallback(doc, res, n, mainData, engineName); 589 } 590 return; 591 } 592 593 if (format === 'text') { 594 const res = await gs.convertPageText({ textStr: ocrRawArr[0] }); 595 596 if (res.length > doc.inputData.pageCount) doc.inputData.pageCount = res.length; 597 598 for (let i = 0; i < res.length; i++) { 599 if (!doc.layoutRegions.pages[i]) doc.layoutRegions.pages[i] = new LayoutPage(i); 600 } 601 602 for (let i = 0; i < res.length; i++) { 603 if (!doc.layoutDataTables.pages[i]) doc.layoutDataTables.pages[i] = new LayoutDataTablePage(i); 604 } 605 606 for (let n = 0; n < res.length; n++) { 607 await convertPageCallback(doc, res[n], n, mainData, engineName); 608 } 609 return; 610 } 611 612 if (format === 'docx') { 613 const res = await gs.convertDocDocx({ docxData: ocrRawArr[0], lineSplitMode: docxLineSplitMode, docId: doc.id }); 614 615 if (res.length > doc.inputData.pageCount) doc.inputData.pageCount = res.length; 616 617 for (let i = 0; i < res.length; i++) { 618 if (!doc.layoutRegions.pages[i]) doc.layoutRegions.pages[i] = new L
618ayoutPage(i); 619 } 620 621 for (let i = 0; i < res.length; i++) { 622 if (!doc.layoutDataTables.pages[i]) doc.layoutDataTables.pages[i] = new LayoutDataTablePage(i); 623 } 624 625 for (let n = 0; n < res.length; n++) { 626 await convertPageCallback(doc, res[n], n, mainData, engineName); 627 } 628 return; 629 } 630 631 for (let n = 0; n < ocrRawArr.length; n++) { 632 promiseArr.push(convertOCRPage(ocrRawArr[n], n, format, scribeMode) 633 .then((res) => convertPageCallback(doc, res, n, mainData, engineName))); 634 } 635 await Promise.all(promiseArr); 636} 637 638/** 639 * @param {ScribeDoc} doc 640 * @param {boolean} legacy 641 * @param {boolean} lstm 642 * @param {boolean} mainData 643 * @param {Array<string>} [langs=['eng']] 644 * @param {boolean} [vanillaMode=false] 645 * @param {Object<string, string>} [config={}] 646 * @param {?boolean[]} [ocrPageMask=null] - Per-page mask. When set, only `true` pages are recognized. 647 */ 648async function recognizeAllPages(doc, legacy = true, lstm = true, mainData = false, langs = ['eng'], vanillaMode = false, config = {}, ocrPageMask = null) { 649 // Render all PDF pages to PNG if needed 650 // This step should not create binarized images as they will be created by Tesseract during recognition. 651 if (doc.inputData.pdfMode) await doc.images.preRenderRange({ min: 0, max: doc.images.pageCount - 1, binary: false }); 652 653 if (legacy) { 654 const oemText = 'Tesseract Legacy'; 655 if (!doc.ocr[oemText]) doc.ocr[oemText] = Array(doc.inputData.pageCount); 656 doc.ocr.active = doc.ocr[oemText]; 657 } 658 659 if (lstm) { 660 const oemText = 'Tesseract LSTM'; 661 if (!doc.ocr[oemText]) doc.ocr[oemText] = Array(doc.inputData.pageCount); 662 doc.ocr.active = doc.ocr[oemText]; 663 } 664 665 // 'Tesseract Latest' includes the last version of Tesseract to run. 666 // It exists only so that data can be consistently displayed during recognition, 667 // should never be enabled after recognition is complete, and should never be editable by the user. 668 { 669 const oemText = 'Tesseract Latest'; 670 if (!doc.ocr[oemText]) doc.ocr[oemText] = Array(doc.inputData.pageCount); 671 doc.ocr.active = doc.ocr[oemText]; 672 } 673 674 await gs.initTesseract({ 675 anyOk: false, vanillaMode, langs, config, 676 }); 677 678 // If Legacy and LSTM are both requested, LSTM completion is tracked by a second array of promises (`promisesB`). 679 // In this case, `convertPageCallbackBrowser` can be run after the Legacy recognition is finished, 680 // however this function only returns after all recognition is completed. 681 // This provides no performance benefit in absolute terms, however halves the amount of time the user has to wait 682 // before seeing the initial recognition results. 683 // When a per-page selection mask is supplied, recognize only the selected pages. 684 // `resolvesA`/`resolvesB` are keyed by page index (sparse) so each result lands in its correct slot, 685 // while `promisesA`/`promisesB` stay dense for `Promise.all`. 686 const inputPages = ocrPageMask ? [...Array(doc.images.pageCount).keys()].filter((i) => ocrPageMask[i]) : [...Array(doc.images.pageCount).keys()]; 687 const promisesA = []; 688 const resolvesA = []; 689 const promisesB = []; 690 const resolvesB = []; 691 692 for (const x of inputPages) { 693 promisesA.push(new Promise((resolve, reject) => { 694 resolvesA[x] = { resolve, reject }; 695 })); 696 promisesB.push(new Promise((resolve, reject) => { 697 resolvesB[x] = { resolve, reject }; 698 })); 699 } 700 701 // Upscaling is enabled only for image data, and only if the user has explicitly enabled it. 702 // For PDF data, if upscaling is desired, that should be handled by rendering the PDF at a higher resolution. 703 const upscale = doc.inputData.imageMode && scribeDocDefaults.enableUpscale; 704 705 const configPage = { upscale }; 706 707 for (const x of inputPages) { 708 recognizePageImp(doc, x, legacy, lstm, false, configPage, scribeDocDefaults.debugVis, langs, vanillaMode).then(async (resArr) => { 709 const res0 = await resArr[0]; 710 711 if (res0.recognize.debugVis) { 712 const { ScrollView } = await import('../scrollview-web/scrollview/ScrollView.js'); 713 const sv = new ScrollView({ 714 lightTheme: true, 715 }); 716 await sv.processVisStr(res0.recognize.debugVis); 717 doc.vis[x] = await sv.getAll(true); 718 } 719 720 if (legacy) { 721 await convertPageCallback(doc, res0.convert.legacy, x, mainData, 'Tesseract Legacy'); 722 resolvesA[x].resolve(); 723 } else if (lstm) { 724 await convertPageCallback(doc, res0.convert.lstm, x, false, 'Tesseract LSTM'); 725 resolvesA[x].resolve(); 726 } 727 728 if (legacy && lstm) { 729 (async () => { 730 const res1 = await resArr[1]; 731 await convertPageCallback(doc, res1.convert.lstm, x, false, 'Tesseract LSTM'); 732 resolvesB[x].resolve(); 733 })(); 734 } 735 }); 736 } 737 738 await Promise.all(promisesA); 739 740 if (mainData) { 741 await checkCharWarn(doc, doc.convertPageWarn); 742 } 743 744 if (legacy && lstm) await Promise.all(promisesB); 745 746 if (lstm) { 747 const oemText = 'Tesseract LSTM'; 748 doc.ocr.active = doc.ocr[oemText]; 749 } else { 750 const oemText = 'Tesseract Legacy'; 751 doc.ocr.active = doc.ocr[oemText]; 752 } 753} 754 755/** 756 * Convert a page of raw model output (HOCR string, Textract JSON string, etc.) into 757 * the internal OcrPage model and store it under the given engine name. Shared between 758 * the per-image and document-mode custom recognition paths. 759 * @param {ScribeDoc} doc 760 * @param {string} rawData 761 * @param {number} n - Page index. 762 * @param {RecognitionModel} model 763 */ 764async function convertModelRawPage(doc, rawData, n, model) { 765 const engineName = model.config.name; 766 const outputFormat = model.config.outputFormat; 767 if (model.convertPage) { 768 const convertResult = await model.convertPage(rawData, n); 769 await convertPageCallback(doc, convertResult, n, true, engineName); 770 } else if (outputFormat === 'textract') { 771 const pageDims = [doc.pageMetrics[n].dims]; 772 const res = await gs.convertDocTextract({ ocrStr: rawData, pageDims, pageNum: n }); 773 for (let i = 0; i < res.length; i++) { 774 await convertPageCallback(doc, res[i], n + i, true, engineName); 775 } 776 } else if (outputFormat === 'azure_doc_intel') { 777 const pageDims = [doc.pageMetrics[n].dims]; 778 const res = await gs.convertDocAzureDocIntel({ ocrStr: rawData, pageDims }); 779 for (let i = 0; i < res.length; i++) { 780 await convertPageCallback(doc, res[i], n + i, true, engineName); 781 } 782 } else if (outputFormat === 'google_doc_ai') { 783 const pageDims = [doc.pageMetrics[n].dims]; 784 const res = await gs.convertDocGoogleDocAI({ ocrStr: rawData, pageDims, pageNum: n }); 785 for (let i = 0; i < res.length; i++) { 786 await convertPageCallback(doc, res[i], n + i, true, engineName); 787 } 788 } else if (outputFormat === 'google_vision') { 789 const res = await gs.convertPageGoogleVision({ ocrStr: rawData, n, pageDims: doc.pageMetrics[n].dims }); 790 await convertPageCallback(doc, res, n, true, engineName); 791 } else { 792 const res = await convertOCRPage(rawData, n, /** @type {TextSource} */ (outputFormat)); 793 await convertPageCallback(doc, res, n, true, engineName); 794 } 795} 796 797/** 798 * Document-mode recognition path: the model consumes the whole PDF at once (e.g. a 799 * server proxy that renders pages and runs OCR remotely) and streams back per-page raw 800 * results. Skips browser-side pre-rendering and the per-image dispatch loop entirely. 801 * @param {ScribeDoc} doc 802 * @param {Object} options 803 * @param {RecognitionModel} options.model 804 * @param {Object} [options.modelOptions] 805 * @param {AbortSignal} [options.signal] 806 * @param {?boolean[]} [ocrPageMask=null] - Per-page mask from `recognize`. 807 * The whole-PDF API cannot select individual pages, so the mask is applied at document granularity. 808 * No page selected skips OCR entirely. Any other selection OCRs the whole PDF, with a warning when the selection was partial. 809 */ 810async function recognizeCustomModelDocumentMode(doc, options, ocrPageMask = null) { 811 const model = options.model; 812 const modelOptions = options.modelOptions || {}; 813 const signal = options.signal; 814 const engineName = model.config.name; 815 // modelOptions passed to the model has `signal` merged in so network-backed models 816 // can cancel their in-flight HTTP requests. We keep the caller's modelOptions object 817 // untouched to avoid mutating a user-supplied reference. 818 const modelOptionsWithSignal = { ...modelOptions, signal }; 819 820 if (!doc.ocr[engineName]) doc.ocr[engineName] = Array(doc.inputData.pageCount); 821 if (scribeDocDefaults.keepRawData && !doc.ocrRaw[engineName]) doc.ocrRaw[engineName] = Array(doc.inputData.pageCount); 822 823 if (ocrPageMask && !ocrPageMask.some(Boolean)) { 824 for (let n = 0; n < doc.inputData.pageCount; n++) { 825 doc.ocr[engineName][n] = (doc.ocr.pdf && doc.ocr.pdf[n]) || new OcrPage(n, doc.pageMetrics[n].dims); 826 } 827 doc.ocr.active = doc.ocr[engineName]; 828 return doc.ocr.active; 829 } 830 if (ocrPageMask && !ocrPageMask.every(Boolean)) {
831 const ocrCount = ocrPageMask.filter(Boolean).length; 832 doc.warningHandler({ message: `Document-mode recognition OCRs the whole PDF; per-page OCR selection (${ocrCount}/${doc.inputData.pageCount} selected) cannot be applied and all pages will be sent.` }); 833 } 834 // The whole PDF is OCR'd in document mode, so every page's active layer becomes OCR. 835 doc.inputData.ocrApplied = Array(doc.inputData.pageCount).fill(true); 836 837 throwIfAborted(signal); 838 839 const pageDims = doc.pageMetrics.map((m) => m.dims); 840 const pdfBytes = doc.inputData.pdfMode && doc.images.pdfData ? new Uint8Array(doc.images.pdfData) : null; 841 842 const stream = await model.recognizeDocument( 843 { pdfBytes, pageCount: doc.inputData.pageCount, pageDims }, 844 modelOptionsWithSignal, 845 ); 846 847 const failedPagesDoc = []; 848 let lastErrMsg = ''; 849 let docAborted = false; 850 try { 851 for await (const entry of stream) { 852 if (signal && signal.aborted) { docAborted = true; break; } 853 if (!entry) continue; 854 if (entry.error) { 855 const errMsg = entry.error.message || String(entry.error); 856 failedPagesDoc.push(entry.pageNum); 857 lastErrMsg = errMsg; 858 doc.warningHandler({ message: `Recognition failed for page ${entry.pageNum}: ${errMsg}`, page: entry.pageNum }); 859 doc.ocr[engineName][entry.pageNum] = new OcrPage(entry.pageNum, doc.pageMetrics[entry.pageNum].dims); 860 continue; 861 } 862 const { pageNum, rawData } = entry; 863 if (scribeDocDefaults.keepRawData) doc.ocrRaw[engineName][pageNum] = rawData; 864 doc.progressHandler({ 865 n: pageNum, type: 'recognize', info: { status: 'received', engineName, timestamp: Date.now() }, 866 }); 867 await convertModelRawPage(doc, rawData, pageNum, model); 868 } 869 } finally { 870 // Best-effort: if the model's generator supports early termination via return(), 871 // give it a chance to clean up (e.g. cancel an in-flight fetch). Works for both 872 // native async generators (have .return) and hand-rolled async iterators. 873 if (docAborted && stream && typeof stream.return === 'function') { 874 try { await stream.return(); } catch (_) { /* ignore */ } 875 } 876 } 877 878 // Preserve partial results on abort â caller (and any resume layer) may want them. 879 doc.ocr.active = doc.ocr[engineName]; 880 if (scribeDocDefaults.keepRawData) doc.ocrRaw.active = doc.ocrRaw[engineName]; 881 882 // Always throw if the signal is aborted, regardless of whether the library or the 883 // model's own early-return terminated the for-await loop first. 884 throwIfAborted(signal); 885 886 if (failedPagesDoc.length === doc.inputData.pageCount) { 887 throw new Error(`Recognition failed for all pages. Last error message: ${lastErrMsg}`); 888 } 889 if (failedPagesDoc.length > 0) { 890 failedPagesDoc.sort((a, b) => a - b); 891 doc.warningHandler({ message: `Recognition failed for ${failedPagesDoc.length} page(s) (${failedPagesDoc.join(', ')}). These pages will have no OCR data.` }); 892 } 893 894 return doc.ocr.active; 895} 896 897/** 898 * Recognize all pages using a custom (external) recognition model. 899 * Called by `recognize` when `options.model` is provided. 900 * 901 * @param {ScribeDoc} doc 902 * @param {Object} options - Options object from `recognize`, guaranteed non-null with `model` set. 903 * @param {RecognitionModel} options.model 904 * @param {Object} [options.modelOptions] 905 * @param {Array<string>} [options.langs] 906 * @param {AbortSignal} [options.signal] - Optional abort signal. 907 * When aborted, recognition stops scheduling new pages, drains in-flight work, 908 * preserves whatever pages completed, and throws an AbortError. 909 * @param {?boolean[]} [ocrPageMask=null] - Per-page mask from `recognize`. 910 * When set, only `true` pages are sent to the model and skipped pages keep their native (PDF) text. 911 */ 912async function recognizeCustomModel(doc, options, ocrPageMask = null) { 913 const model = options.model; 914 const modelOptions = options.modelOptions || {}; 915 const signal = options.signal; 916 const engineName = model.config.name; 917 const outputFormat = model.config.outputFormat; 918 // modelOptions passed to the model has `signal` merged in so network-backed models 919 // can cancel their in-flight HTTP requests. We keep the caller's modelOptions object 920 // untouched to avoid mutating a user-supplied reference. 921 const modelOptionsWithSignal = { ...modelOptions, signal }; 922 923 const knownFormats = ['hocr', 'abbyy', 'alto', 'textract', 'azure_doc_intel', 'google_doc_ai', 'google_vision', 'stext', 'text']; 924 if (!knownFormats.includes(outputFormat) && !model.convertPage) { 925 throw new Error(`Model output format '${outputFormat}' is not supported. Provide a convertPage method on the model.`); 926 } 927 928 await gs.getGeneralScheduler(); 929 930 // Document-mode models OCR the whole PDF in a single call, so route them to their own path. 931 if (model.config.documentMode) return recognizeCustomModelDocumentMode(doc, options, ocrPageMask); 932 933 // Initialize array for custom model results 934 if (!doc.ocr[engineName]) doc.ocr[engineName] = Array(doc.inputData.pageCount); 935 if (scribeDocDefaults.keepRawData && !doc.ocrRaw[engineName]) doc.ocrRaw[engineName] = Array(doc.inputData.pageCount); 936 937 // No page selected: skip the model entirely, filling each page from its native (PDF) text. 938 if (ocrPageMask && !ocrPageMask.some(Boolean)) { 939 for (let n = 0; n < doc.inputData.pageCount; n++) { 940 doc.ocr[engineName][n] = (doc.ocr.pdf && doc.ocr.pdf[n]) || new OcrPage(n, doc.pageMetrics[n].dims); 941 } 942 doc.ocr.active = doc.ocr[engineName]; 943 return doc.ocr.active; 944 } 945 946 // Different cloud providers implement usage quotas in different ways. 947 // AWS Textract (Sync) uses transactions per second (TPS). 948 // AWS Textract (Async) uses both transactions per second (TPS) and concurrent request limits. 949 // Google Vision (Sync) uses requests per minute (RPM). 950 // The core distinction is that TPS limits the number of requests SENT per second, 951 // rather than the number of live requests at any given time. 952 const configRateLimit = modelOptions.rateLimit ?? model.config.rateLimit ?? null; 953 const regionCount = Array.isArray(modelOptions?.region) ? modelOptions.region.length : 1; 954 const baseTps = configRateLimit?.tps ?? (configRateLimit?.rpm ? configRateLimit.rpm / 60 : null); 955 const tps = baseTps != null ? baseTps * regionCount : null; 956 let adaptiveTps = tps; 957 let lastRequestTime = 0; 958 959 let concurrency; 960 if (modelOptions.maxConcurrency != null) { 961 concurrency = modelOptions.maxConcurrency; 962 } else if (tps != null) { 963 // When tps is set, that is the primary means of limiting concurrency. 964 // This is set to a large number as a safeguard. 965 concurrency = 30; 966 } else if (opt.workerN) { 967 concurrency = opt.workerN; 968 } else if (typeof process === 'undefined') { 969 concurrency = Math.min(Math.round((globalThis.navigator.hardwareConcurrency || 8) / 2), 6); 970 } else { 971 const cpuN = Math.floor((await import('node:os')).cpus().length / 2); 972 concurrency = Math.max(Math.min(cpuN - 1, 8), 1); 973 } 974 975 // Process all selected pages with limited concurrency. 976 // Skipped pages keep their native (PDF) text so `doc.ocr[engineName]` (the active layer below) has no holes. 977 const pages = [...Array(doc.images.pageCount).keys()].filter((n) => !ocrPageMask || ocrPageMask[n]); 978 if (ocrPageMask) { 979 for (let n = 0; n < doc.images.pageCount; n++) { 980 if (!ocrPageMask[n]) doc.ocr[engineName][n] = (doc.ocr.pdf && doc.ocr.pdf[n]) || new OcrPage(n, doc.pageMetrics[n].dims); 981 } 982 } 983 const executing = new Set(); 984 985 const maxConsecutiveFailures = 3; 986 let consecutiveFailures = 0; 987 let lastErrorMessage = ''; 988 let quitEarly = false;
989 /** @type {number[]} */ 990 const failedPages = []; 991 992 for (const n of pages) { 993 if (quitEarly) break; 994 if (signal && signal.aborted) break; 995 // eslint-disable-next-line no-loop-func 996 const p = (async () => { 997 if (quitEarly) return; 998 if (signal && signal.aborted) return; 999 1000 const nativeN = await doc.images.getNative(n); 1001 if (!nativeN) { 1002 doc.warningHandler({ message: `No image found for page ${n}, skipping.`, page: n }); 1003 doc.ocr[engineName][n] = new OcrPage(n, doc.pageMetrics[n].dims); 1004 return; 1005 } 1006 1007 // Convert base64 data URL to Uint8Array for the model 1008 const base64Data = nativeN.src.split(',')[1]; 1009 const binaryStr = atob(base64Data); 1010 const imageData = new Uint8Array(binaryStr.length); 1011 for (let i = 0; i < binaryStr.length; i++) { 1012 imageData[i] = binaryStr.charCodeAt(i); 1013 } 1014 1015 // Drop the cached render once its bytes are copied. 1016 // Without this every page's rendered image stays cached for the whole run, 1017 // and a large document exhausts the heap before recognition finishes. 1018 // TODO: This will delete images at the user's current position in the viewer. 1019 // Additionally, rendering 1k pages in the viewer will still cause a crash. 1020 // We should switch to a more robust system for clearing cache. 1021 // Gated to large documents for now since small docs are not a memory risk. 1022 if (doc.inputData.pdfMode && doc.inputData.pageCount > 100) delete doc.images.native[n]; 1023 1024 const maxThrottleRetries = 3; 1025 /** @type {RecognitionResult} */ 1026 let result = { success: false, format: '' }; 1027 1028 // Attempt recognition with up to maxThrottleRetries retries for throttling errors. 1029 // Attempt 0 is the initial request; attempts 1âmaxThrottleRetries are retries. 1030 for (let attempt = 0; attempt <= maxThrottleRetries; attempt++) { 1031 if (signal && signal.aborted) return; 1032 // TPS pacing: claim the next available dispatch slot before yielding. 1033 if (adaptiveTps != null && adaptiveTps > 0) { 1034 const now = Date.now(); 1035 // Using a number slightly above 1 second to account for variation. 1036 const minInterval = 1050 / adaptiveTps; 1037 const targetTime = Math.max(now, lastRequestTime + minInterval); 1038 lastRequestTime = targetTime; 1039 const waitMs = targetTime - now; 1040 if (waitMs > 0) { 1041 await abortableDelay(waitMs, signal); 1042 if (signal && signal.aborted) return; 1043 } 1044 } 1045 1046 doc.progressHandler({ n, type: 'recognize', info: { status: 'sending', engineName, timestamp: Date.now() } }); 1047 const recognizeStart = Date.now(); 1048 result = await model.recognizeImage(imageData, modelOptionsWithSignal); 1049 1050 if (result.success) { 1051 const elapsedSec = (Date.now() - recognizeStart) / 1000; 1052 if (scribeDocDefaults.printRecognitionTime === true || (typeof scribeDocDefaults.printRecognitionTime === 'number' && elapsedSec > scribeDocDefaults.printRecognitionTime)) { 1053 console.log(`Page ${n} recognition time: ${elapsedSec.toFixed(2)}s`); 1054 } 1055 break; 1056 } 1057 1058 // Only throttling errors are retried. 1059 const isThrottle = model.isThrottlingError && result.error && model.isThrottlingError(result.error); 1060 if (!isThrottle) break; 1061 1062 if (attempt === maxThrottleRetries) { 1063 doc.warningHandler({ message: `Page ${n}: throttled ${maxThrottleRetries + 1} times, giving up.`, page: n }); 1064 break; 1065 } 1066 const backoffMs = Math.min(1000 * (2 ** attempt), 16000); 1067 doc.warningHandler({ message: `Page ${n}: throttled by API, retrying in ${backoffMs}ms (attempt ${attempt + 1}/${maxThrottleRetries})`, page: n }); 1068 if (adaptiveTps != null && adaptiveTps > 0.5) { 1069 adaptiveTps *= 0.9; 1070 } 1071 await abortableDelay(backoffMs, signal); 1072 } 1073 1074 if (signal && signal.aborted) return; 1075 1076 if (!result.success || !result.rawData) { 1077 const errMsg = result.error ? result.error.message : 'Unknown error'; 1078 failedPages.push(n); 1079 doc.warningHandler({ message: `Recognition failed for page ${n}: ${errMsg}`, page: n }); 1080 doc.ocr[engineName][n] = new OcrPage(n, doc.pageMetrics[n].dims); 1081 consecutiveFailures++; 1082 lastErrorMessage = errMsg; 1083 if (consecutiveFailures >= maxConsecutiveFailures) { 1084 quitEarly = true; 1085 } 1086 return; 1087 } 1088 1089 consecutiveFailures = 0; 1090 1091 const rawData = result.rawData; 1092 if (scribeDocDefaults.keepRawData) doc.ocrRaw[engineName][n] = rawData; 1093 1094 await convertModelRawPage(doc, rawData, n, model); 1095 })().then(() => executing.delete(p)); 1096 1097 executing.add(p); 1098 if (executing.size >= concurrency) await Promise.race(executing); 1099 } 1100 1101 await Promise.allSettled(executing); 1102 1103 // On abort: preserve whatever pages completed and throw an AbortError. 1104 // A caller (e.g. the server proxy's resume-cache layer) may want the partial results. 1105 if (signal && signal.aborted) { 1106 doc.ocr.active = doc.ocr[engineName]; 1107 if (scribeDocDefaults.keepRawData) doc.ocrRaw.active = doc.ocrRaw[engineName]; 1108 throwIfAborted(signal); 1109 } 1110 1111 if (consecutiveFailures === doc.images.pageCount) { 1112 throw new Error( 1113 `Recognition failed for all pages. Last error message: ${lastErrorMessage}`, 1114 ); 1115 } 1116 1117 if (quitEarly) { 1118 throw new Error( 1119 `Recognition aborted after ${consecutiveFailures} consecutive failures. Last error message: ${lastErrorMessage}`, 1120 ); 1121 } 1122 1123 if (failedPages.length > 0) { 1124 failedPages.sort((a, b) => a - b); 1125 doc.warningHandler({ message: `Recognition failed for ${failedPages.length} page(s) (${failedPages.join(', ')}). These pages will have no OCR data.` }); 1126 } 1127 1128 // Set active OCR to custom model results 1129 doc.ocr.active = doc.ocr[engineName]; 1130 if (scribeDocDefaults.keepRawData) { 1131 doc.ocrRaw.active = doc.ocrRaw[engineName]; 1132 } 1133 return doc.ocr.active; 1134} 1135
1136/** 1137 * Recognize all pages in this document. 1138 * Files for recognition should already be imported using `importFiles` before calling this function. 1139 * The results of recognition can be exported by calling `exportData` after this function. 1140 * @param {ScribeDoc} doc 1141 * @param {Object} options 1142 * @param {'speed'|'quality'} [options.mode='quality'] - Recognition mode. 1143 * @param {Array<string>} [options.langs=['eng']] - Language(s) in document. 1144 * @param {'lstm'|'legacy'|'combined'} [options.modeAdv='combined'] - Alternative method of setting recognition mode. 1145 * @param {'conf'|'data'|'none'} [options.combineMode='data'] - Method of combining OCR results. Used if OCR data already exists. 1146 * @param {('all'|'auto'|'autoShallow'|'autoDeep'|'none'|boolean[])} [options.ocrPages] - Which pages to OCR. Defaults to `scribeDocDefaults.ocrPages` (`'all'`). 1147 * `'autoShallow'` decides per document: it skips a text-native document and OCRs an image-based one in full, 1148 * re-OCRing one that already carries an OCR layer unless `usePDFText.ocr.main` trusts that layer. 1149 * `'autoDeep'` (alias `'auto'`) is a strict superset of `'autoShallow'` that also OCRs the pages of an otherwise-skipped document that may hold baked-in text. 1150 * `'all'`/`'none'` force every/no page; a boolean array (length === page count) selects pages explicitly. 1151 * Image inputs, and any document with uploaded OCR, always OCR every page. 1152 * @param {typeof scribeDocDefaults.usePDFText} [options.usePDFText] - How to use a PDF's own extracted text, for this call. 1153 * Defaults to `scribeDocDefaults.usePDFText`. For a document with an existing OCR layer, `ocr.main: true` trusts that 1154 * layer as primary and skips OCR; `ocr.supp: true` merges it into a fresh OCR run; both false re-OCRs and discards it. 1155 * @param {boolean} [options.vanillaMode=false] - Whether to use the vanilla Tesseract.js model. 1156 * @param {Object<string, string>} [options.config={}] - Config params to pass to to Tesseract.js. 1157 * @param {RecognitionModel} [options.model] - Custom recognition model. See docs. 1158 * @param {Object} [options.modelOptions={}] - Options passed to the model's `recognizeImage` method. 1159 * @param {AbortSignal} [options.signal] - Optional abort signal for cancelling a custom-model 1160 * recognition run. When aborted, scribe.js stops scheduling new pages, drains any in-flight 1161 * page requests (so their network activity is not wasted), preserves the OCR data of pages 1162 * that already completed, and throws an AbortError. Only applies when `options.model` is set. 1163 */ 1164export async function recognize(doc, options = {}) { 1165 if (!doc.inputData.pdfMode && !doc.inputData.imageMode) throw new Error('No PDF or image data found to recognize.'); 1166 1167 // Decide which pages require OCR based on document contents and options specified. 1168 const ocrPages = options.ocrPages ?? scribeDocDefaults.ocrPages; 1169 const usePDFText = options.usePDFText ?? scribeDocDefaults.usePDFText; 1170 const pageCount = doc.inputData.pageCount; 1171 const stats = doc.inputData.pageStats; 1172 1173 /** @type {boolean[]} */ 1174 let ocrPageMask; 1175 if (Array.isArray(ocrPages) && !doc.ocr['User Upload']) { 1176 // An explicit per-page mask is used directly, independent of parse-time stats. 1177 if (ocrPages.length !== pageCount) { 1178 throw new Error(`ocrPages array length (${ocrPages.length}) must equal the page count (${pageCount}).`); 1179 } 1180 ocrPageMask = ocrPages.map(Boolean); 1181 } else if (doc.ocr['User Upload'] || !stats || stats.length !== pageCount) { 1182 // Uploaded OCR keeps the existing whole-document combine path (back-compat): OCR every page unless explicitly told `'none'`. 1183 // The same whole-document fallback applies when per-page stats are unavailable. 1184 ocrPageMask = Array(pageCount).fill(ocrPages !== 'none'); 1185 } else { 1186 ocrPageMask = selectOcrPages(stats, doc.inputData.pdfType, /** @type {'all'|'none'|'auto'|'autoShallow'|'autoDeep'} */ (ocrPages), usePDFText); 1187 } 1188 doc.inputData.ocrApplied = ocrPageMask.slice(); 1189 const fullOcr = ocrPageMask.every(Boolean); 1190 // The keep/discard gate runs only for the `auto*` ocrPages modes (not `all`, `none`, or an explicit mask). 1191 const gateApplies = ocrPages === 'autoDeep' || ocrPages === 'auto' || ocrPages === 'autoShallow'; 1192 1193 // Custom recognition model path 1194 if (options.model) { 1195 await recognizeCustomModel(doc, /** @type {{ model: RecognitionModel }} */ (options), ocrPageMask); 1196 buildCombinedLayer(doc, doc.ocr.active, ocrPageMask, gateApplies, fullOcr); 1197 return doc.ocr.active; 1198 } 1199 1200 if (!ocrPageMask.some(Boolean)) { 1201 // No page needs OCR: keep the parsed native/existing text layer as the active layer and skip recognition. 1202 if (doc.ocr.pdf) doc.ocr.active = doc.ocr.pdf; 1203 return doc.ocr.active; 1204 } 1205 1206 await gs.getGeneralScheduler(); 1207 1208 const combineMode = options && options.combineMode ? options.combineMode : 'data'; 1209 const vanillaMode = options && options.vanillaMode !== undefined ? options.vanillaMode : false;
1210 const config = options && options.config ? options.config : {}; 1211 1212 const langs = options && options.langs ? options.langs : ['eng']; 1213 let oemMode = 'combined'; 1214 if (options && options.modeAdv) { 1215 oemMode = options.modeAdv; 1216 } else if (options && options.mode) { 1217 oemMode = options.mode === 'speed' ? 'lstm' : 'legacy'; 1218 } 1219 1220 const fontPromiseArr = []; 1221 // Chinese requires loading a separate font. 1222 if (langs.includes('chi_sim')) fontPromiseArr.push(loadChiSimFont()); 1223 // Greek and Cyrillic require loading a version of the base fonts that include these characters. 1224 if (langs.includes('rus') || langs.includes('ukr') || langs.includes('ell')) fontPromiseArr.push(loadBuiltInFontsRaw('all')); 1225 await Promise.all(fontPromiseArr); 1226 1227 let forceMainData = false; 1228 let existingOCR; 1229 if (doc.ocr['User Upload']) { 1230 existingOCR = doc.ocr['User Upload']; 1231 } else if ( 1232 doc.ocr.pdf 1233 && ((doc.inputData.pdfType === 'text' && usePDFText.native.supp) 1234 || (doc.inputData.pdfType === 'ocr' && usePDFText.ocr.supp)) 1235 ) { 1236 existingOCR = doc.ocr.pdf; 1237 // If the PDF text is not the active data, it is assumed to be for supplemental purposes only. 1238 forceMainData = doc.ocr.pdf !== doc.ocr.active; 1239 } 1240 1241 // A single Tesseract engine can be used (Legacy or LSTM) or the results from both can be used and combined. 1242 if (oemMode === 'legacy' || oemMode === 'lstm') { 1243 // Tesseract is used as the "main" data unless user-uploaded data exists and only the LSTM model is being run. 1244 // This is because Tesseract Legacy provides very strong metrics, and Abbyy often does not. 1245 await recognizeAllPages(doc, oemMode === 'legacy', oemMode === 'lstm', !existingOCR, langs, vanillaMode, config, ocrPageMask); 1246 1247 // Metrics from the LSTM model are so inaccurate they are not worth using. 1248 if (oemMode === 'legacy') { 1249 const charMetrics = calcCharMetricsFromPages(doc.ocr['Tesseract Legacy']); 1250 if (Object.keys(charMetrics).length > 0) { 1251 clearObjectProperties(doc.fonts.state.charMetrics); 1252 Object.assign(doc.fonts.state.charMetrics, charMetrics); 1253 } 1254 await doc.runOptimization(doc.ocr['Tesseract Legacy']); 1255 } 1256 } else if (oemMode === 'combined') { 1257 await recognizeAllPages(doc, true, true, !existingOCR, langs, vanillaMode, config, ocrPageMask); 1258 1259 const progressCb = () => doc.progressHandler({ type: 'recognize' }); 1260 1261 if (scribeDocDefaults.saveDebugImages) { 1262 doc.debug.debugImg.Combined = new Array(doc.images.pageCount); 1263 for (let i = 0; i < doc.images.pageCount; i++) { 1264 doc.debug.debugImg.Combined[i] = []; 1265 } 1266 } 1267 1268 if (existingOCR) { 1269 const oemText = 'Tesseract Combined'; 1270 if (!doc.ocr[oemText]) doc.ocr[oemText] = Array(doc.inputData.pageCount); 1271 doc.ocr.active = doc.ocr[oemText]; 1272 1273 if (scribeDocDefaults.saveDebugImages) { 1274 doc.debug.debugImg['Tesseract Combined'] = new Array(doc.images.pageCount); 1275 for (let i = 0; i < doc.images.pageCount; i++) { 1276 doc.debug.debugImg['Tesseract Combined'][i] = []; 1277 } 1278 } 1279 } 1280 1281 // A new version of OCR data is created for font optimization and validation purposes. 1282 // This version has the bounding box and style data from the Legacy data, however uses the text from the LSTM data whenever conflicts occur. 1283 // Additionally, confidence is set to 0 when conflicts occur. Using this version benefits both font optimiztion and validation. 1284 // For optimization, using this version rather than Tesseract Legacy excludes data that conflicts with Tesseract LSTM and is therefore likely incorrect, 1285 // as low-confidence words are excluded when calculating overall character metrics. 1286 // For validation, this version is superior to both Legacy and LSTM, as it combines the more accurate bounding boxes/style data from Legacy 1287 // with the more accurate (on average) text data from LSTM. 1288 if (!doc.ocr['Tesseract Combined Temp']) doc.ocr['Tesseract Combined Temp'] = Array(doc.inputData.pageCount); 1289 1290 { 1291 /** @type {Parameters<typeof doc.compareOCR>[2]} */ 1292 const compOptions = { 1293 mode: 'comb', 1294 evalConflicts: false, 1295 legacyLSTMComb: true, 1296 }; 1297 1298 const res = await compareOCR(doc, doc.ocr['Tesseract Legacy'], doc.ocr['Tesseract LSTM'], compOptions, progressCb); 1299 1300 clearObjectProperties(doc.ocr['Tesseract Combined Temp']); 1301 Object.assign(doc.ocr['Tesseract Combined Temp'], res.ocr); 1302 } 1303 1304 // Evaluate default fonts using up to 5 pages. 1305 const pageNum = Math.min(doc.images.pageCount - 1, 5); 1306 await doc.images.preRenderRange({ min: 0, max: pageNum, binary: true }); 1307 const charMetrics = calcCharMetricsFromPages(doc.ocr['Tesseract Combined Temp']); 1308 if (Object.keys(charMetrics).length > 0) { 1309 clearObjectProperties(doc.fonts.state.charMetrics); 1310 Object.assign(doc.fonts.state.charMetrics, charMetrics); 1311 } 1312 await doc.runOptimization(doc.ocr['Tesseract Combined Temp']); 1313 1314 const oemText = 'Combined'; 1315 if (!doc.ocr[oemText]) doc.ocr[oemText] = Array(doc.inputData.pageCount); 1316 doc.ocr.active = doc.ocr[oemText]; 1317 1318 { 1319 const tessCombinedLabel = existingOCR ? 'Tesseract Combined' : 'Combined'; 1320 1321 /** @type {Parameters<import('./worker/compareOCRModule.js').compareOCRPageImp>[0]['options']} */ 1322 const compOptions = { 1323 mode: 'comb', 1324 debugLabel: scribeDocDefaults.saveDebugImages ? tessCombinedLabel : undefined, 1325 ignoreCap: scribeDocDefaults.ignoreCap, 1326 ignorePunct: scribeDocDefaults.ignorePunct, 1327 confThreshHigh: scribeDocDefaults.confThreshHigh, 1328 confThreshMed: scribeDocDefaults.confThreshMed, 1329 legacyLSTMComb: true, 1330 }; 1331 1332 const res = await compareOCR(doc, doc.ocr['Tesseract Legacy'], doc.ocr['Tesseract LSTM'], compOptions, progressCb); 1333 1334 if (doc.debug.debugImg[tessCombinedLabel]) doc.debug.debugImg[tessCombinedLabel] = res.debug; 1335 1336 clearObjectProperties(doc.ocr[tessCombinedLabel]); 1337 Object.assign(doc.ocr[tessCombinedLabel], res.ocr); 1338 } 1339 1340 // Compare the existing text layer against a secondary text layer word-by-word. 1341 // Runs for a whole-document OCR pass or for User-Upload data. 1342 if (existingOCR && (doc.ocr['User Upload'] || fullOcr)) { 1343 if (combineMode === 'conf') { 1344 /** @type {Parameters<import('./worker/compareOCRModule.js').compareOCRPageImp>[0]['options']} */ 1345 const compOptions = { 1346 debugLabel: scribeDocDefaults.saveDebugImages ? 'Combined' : undefined, 1347 supplementComp: true, 1348 ignoreCap: scribeDocDefaults.ignoreCap, 1349 ignorePunct: scribeDocDefaults.ignorePunct, 1350 confThreshHigh: scribeDocDefaults.confThreshHigh, 1351 confThreshMed: scribeDocDefaults.confThreshMed, 1352 editConf: true, 1353 }; 1354 1355 const res = await compareOCR(doc, existingOCR, doc.ocr['Tesseract Combined'], compOptions, progressCb); 1356 1357 if (doc.debug.debugImg.Combined) doc.debug.debugImg.Combined = res.debug; 1358 1359 clearObjectProperties(doc.ocr.Combined); 1360 Object.assign(doc.ocr.Combined, res.ocr); 1361 } else if (combineMode === 'data') { 1362 /** @type {Parameters<import('./worker/compareOCRModule.js').compareOCRPageImp>[0]['options']} */ 1363 const compOptions = { 1364 mode: 'comb', 1365 debugLabel: 'Combined', 1366 ignoreCap: scribeDocDefaults.ignoreCap, 1367 ignorePunct: scribeDocDefaults.ignorePunct, 1368 confThreshHigh: scribeDocDefaults.confThreshHigh, 1369 confThreshMed: scribeDocDefaults.confThreshMed, 1370 // If the existing data was invisible OCR text extracted from a PDF, it is assumed to not have accurate bounding boxes. 1371 useBboxB: !forceMainData && existingOCR === doc.ocr.pdf && doc.inputData.pdfMode && !!doc.inputData.pdfType && ['image', 'ocr'].includes(doc.inputData.pdfType), 1372 }; 1373 1374 let res; 1375 if (forceMainData) { 1376 res = await compareOCR(doc, doc.ocr['Tesseract Combined'], existingOCR, compOptions, progressCb); 1377 } else { 1378 res = await compareOCR(doc, existingOCR, doc.ocr['Tesseract Combined'], compOptions, progressCb); 1379 } 1380 1381 if (doc.debug.debugImg.Combined) doc.debug.debugImg.Combined = res.debug; 1382 1383 clearObjectProperties(doc.ocr.Combined); 1384 Object.assign(doc.ocr.Combined, res.ocr); 1385 } 1386 } 1387 } 1388 1389 // The engine's OCR layer to route: 'Tesseract Combined' for an existing-OCR run (where `active` points elsewhere), otherwise `active` itself. 1390 const tessSource = (existingOCR && doc.ocr['Tesseract Combined']) ? doc.ocr['Tesseract Combined'] : doc.ocr.active; 1391 buildCombinedLayer(doc, tessSource, ocrPageMask, gateApplies, fullOcr); 1392 return (doc.ocr.active); 1393}
Line numbers count LF bytes from the start of the resource, as the search results do. Vendor segments are library code the classifier recognised; they are stored but not indexed. Bytes are shown as Latin1 characters, one per byte.