1import { scribeDocDefaults } from '../containers/scribeDocDefaults.js'; 2import ocr from '../objects/ocrObjects.js'; 3import { calcWordMetrics } from './fontUtils.js'; 4import { calcBboxUnion, getRandomAlphanum } from './miscUtils.js'; 5 6/** 7 * Count words above the high-confidence threshold across `pages`. 8 * @param {Array<OcrPage>} pages 9 * @param {number} [confThreshHigh] 10 */ 11export const calcConf = (pages, confThreshHigh = scribeDocDefaults.confThreshHigh) => { 12 let wordsTotal = 0; 13 let wordsHighConf = 0; 14 for (let i = 0; i < pages.length; i++) { 15 const words = ocr.getPageWords(pages[i]); 16 for (let j = 0; j < words.length; j++) { 17 const word = words[j]; 18 wordsTotal += 1; 19 if (word.conf > confThreshHigh) wordsHighConf += 1; 20 } 21 } 22 return { total: wordsTotal, highConf: wordsHighConf }; 23}; 24 25/** 26 * 27 * @param {OcrWord} word 28 * @param {number} splitIndex 29 * @param {import('../containers/fontContainer.js').DocFonts} docFonts - Fonts used to estimate the 30 * split point when character-level metrics are missing or unreliable. 31 * @returns 32 */ 33export function splitOcrWord(word, splitIndex, docFonts) { 34 const wordA = ocr.cloneWord(word); 35 const wordB = ocr.cloneWord(word); 36 37 // Character-level metrics are often present and reliable, however may not be. 38 // If a user edits the text, then the character-level metrics from the OCR engine will not match. 39 // Therefore, a fallback strategy is used in this case to calculate where to split the word. 40 const validCharData = word.chars && word.chars.map((x) => x.text).join('') === word.text; 41 if (wordA.chars && wordB.chars) { 42 wordA.chars.splice(splitIndex); 43 wordB.chars.splice(0, splitIndex); 44 if (validCharData) { 45 wordA.bbox = calcBboxUnion(wordA.chars.map((x) => x.bbox)); 46 wordB.bbox = calcBboxUnion(wordB.chars.map((x) => x.bbox)); 47 } 48 } 49 50 // TODO: This is a quick fix; figure out how to get this math correct. 51 if (!validCharData) { 52 const metrics = calcWordMetrics(wordA, docFonts); 53 wordA.bbox.right -= metrics.advanceArr.slice(splitIndex).reduce((a, b) => a + b, 0); 54 wordB.bbox.left = wordA.bbox.right; 55 } 56 57 wordA.text = wordA.text.split('').slice(0, splitIndex).join(''); 58 wordB.text = wordB.text.split('').slice(splitIndex).join(''); 59 60 wordA.id = `${word.id}a`; 61 wordB.id = `${word.id}b`; 62 63 return { wordA, wordB }; 64} 65 66/** 67 * 68 * @param {Array<OcrWord>} words 69 * @returns 70 */ 71export function mergeOcrWords(words) { 72 words.sort((a, b) => a.bbox.left - b.bbox.left); 73 const wordA = ocr.cloneWord(words[0]); 74 wordA.bbox.right = words[words.length - 1].bbox.right; 75 wordA.text = words.map((x) => x.text).join(''); 76 if (wordA.chars) wordA.chars = words.flatMap((x) => x.chars || []); 77 return wordA; 78} 79 80/** 81 * 82 * @param {Array<OcrWord>} words 83 * @returns 84 */ 85export const checkOcrWordsAdjacent = (words) => { 86 const sortedWords = words.slice().sort((a, b) => a.bbox.left - b.bbox.left); 87 const lineWords = words[0].line.words; 88 lineWords.sort((a, b) => a.bbox.left - b.bbox.left); 89 90 const firstIndex = lineWords.findIndex((x) => x.id === sortedWords[0].id); 91 const lastIndex = lineWords.findIndex((x) => x.id === sortedWords[sortedWords.length - 1].id); 92 return lastIndex - firstIndex === sortedWords.length - 1; 93}; 94 95/** 96 * 97 * @param {OcrLine} line 98 */ 99export const splitLineAgressively = (line) => { 100 /** @type {Array<OcrLine>} */ 101 const linesOut = []; 102 const lineHeight = line.bbox.bottom - line.bbox.top; 103 let wordPrev = line.words[0]; 104 let lineCurrent = ocr.cloneLine(line); 105 lineCurrent.words = [line.words[0]]; 106 for (let i = 1; i < line.words.length; i++) { 107 const word = ocr.cloneWord(line.words[i]); 108 if (word.bbox.left - wordPrev.bbox.right > lineHeight) { 109 linesOut.push(lineCurrent); 110 lineCurrent = ocr.cloneLine(line); 111 word.line = lineCurrent; 112 lineCurrent.words = [word]; 113 } else { 114 word.line = lineCurrent; 115 lineCurrent.words.push(word); 116 } 117 wordPrev = word; 118 } 119 linesOut.push(lineCurrent); 120 121 linesOut.forEach((x) => { 122 ocr.updateLineBbox(x); 123 }); 124 125 // Generate new IDs for all split lines except the first (which keeps the original ID) 126 for (let i = 1; i < linesOut.length; i++) { 127 linesOut[i].id = getRandomAlphanum(8); 128 } 129 130 return linesOut; 131};
Line numbers count LF bytes from the start of the resource, as the search results do. Vendor segments are library code the classifier recognised; they are stored but not indexed. Bytes are shown as Latin1 characters, one per byte.