PageSourceSearch

https://scribeocr.com/scribe.js/js/utils/ocrUtils.js

js scribeocr.com collected 2026-09-25 20:05:15 UTC 4,278 bytes, 131 lines download raw bytes

1import { scribeDocDefaults } from '../containers/scribeDocDefaults.js';
2import ocr from '../objects/ocrObjects.js';
3import { calcWordMetrics } from './fontUtils.js';
4import { calcBboxUnion, getRandomAlphanum } from './miscUtils.js';
5
6/**
7 * Count words above the high-confidence threshold across `pages`.
8 * @param {Array<OcrPage>} pages
9 * @param {number} [confThreshHigh]
10 */
11export const calcConf = (pages, confThreshHigh = scribeDocDefaults.confThreshHigh) => {
12  let wordsTotal = 0;
13  let wordsHighConf = 0;
14  for (let i = 0; i < pages.length; i++) {
15    const words = ocr.getPageWords(pages[i]);
16    for (let j = 0; j < words.length; j++) {
17      const word = words[j];
18      wordsTotal += 1;
19      if (word.conf > confThreshHigh) wordsHighConf += 1;
20    }
21  }
22  return { total: wordsTotal, highConf: wordsHighConf };
23};
24
25/**
26 *
27 * @param {OcrWord} word
28 * @param {number} splitIndex
29 * @param {import('../containers/fontContainer.js').DocFonts} docFonts - Fonts used to estimate the
30 *   split point when character-level metrics are missing or unreliable.
31 * @returns
32 */
33export function splitOcrWord(word, splitIndex, docFonts) {
34  const wordA = ocr.cloneWord(word);
35  const wordB = ocr.cloneWord(word);
36
37  // Character-level metrics are often present and reliable, however may not be.
38  // If a user edits the text, then the character-level metrics from the OCR engine will not match.
39  // Therefore, a fallback strategy is used in this case to calculate where to split the word.
40  const validCharData = word.chars && word.chars.map((x) => x.text).join('') === word.text;
41  if (wordA.chars && wordB.chars) {
42    wordA.chars.splice(splitIndex);
43    wordB.chars.splice(0, splitIndex);
44    if (validCharData) {
45      wordA.bbox = calcBboxUnion(wordA.chars.map((x) => x.bbox));
46      wordB.bbox = calcBboxUnion(wordB.chars.map((x) => x.bbox));
47    }
48  }
49
50  // TODO: This is a quick fix; figure out how to get this math correct.
51  if (!validCharData) {
52    const metrics = calcWordMetrics(wordA, docFonts);
53    wordA.bbox.right -= metrics.advanceArr.slice(splitIndex).reduce((a, b) => a + b, 0);
54    wordB.bbox.left = wordA.bbox.right;
55  }
56
57  wordA.text = wordA.text.split('').slice(0, splitIndex).join('');
58  wordB.text = wordB.text.split('').slice(splitIndex).join('');
59
60  wordA.id = `${word.id}a`;
61  wordB.id = `${word.id}b`;
62
63  return { wordA, wordB };
64}
65
66/**
67 *
68 * @param {Array<OcrWord>} words
69 * @returns
70 */
71export function mergeOcrWords(words) {
72  words.sort((a, b) => a.bbox.left - b.bbox.left);
73  const wordA = ocr.cloneWord(words[0]);
74  wordA.bbox.right = words[words.length - 1].bbox.right;
75  wordA.text = words.map((x) => x.text).join('');
76  if (wordA.chars) wordA.chars = words.flatMap((x) => x.chars || []);
77  return wordA;
78}
79
80/**
81 *
82 * @param {Array<OcrWord>} words
83 * @returns
84 */
85export const checkOcrWordsAdjacent = (words) => {
86  const sortedWords = words.slice().sort((a, b) => a.bbox.left - b.bbox.left);
87  const lineWords = words[0].line.words;
88  lineWords.sort((a, b) => a.bbox.left - b.bbox.left);
89
90  const firstIndex = lineWords.findIndex((x) => x.id === sortedWords[0].id);
91  const lastIndex = lineWords.findIndex((x) => x.id === sortedWords[sortedWords.length - 1].id);
92  return lastIndex - firstIndex === sortedWords.length - 1;
93};
94
95/**
96 *
97 * @param {OcrLine} line
98 */
99export const splitLineAgressively = (line) => {
100  /** @type {Array<OcrLine>} */
101  const linesOut = [];
102  const lineHeight = line.bbox.bottom - line.bbox.top;
103  let wordPrev = line.words[0];
104  let lineCurrent = ocr.cloneLine(line);
105  lineCurrent.words = [line.words[0]];
106  for (let i = 1; i < line.words.length; i++) {
107    const word = ocr.cloneWord(line.words[i]);
108    if (word.bbox.left - wordPrev.bbox.right > lineHeight) {
109      linesOut.push(lineCurrent);
110      lineCurrent = ocr.cloneLine(line);
111      word.line = lineCurrent;
112      lineCurrent.words = [word];
113    } else {
114      word.line = lineCurrent;
115      lineCurrent.words.push(word);
116    }
117    wordPrev = word;
118  }
119  linesOut.push(lineCurrent);
120
121  linesOut.forEach((x) => {
122    ocr.updateLineBbox(x);
123  });
124
125  // Generate new IDs for all split lines except the first (which keeps the original ID)
126  for (let i = 1; i < linesOut.length; i++) {
127    linesOut[i].id = getRandomAlphanum(8);
128  }
129
130  return linesOut;
131};

Line numbers count LF bytes from the start of the resource, as the search results do. Vendor segments are library code the classifier recognised; they are stored but not indexed. Bytes are shown as Latin1 characters, one per byte.