PageSourceSearch

https://scribeocr.com/scribe.js/js/import/convertPageGoogleVision.js

js scribeocr.com collected 2026-09-25 20:05:59 UTC 6,834 bytes, 220 lines download raw bytes

1import ocr from '../objects/ocrObjects.js';
2
3import {
4  calcBboxUnion,
5  mean50,
6} from '../utils/miscUtils.js';
7
8import {
9  LayoutDataTablePage,
10} from '../objects/layoutObjects.js';
11import { pass3 } from './convertPageShared.js';
12
13const debugMode = false;
14
15/**
16 * @param {Object} params
17 * @param {string} params.ocrStr - String or array of strings containing Google Vision JSON data.
18 * @param {number} params.n
19 * @param {dims} [params.pageDims]
20 */
21export async function convertPageGoogleVision({ ocrStr, n, pageDims }) {
22  const ocrJson = JSON.parse(ocrStr);
23  let visionResult;
24  if (ocrJson.fullTextAnnotation) {
25    visionResult = ocrJson;
26  } else if (ocrJson?.responses?.[0]?.fullTextAnnotation) {
27    visionResult = ocrJson?.responses?.[0];
28  } else {
29    visionResult = ocrJson?.[0];
30  }
31
32  if (!visionResult || !visionResult.fullTextAnnotation) {
33    throw new Error('Failed to parse Google Vision OCR data.');
34  }
35
36  const pageVision = /** @type {GoogleVisionPage} */ (visionResult.fullTextAnnotation?.pages?.[0]);
37  const pageWidth = pageVision.width;
38  const pageHeight = pageVision.height;
39  if (!pageWidth || !pageHeight) {
40    throw new Error('Failed to parse page dimensions.');
41  }
42
43  const scaleX = pageDims ? pageDims.width / pageWidth : 1;
44  const scaleY = pageDims ? pageDims.height / pageHeight : 1;
45
46  /**
47   * @param {GoogleVisionParagraph["boundingBox"]} boundingBox - The bounding box object.
48   * @returns {Array<{x: number, y: number}>} - An array of vertex coordinates.
49   */
50  const getVertices = (boundingBox) => {
51    if (boundingBox.vertices) {
52      return boundingBox.vertices.map((v) => ({
53        x: (v.x || 0) * scaleX,
54        y: (v.y || 0) * scaleY,
55      }));
56    }
57    if (boundingBox.normalizedVertices) {
58      return boundingBox.normalizedVertices.map((v) => ({
59        x: (v.x || 0) * pageWidth * scaleX,
60        y: (v.y || 0) * pageHeight * scaleY,
61      }));
62    }
63    throw new Error('No vertices found in bounding box.');
64  };
65
66  const pageDimsOut = pageDims || { width: pageWidth, height: pageHeight };
67
68  const pageObj = new ocr.OcrPage(n, pageDimsOut);
69
70  if (!pageVision.blocks || pageVision.blocks.length === 0) {
71    const warn = { char: 'char_error' };
72    return {
73      pageObj,
74      charMetricsObj: {},
75      dataTables: new LayoutDataTablePage(n),
76      warn,
77    };
78  }
79
80  const tablesPage = new LayoutDataTablePage(n);
81
82  /** @type {Array<number>} */
83  const angleRisePage = [];
84
85  pageVision.blocks.forEach((block, blockIndex) => {
86    if (!block.paragraphs) return;
87
88    block.paragraphs.forEach((paragraph, paragraphIndex) => {
89      const wordsVision = paragraph.words;
90      if (!wordsVision || wordsVision.length === 0) return;
91
92      const parVertices = getVertices(paragraph.boundingBox);
93      const xsPar = parVertices.map((v) => v.x || 0);
94      const ysPar = parVertices.map((v) => v.y || 0);
95
96      const bboxPar = {
97        left: Math.min(...xsPar),
98        top: Math.min(...ysPar),
99        right: Math.max(...xsPar),
100        bottom: Math.max(...ysPar),
101      };
102
103      const parObj = new ocr.OcrPar(pageObj, bboxPar);
104      parObj.reason = String(block.blockType || 'TEXT');
105
106      if (debugMode) {
107        parObj.debug.sourceType = block.blockType || null;
108      }
109
110      let lineObj = new ocr.OcrLine(pageObj, null, [0, 0]);
111      let lineIndex = 0;
112
113      wordsVision.forEach((word, wordIndex) => {
114        if (!word.symbols || word.symbols.length === 0) return;
115
116        const wordVertices = getVertices(word.boundingBox);
117        const xs = wordVertices.map((v) => v.x || 0);
118        const ys = wordVertices.map((v) => v.y || 0);
119
120        const bboxWord = {
121          left: Math.min(...xs),
122          top: Math.min(...ys),
123          right: Math.max(...xs),
124          bottom: Math.max(...ys),
125        };
126
127        const id = `word_${n + 1}_${blockIndex + 1}_${paragraphIndex + 1}_${lineIndex + 1}_${wordIndex + 1}`;
128
129        const wordText = word.symbols.map((symbol) => symbol.text || '').join('');
130
131        const incChars = false;
132        let charObjs = /** @type {?OcrChar[]} */ (null);
133        if (incChars) {
134          charObjs = [];
135          if (word.symbols) {
136            word.symbols.forEach((symbol) => {
137              const charVertices = getVertices(symbol.boundingBox);
138              const charXs = charVertices.map((v) => v.x || 0);
139              const charYs = charVertices.map((v) => v.y || 0);
140              const charBbox = {
141                left: Math.min(...charXs),
142                top: Math.min(...charYs),
143                right: Math.max(...charXs),
144                bottom: Math.max(...charYs),
145              };
146              const charObj = new ocr.OcrChar(symbol.text || '', charBbox);
147              charObjs.push(charObj);
148            });
149          }
150        }
151
152        const wordObj = new ocr.OcrWord(lineObj, id, wordText, bboxWord);
153        wordObj.conf = (word.confidence || 0) * 100;
154        wordObj.chars = charObjs;
155
156        if (debugMode) {
157          wordObj.debug.raw = JSON.stringify(word);
158        }
159
160        lineObj.words.push(wordObj);
161
162        const hasLineBreak = word.symbols.some((symbol) => {
163          const breakType = symbol.property?.detectedBreak?.type;
164          return breakType === 'LINE_BREAK' || breakType === 'EOL_SURE_SPACE';
165        });
166
167        if (hasLineBreak || wordIndex === wordsVision.length - 1) {
168          if (lineObj.words.length > 0) {
169            const wordBboxes = lineObj.words.map((w) => w.bbox);
170            lineObj.bbox = calcBboxUnion(wordBboxes);
171
172            calculateTextMetrics(lineObj);
173
174            pageObj.lines.push(lineObj);
175            parObj.lines.push(lineObj);
176            lineObj.par = parObj;
177            lineIndex++;
178          }
179
180          if (wordIndex !== wordsVision.length - 1) {
181            lineObj = new ocr.OcrLine(pageObj, null, [0, 0]);
182          }
183        }
184      });
185
186      if (parObj.lines.length > 0) {
187        pageObj.pars.push(parObj);
188      }
189    });
190  });
191
192  pageObj.lines.forEach((line) => {
193    const wordBoxArr = line.words.map((x) => x.bbox);
194    line.bbox = calcBboxUnion(wordBoxArr);
195  });
196
197  const angleRiseMedian = mean50(angleRisePage) || 0;
198  const angleOut = Math.asin(angleRiseMedian) * (180 / Math.PI);
199  pageObj.angle = angleOut;
200  pageObj.textSource = 'google_vision';
201
202  const langSet = pass3(pageObj);
203
204  return { pageObj, dataTables: tablesPage, langSet };
205}
206
207/**
208 *
209 * @param {OcrLine} lineObj - The line object to update
210 */
211function calculateTextMetrics(lineObj) {
212  const wordHeights = lineObj.words.map((w) => w.bbox.bottom - w.bbox.top);
213  if (wordHeights.length === 0) return;
214
215  const sortedHeights = [...wordHeights].sort((a, b) => a - b);
216  const medianHeight = sortedHeights[Math.floor(sortedHeights.length / 2)];
217
218  lineObj.ascHeight = medianHeight * 2 / 3;
219  lineObj.baseline[1] = medianHeight * -1 / 3;
220}

Line numbers count LF bytes from the start of the resource, as the search results do. Vendor segments are library code the classifier recognised; they are stored but not indexed. Bytes are shown as Latin1 characters, one per byte.