PageSourceSearch

https://scribeocr.com/scribe.js/js/export/writeHocr.js

js scribeocr.com collected 2026-09-25 20:05:02 UTC 6,860 bytes, 152 lines download raw bytes

1import ocr from '../objects/ocrObjects.js';
2import { round6 } from '../utils/miscUtils.js';
3
4/**
5 *
6 * @param {Object} params
7 * @param {Array<OcrPage>} params.ocrData
8 * @param {?Array<number>} [params.pageArr=null] - Array of 0-based page indices to include. Overrides minValue/maxValue when provided.
9 * @param {number} [params.minValue]
10 * @param {number} [params.maxValue]
11 * @param {import('../containers/fontContainer.js').DocFonts} [params.docFonts] - Per-document fonts. Required.
12 * @param {{ pages: Array<LayoutPage> }} [params.layoutRegions] - This document's layout regions. Required.
13 * @param {Array<PageMetrics>} [params.pageMetrics] - This document's page metrics. Required.
14 * @param {string} [params.dataTablesSerialized] - This document's serialized layout data tables. Required.
15 * @param {?import('../containers/scribeDoc.js').ScribeDoc} [params.doc=null] - Owning document for progress reporting.
16\ */
17export function writeHocr({
18  ocrData, pageArr = null, minValue, maxValue,
19  docFonts, layoutRegions: layoutRegionsArg, pageMetrics, dataTablesSerialized, doc = null,
20}) {
21  const fonts = docFonts;
22  const layoutRegionsPages = layoutRegionsArg.pages;
23  const pageMetricsArr = pageMetrics;
24
25  if (!pageArr) {
26    if (minValue === null || minValue === undefined) minValue = 0;
27    if (maxValue === null || maxValue === undefined || maxValue < 0) maxValue = ocrData.length - 1;
28    pageArr = [];
29    for (let i = minValue; i <= maxValue; i++) pageArr.push(i);
30  }
31
32  const meta = {
33    'font-metrics': fonts.state.charMetrics,
34    'default-font': fonts.state.defaultFontName,
35    'sans-font': fonts.state.sansDefaultName,
36    'serif-font': fonts.state.serifDefaultName,
37    'enable-opt': fonts.state.enableOpt,
38    layout: layoutRegionsPages,
39    'layout-data-table': dataTablesSerialized,
40  };
41
42  let hocrOut = String.raw`<?xml version="1.0" encoding="UTF-8"?>
43<!DOCTYPE html PUBLIC "-//W3C//DTD XHTML 1.0 Transitional//EN"
44    "http://www.w3.org/TR/xhtml1/DTD/xhtml1-transitional.dtd">
45<html xmlns="http://www.w3.org/1999/xhtml" xml:lang="en" lang="en">`;
46
47  hocrOut += '<head>';
48  hocrOut += '\n\t<title></title>';
49
50  // Add <meta> nodes provided by argument
51  for (const [key, value] of Object.entries(meta)) {
52    const valueStr = typeof value === 'object' ? JSON.stringify(value) : value;
53    hocrOut += `\n\t<meta name='${key}' content='${valueStr}'></meta>`;
54  }
55
56  hocrOut += '\n\t<meta http-equiv="Content-Type" content="text/html;charset=utf-8"/>';
57  hocrOut += '\n\t<meta name=\'ocr-system\' content=\'scribeocr\' />';
58  hocrOut += '\n\t<meta name=\'ocr-capabilities\' content=\'ocr_page ocr_carea ocr_par ocr_line ocrx_word ocrp_wconf ocrp_lang ocrp_dir ocrp_font ocrp_fsize\'/>';
59  hocrOut += '\n</head>';
60  hocrOut += '\n<body>';
61
62  for (const i of pageArr) {
63    const pageObj = ocrData[i];
64
65    // Handle case where ocrPage object does not exist.
66    if (!pageObj) {
67      hocrOut += `\n\t<div class='ocr_page' title='bbox 0 0 ${pageMetricsArr[i].dims.width} ${pageMetricsArr[i].dims.height}'>`;
68      hocrOut += '\n\t</div>';
69      continue;
70    }
71
72    hocrOut += `\n\t<div class='ocr_page' title='bbox 0 0 ${pageObj.dims.width} ${pageObj.dims.height}'>`;
73    for (const lineObj of pageObj.lines) {
74      hocrOut += `\n\t\t<span class='ocr_line' title="bbox ${lineObj.bbox.left} ${lineObj.bbox.top} ${lineObj.bbox.right} ${lineObj.bbox.bottom}`;
75      hocrOut += `; baseline ${round6(lineObj.baseline[0])} ${Math.round(lineObj.baseline[1])}`;
76
77      // These metrics are specific to ScribeOCR, and are different from the Tesseract HOCR output.
78      // Per the HOCR spec, these properties must (1) be prefixed with `x_` and (2) only consist of lowercase letters and numbers (no camel case).
79      // The name of a property must only consist of lowercase letters and numbers.
80      // Property names must be either from those defined in §4 The properties of hOCR or begin with x_ to denote implementation-specific extensions.
81      // https://kba.github.io/hocr-spec/1.2/#definition-property
82      if (lineObj.xHeight) hocrOut += `; x_x_height ${lineObj.xHeight}`;
83      if (lineObj.ascHeight) hocrOut += `; x_asc_height ${lineObj.ascHeight}`;
84      hocrOut += '">';
85      for (const wordObj of lineObj.words) {
86        hocrOut += `\n\t\t\t<span class='ocrx_word' id='${wordObj.id}' title='`;
87        // The HOCR specification requires that the bounding box be roun
87ded to the nearest integer, however the Scribe internal data structure does not.
88        hocrOut += `bbox ${Math.round(wordObj.bbox.left)} ${Math.round(wordObj.bbox.top)} ${Math.round(wordObj.bbox.right)} ${Math.round(wordObj.bbox.bottom)}`;
89        hocrOut += `;x_wconf ${wordObj.conf}`;
90
91        if (wordObj.style.font && wordObj.style.font !== 'Default') {
92          hocrOut += `;x_font ${wordObj.style.font}`;
93        }
94
95        if (wordObj.style.size) {
96          hocrOut += `;x_fsize ${wordObj.style.size}`;
97        }
98
99        hocrOut += "'";
100
101        // Tesseract HOCR specifies default language for a paragraph in the "ocr_par" element,
102        // however as ScribeOCR does not currently have a paragarph object, every word must have its language specified.
103        if (wordObj.lang) hocrOut += ` lang='${wordObj.lang}'`;
104
105        // TODO: Why are we representing font family and style using the `style` HTML element here?
106        // This is not how Tesseract does things, and our own parsing script does not appear to be written to re-import it properly.
107        // Add "style" attribute (if applicable)
108        if (wordObj.style.bold || wordObj.style.italic || wordObj.style.smallCaps || (wordObj.style.font && wordObj.style.font !== 'Default')) {
109          hocrOut += ' style=\'';
110
111          if (wordObj.style.italic) {
112            hocrOut += 'font-style:italic;';
113          }
114
115          if (wordObj.style.bold) {
116            hocrOut += 'font-weight:bold;';
117          }
118
119          if (wordObj.style.smallCaps) {
120            hocrOut += 'font-variant:small-caps;';
121          }
122
123          if (wordObj.style.font && wordObj.style.font !== 'Default') {
124            hocrOut += `font-family:${wordObj.style.font}`;
125          }
126
127          hocrOut += '\'>';
128        } else {
129          hocrOut += '>';
130        }
131
132        // Add word text, along with any formatting that uses nested elements rather than attributes
133        if (wordObj.style.sup) {
134          hocrOut += `<sup>${ocr.escapeXml(wordObj.text)}</sup>`;
135        } else if (wordObj.style.dropcap) {
136          hocrOut += `<span class='ocr_dropcap'>${ocr.escapeXml(wordObj.text)}</span>`;
137        } else {
138          hocrOut += ocr.escapeXml(wordObj.text);
139        }
140        hocrOut += '</span>';
141      }
142      hocrOut += '\n\t\t</span>';
143    }
144    hocrOut += '\n\t</div>';
145
146    doc?.progressHandler({ n: i, type: 'export', info: { } });
147  }
148
149  hocrOut += '\n</body>\n</html>';
150
151  return hocrOut;
152}

Line numbers count LF bytes from the start of the resource, as the search results do. Vendor segments are library code the classifier recognised; they are stored but not indexed. Bytes are shown as Latin1 characters, one per byte.