1import ocr from '../objects/ocrObjects.js'; 2import { round6 } from '../utils/miscUtils.js'; 3 4/** 5 * 6 * @param {Object} params 7 * @param {Array<OcrPage>} params.ocrData 8 * @param {?Array<number>} [params.pageArr=null] - Array of 0-based page indices to include. Overrides minValue/maxValue when provided. 9 * @param {number} [params.minValue] 10 * @param {number} [params.maxValue] 11 * @param {import('../containers/fontContainer.js').DocFonts} [params.docFonts] - Per-document fonts. Required. 12 * @param {{ pages: Array<LayoutPage> }} [params.layoutRegions] - This document's layout regions. Required. 13 * @param {Array<PageMetrics>} [params.pageMetrics] - This document's page metrics. Required. 14 * @param {string} [params.dataTablesSerialized] - This document's serialized layout data tables. Required. 15 * @param {?import('../containers/scribeDoc.js').ScribeDoc} [params.doc=null] - Owning document for progress reporting. 16\ */ 17export function writeHocr({ 18 ocrData, pageArr = null, minValue, maxValue, 19 docFonts, layoutRegions: layoutRegionsArg, pageMetrics, dataTablesSerialized, doc = null, 20}) { 21 const fonts = docFonts; 22 const layoutRegionsPages = layoutRegionsArg.pages; 23 const pageMetricsArr = pageMetrics; 24 25 if (!pageArr) { 26 if (minValue === null || minValue === undefined) minValue = 0; 27 if (maxValue === null || maxValue === undefined || maxValue < 0) maxValue = ocrData.length - 1; 28 pageArr = []; 29 for (let i = minValue; i <= maxValue; i++) pageArr.push(i); 30 } 31 32 const meta = { 33 'font-metrics': fonts.state.charMetrics, 34 'default-font': fonts.state.defaultFontName, 35 'sans-font': fonts.state.sansDefaultName, 36 'serif-font': fonts.state.serifDefaultName, 37 'enable-opt': fonts.state.enableOpt, 38 layout: layoutRegionsPages, 39 'layout-data-table': dataTablesSerialized, 40 }; 41 42 let hocrOut = String.raw`<?xml version="1.0" encoding="UTF-8"?> 43<!DOCTYPE html PUBLIC "-//W3C//DTD XHTML 1.0 Transitional//EN" 44 "http://www.w3.org/TR/xhtml1/DTD/xhtml1-transitional.dtd"> 45<html xmlns="http://www.w3.org/1999/xhtml" xml:lang="en" lang="en">`; 46 47 hocrOut += '<head>'; 48 hocrOut += '\n\t<title></title>'; 49 50 // Add <meta> nodes provided by argument 51 for (const [key, value] of Object.entries(meta)) { 52 const valueStr = typeof value === 'object' ? JSON.stringify(value) : value; 53 hocrOut += `\n\t<meta name='${key}' content='${valueStr}'></meta>`; 54 } 55 56 hocrOut += '\n\t<meta http-equiv="Content-Type" content="text/html;charset=utf-8"/>'; 57 hocrOut += '\n\t<meta name=\'ocr-system\' content=\'scribeocr\' />'; 58 hocrOut += '\n\t<meta name=\'ocr-capabilities\' content=\'ocr_page ocr_carea ocr_par ocr_line ocrx_word ocrp_wconf ocrp_lang ocrp_dir ocrp_font ocrp_fsize\'/>'; 59 hocrOut += '\n</head>'; 60 hocrOut += '\n<body>'; 61 62 for (const i of pageArr) { 63 const pageObj = ocrData[i]; 64 65 // Handle case where ocrPage object does not exist. 66 if (!pageObj) { 67 hocrOut += `\n\t<div class='ocr_page' title='bbox 0 0 ${pageMetricsArr[i].dims.width} ${pageMetricsArr[i].dims.height}'>`; 68 hocrOut += '\n\t</div>'; 69 continue; 70 } 71 72 hocrOut += `\n\t<div class='ocr_page' title='bbox 0 0 ${pageObj.dims.width} ${pageObj.dims.height}'>`; 73 for (const lineObj of pageObj.lines) { 74 hocrOut += `\n\t\t<span class='ocr_line' title="bbox ${lineObj.bbox.left} ${lineObj.bbox.top} ${lineObj.bbox.right} ${lineObj.bbox.bottom}`; 75 hocrOut += `; baseline ${round6(lineObj.baseline[0])} ${Math.round(lineObj.baseline[1])}`; 76 77 // These metrics are specific to ScribeOCR, and are different from the Tesseract HOCR output. 78 // Per the HOCR spec, these properties must (1) be prefixed with `x_` and (2) only consist of lowercase letters and numbers (no camel case). 79 // The name of a property must only consist of lowercase letters and numbers. 80 // Property names must be either from those defined in §4 The properties of hOCR or begin with x_ to denote implementation-specific extensions. 81 // https://kba.github.io/hocr-spec/1.2/#definition-property 82 if (lineObj.xHeight) hocrOut += `; x_x_height ${lineObj.xHeight}`; 83 if (lineObj.ascHeight) hocrOut += `; x_asc_height ${lineObj.ascHeight}`; 84 hocrOut += '">'; 85 for (const wordObj of lineObj.words) { 86 hocrOut += `\n\t\t\t<span class='ocrx_word' id='${wordObj.id}' title='`; 87 // The HOCR specification requires that the bounding box be roun
87ded to the nearest integer, however the Scribe internal data structure does not. 88 hocrOut += `bbox ${Math.round(wordObj.bbox.left)} ${Math.round(wordObj.bbox.top)} ${Math.round(wordObj.bbox.right)} ${Math.round(wordObj.bbox.bottom)}`; 89 hocrOut += `;x_wconf ${wordObj.conf}`; 90 91 if (wordObj.style.font && wordObj.style.font !== 'Default') { 92 hocrOut += `;x_font ${wordObj.style.font}`; 93 } 94 95 if (wordObj.style.size) { 96 hocrOut += `;x_fsize ${wordObj.style.size}`; 97 } 98 99 hocrOut += "'"; 100 101 // Tesseract HOCR specifies default language for a paragraph in the "ocr_par" element, 102 // however as ScribeOCR does not currently have a paragarph object, every word must have its language specified. 103 if (wordObj.lang) hocrOut += ` lang='${wordObj.lang}'`; 104 105 // TODO: Why are we representing font family and style using the `style` HTML element here? 106 // This is not how Tesseract does things, and our own parsing script does not appear to be written to re-import it properly. 107 // Add "style" attribute (if applicable) 108 if (wordObj.style.bold || wordObj.style.italic || wordObj.style.smallCaps || (wordObj.style.font && wordObj.style.font !== 'Default')) { 109 hocrOut += ' style=\''; 110 111 if (wordObj.style.italic) { 112 hocrOut += 'font-style:italic;'; 113 } 114 115 if (wordObj.style.bold) { 116 hocrOut += 'font-weight:bold;'; 117 } 118 119 if (wordObj.style.smallCaps) { 120 hocrOut += 'font-variant:small-caps;'; 121 } 122 123 if (wordObj.style.font && wordObj.style.font !== 'Default') { 124 hocrOut += `font-family:${wordObj.style.font}`; 125 } 126 127 hocrOut += '\'>'; 128 } else { 129 hocrOut += '>'; 130 } 131 132 // Add word text, along with any formatting that uses nested elements rather than attributes 133 if (wordObj.style.sup) { 134 hocrOut += `<sup>${ocr.escapeXml(wordObj.text)}</sup>`;
135 } else if (wordObj.style.dropcap) { 136 hocrOut += `<span class='ocr_dropcap'>${ocr.escapeXml(wordObj.text)}</span>`; 137 } else { 138 hocrOut += ocr.escapeXml(wordObj.text); 139 } 140 hocrOut += '</span>'; 141 } 142 hocrOut += '\n\t\t</span>'; 143 } 144 hocrOut += '\n\t</div>'; 145 146 doc?.progressHandler({ n: i, type: 'export', info: { } }); 147 } 148 149 hocrOut += '\n</body>\n</html>'; 150 151 return hocrOut; 152}
Line numbers count LF bytes from the start of the resource, as the search results do. Vendor segments are library code the classifier recognised; they are stored but not indexed. Bytes are shown as Latin1 characters, one per byte.