1const normalizeWhitespace = str => str ? str 2 .replace(/[\t\n\f\r ]+/g, ' ') 3 .replace(/^[\t\n\f\r ]+/, '') 4 .replace(/[\t\n\f\r ]+$/, '') : '' 5const getElementText = el => normalizeWhitespace(el?.textContent) 6 7const NS = { 8 XLINK: 'http://www.w3.org/1999/xlink', 9 EPUB: 'http://www.idpf.org/2007/ops', 10} 11 12const MIME = { 13 XML: 'application/xml', 14 XHTML: 'application/xhtml+xml', 15} 16 17const STYLE = { 18 'strong': ['strong', 'self'], 19 'emphasis': ['em', 'self'], 20 'style': ['span', 'self'], 21 'a': 'anchor', 22 'strikethrough': ['s', 'self'], 23 'sub': ['sub', 'self'], 24 'sup': ['sup', 'self'], 25 'code': ['code', 'self'], 26 'image': 'image', 27} 28 29const TABLE = { 30 'tr': ['tr', { 31 'th': ['th', STYLE, ['colspan', 'rowspan', 'align', 'valign']], 32 'td': ['td', STYLE, ['colspan', 'rowspan', 'align', 'valign']], 33 }, ['align']], 34} 35 36const POEM = { 37 'epigraph': ['blockquote'], 38 'subtitle': ['h2', STYLE], 39 'text-author': ['p', STYLE], 40 'date': ['p', STYLE], 41 'stanza': 'stanza', 42} 43 44const SECTION = { 45 'title': ['header', { 46 'p': ['h1', STYLE], 47 'empty-line': ['br'], 48 }], 49 'epigraph': ['blockquote', 'self'], 50 'image': 'image', 51 'annotation': ['aside'], 52 'section': ['section', 'self'], 53 'p': ['p', STYLE], 54 'poem': ['blockquote', POEM], 55 'subtitle': ['h2', STYLE], 56 'cite': ['blockquote', 'self'], 57 'empty-line': ['br'], 58 'table': ['table', TABLE], 59 'text-author': ['p', STYLE], 60} 61POEM['epigraph'].push(SECTION) 62 63const BODY = { 64 'image': 'image', 65 'title': ['section', { 66 'p': ['h1', STYLE], 67 'empty-line': ['br'], 68 }], 69 'epigraph': ['section', SECTION], 70 'section': ['section', SECTION], 71} 72 73class FB2Converter { 74 constructor(fb2) { 75 this.fb2 = fb2 76 this.doc = document.implementation.createDocument(NS.XHTML, 'html') 77 // use this instead of `getElementById` to allow images like 78 // `<image l:href="#img1.jpg" id="img1.jpg" />` 79 this.bins = new Map(Array.from(this.fb2.getElementsByTagName('binary'), 80 el => [el.id, el])) 81 } 82 getImageSrc(el) { 83 const href = el.getAttributeNS(NS.XLINK, 'href') 84 if (!href) return 'data:,' 85 const [, id] = href.split('#') 86 if (!id) return href 87 const bin = this.bins.get(id) 88 return bin 89 ? `data:${bin.getAttribute('content-type')};base64,${bin.textContent}` 90 : href 91 } 92 image(node) { 93 const el = this.doc.createElement('img') 94 el.alt = node.getAttribute('alt') 95 el.title = node.getAttribute('title') 96 el.setAttribute('src', this.getImageSrc(node)) 97 return el 98 } 99 anchor(node) { 100 const el = this.convert(node, { 'a': ['a', STYLE] }) 101 el.setAttribute('href', node.getAttributeNS(NS.XLINK, 'href')) 102 if (node.getAttribute('type') === 'note') 103 el.setAttributeNS(NS.EPUB, 'epub:type', 'noteref') 104 return el 105 } 106 stanza(node) { 107 const el = this.convert(node, { 108 'stanza': ['p', { 109 'title': ['header', { 110 'p': ['strong', STYLE], 111 'empty-line': ['br'], 112 }], 113 'subtitle': ['p', STYLE], 114 }], 115 }) 116 for (const child of node.children) if (child.nodeName === 'v') { 117 el.append(this.doc.createTextNode(child.textContent)) 118 el.append(this.doc.createElement('br')) 119 } 120 return el 121 } 122 convert(node, def) { 123 // not an element; return text content 124 if (node.nodeType === 3) return this.doc.createTextNode(node.textContent) 125 if (node.nodeType === 4) return this.doc.createCDATASection(node.textContent) 126 if (node.nodeType === 8) return this.doc.createComment(node.textContent) 127 128 const d = def?.[node.nodeName] 129 if (!d) return null 130 if (typeof d === 'string') return this[d](node) 131 132 const [name, opts, attrs] = d 133 const el = this.doc.createElement(name) 134 135 // copy the ID, and set class name from original element name 136 if (node.id) el.id = node.id 137 el.classList.add(node.nodeName) 138 139 // copy attributes 140 if (Array.isArray(attrs)) for (const attr of attrs) { 141 const value = node.getAttribute(attr) 142 if (value) el.setAttribute(attr, value) 143 } 144 145 // process child elements recursively 146 const childDef = opts === 'self' ? def : opts 147 let child = node.firstChild 148 while (child) { 149 const childEl = this.convert(child, childDef) 150 if (childEl) el.append(childEl) 151 child = child.nextSibling 152 } 153 return el 154 } 155} 156 157const parseXML = async blob => { 158 const buffer = await blob.arrayBuffer() 159 const str = new TextDecoder('utf-8').decode(buffer) 160 const parser = new DOMParser() 161 const doc = parser.parseFromString(str, MIME.XML) 162 const encoding = doc.xmlEncoding 163 // `Document.xmlEncoding` is deprecated, and already removed in Firefox 164 // so parse the XML declaration manually
165 || str.match(/^<\?xml\s+version\s*=\s*["']1.\d+"\s+encoding\s*=\s*["']([A-Za-z0-9._-]*)["']/)?.[1] 166 if (encoding && encoding.toLowerCase() !== 'utf-8') { 167 const str = new TextDecoder(encoding).decode(buffer) 168 return parser.parseFromString(str, MIME.XML) 169 } 170 return doc 171} 172 173const style = URL.createObjectURL(new Blob([` 174@namespace epub "http://www.idpf.org/2007/ops"; 175body > img, section > img { 176 display: block; 177 margin: auto; 178} 179.title h1 { 180 text-align: center; 181} 182body > section > .title, body.notesBodyType > .title { 183 margin: 3em 0; 184} 185body.notesBodyType > section .title h1 { 186 text-align: start; 187} 188body.notesBodyType > section .title { 189 margin: 1em 0; 190} 191p { 192 text-indent: 1em; 193 margin: 0; 194} 195:not(p) + p, p:first-child { 196 text-indent: 0; 197} 198.poem p { 199 text-indent: 0; 200 margin: 1em 0; 201} 202.text-author, .date { 203 text-align: end; 204} 205.text-author:before { 206 content: "â"; 207} 208table { 209 border-collapse: collapse; 210} 211td, th { 212 padding: .25em; 213} 214a[epub|type~="noteref"] { 215 font-size: .75em; 216 vertical-align: super; 217} 218body:not(.notesBodyType) > .title, body:not(.notesBodyType) > .epigraph { 219 margin: 3em 0; 220} 221`], { type: 'text/css' })) 222 223const template = html => `<?xml version="1.0" encoding="utf-8"?> 224<html xmlns="http://www.w3.org/1999/xhtml"> 225 <head><link href="${style}" rel="stylesheet" type="text/css"/></head> 226 <body>${html}</body> 227</html>` 228 229// name of custom ID attribute for TOC items 230const dataID = 'data-foliate-id' 231 232export const makeFB2 = async blob => { 233 const book = {} 234 const doc = await parseXML(blob) 235 const converter = new FB2Converter(doc) 236 237 const $ = x => doc.querySelector(x) 238 const $$ = x => [...doc.querySelectorAll(x)] 239 const getPerson = el => { 240 const nick = getElementText(el.querySelector('nickname')) 241 if (nick) return nick 242 const first = getElementText(el.querySelector('first-name')) 243 const middle = getElementText(el.querySelector('middle-name')) 244 const last = getElementText(el.querySelector('last-name')) 245 const name = [first, middle, last].filter(x => x).join(' ') 246 const sortAs = last 247 ? [last, [first, middle].filter(x => x).join(' ')].join(', ') 248 : null 249 return { name, sortAs } 250 } 251 const getDate = el => el?.getAttribute('value') ?? getElementText(el) 252 const annotation = $('title-info annotation') 253 book.metadata = { 254 title: getElementText($('title-info book-title')), 255 identifier: getElementText($('document-info id')), 256 language: getElementText($('title-info lang')), 257 author: $$('title-info author').map(getPerson), 258 translator: $$('title-info translator').map(getPerson), 259 contributor: $$('document-info author').map(getPerson) 260 // techincially the program probably shouldn't get the `bkp` role 261 // but it has been so used by calibre, so ¯\_(ã)_/¯ 262 .concat($$('document-info program-used').map(getElementText)) 263 .map(x => Object.assign(typeof x === 'string' ? { name: x } : x, 264 { role: 'bkp' })), 265 publisher: getElementText($('publish-info publisher')), 266 published: getDate($('title-info date')), 267 modified: getDate($('document-info date')), 268 description: annotation ? converter.convert(annotation, 269 { annotation: ['div', SECTION] }).innerHTML : null, 270 subject: $$('title-info genre').map(getElementText), 271 } 272 if ($('coverpage image')) { 273 const src = converter.getImageSrc($('coverpage image')) 274 book.getCover = () => fetch(src).then(res => res.blob()) 275 } else book.getCover = () => null 276 277 // get convert each body 278 const bodyData = Array.from(doc.querySelectorAll('body'), body => { 279 const converted = converter.convert(body, { body: ['body', BODY] }) 280 return [Array.from(converted.children, el => { 281 // get list of IDs in the section 282 const ids = [el, ...el.querySelectorAll('[id]')].map(el => el.id) 283 return { el, ids } 284 }), converted] 285 }) 286 287 const urls = [] 288 const sectionData = bodyData[0][0] 289 // make a separate section for each section in the first body 290 .map(({ el, ids }) => { 291 // set up titles for TOC 292 const titles = Array.from( 293 el.querySelectorAll(':scope > section > .title'), 294 (el, index) => { 295 el.setAttribute(dataID, index) 296 return { title: getElementText(el), index } 297 }) 298 return { ids, titles, el } 299 }) 300 // for additional bodies, only make one section for each body 301 .concat(bodyData.slice(1).map(([sections, body]) => { 302 const ids = sections.map(s => s.ids).flat() 303 body.classList.add('notesBodyType') 304 return { ids, el: body, linear: 'no' } 305 })) 306 .map(({ ids, titles, el, linear }) => { 307 const str = template(el.outerHTML) 308 const blob = new Blob([str], { type: MIME.XHTML }) 309 const url = URL.createObjectURL(blob) 310 urls.push(url) 311 const title = normalizeWhitespace( 312 el.querySelector('.title, .subtitle, p')?.textContent 313 ?? (el.classList.contains('title') ? el.textContent : '')) 314 return { 315 ids, title, titles, load: () => url, 316 createDocument: () => new DOMParser().parseFromString(str, MIME.XHTML), 317 // doo't count image data as it'd skew the size too much 318 size: blob.size - Array.from(el.querySelectorAll('[src]'), 319 el => el.getAttribute('src')?.length ?? 0) 320 .reduce((a, b) => a + b, 0), 321 linear, 322 } 323 }) 324 325 const idMap = new Map() 326 book.sections = sectionData.map((section, index) => { 327 const { ids, load, createDocument, size, linear } = section 328 for (const id of ids) if (id) idMap.set(id, index) 329 return { id: index, load, createDocument, size, linear } 330 }) 331 332 book.toc = sectionData.map(({ title, titles }, index) => { 333 const id = index.toString() 334 return { 335 label: title, 336 href: id,
337 subitems: titles?.length ? titles.map(({ title, index }) => ({ 338 label: title, 339 href: `${id}#${index}`, 340 })) : null, 341 } 342 }).filter(item => item) 343 344 book.resolveHref = href => { 345 const [a, b] = href.split('#') 346 return a 347 // the link is from the TOC 348 ? { index: Number(a), anchor: doc => doc.querySelector(`[${dataID}="${b}"]`) } 349 // link from within the page 350 : { index: idMap.get(b), anchor: doc => doc.getElementById(b) } 351 } 352 book.splitTOCHref = href => href?.split('#')?.map(x => Number(x)) ?? [] 353 book.getTOCFragment = (doc, id) => doc.querySelector(`[${dataID}="${id}"]`) 354 355 book.destroy = () => { 356 for (const url of urls) URL.revokeObjectURL(url) 357 } 358 return book 359}
Line numbers count LF bytes from the start of the resource, as the search results do. Vendor segments are library code the classifier recognised; they are stored but not indexed. Bytes are shown as Latin1 characters, one per byte.