1// ==ClosureCompiler== 2// @output_file_name default.js 3// @compilation_level SIMPLE_OPTIMIZATIONS 4// ==/ClosureCompiler== 5 6/** 7 * @author: Tobias Nickel 8 * @created: 06.04.2015 9 * I needed a small xmlparser chat can be used in a worker. 10 */ 11 12/** 13 * @typedef tNode 14 * @property {string} tagName 15 * @property {object} [attributes] 16 * @property {tNode|string|number[]} children 17 **/ 18 19/** 20 * parseXML / html into a DOM Object. with no validation and some failur tolerance 21 * @param {string} S your XML to parse 22 * @param options {object} all other options: 23 * searchId {string} the id of a single element, that should be returned. using this will increase the speed rapidly 24 * filter {function} filter method, as you know it from Array.filter. but is goes throw the DOM. 25 * simplify {bool} to use tXml.simplify. 26 * @return {tNode[]} 27 */ 28function tXml(S, options) { 29 "use strict"; 30 options = options || {}; 31 32 var pos = options.pos || 0; 33 34 var openBracket = "<"; 35 var openBracketCC = "<".charCodeAt(0); 36 var closeBracket = ">"; 37 var closeBracketCC = ">".charCodeAt(0); 38 var minus = "-"; 39 var minusCC = "-".charCodeAt(0); 40 var slash = "/"; 41 var slashCC = "/".charCodeAt(0); 42 var exclamation = '!'; 43 var exclamationCC = '!'.charCodeAt(0); 44 var singleQuote = "'"; 45 var singleQuoteCC = "'".charCodeAt(0); 46 var doubleQuote = '"'; 47 var doubleQuoteCC = '"'.charCodeAt(0); 48 49 /** 50 * parsing a list of entries 51 */ 52 function parseChildren() { 53 var children = []; 54 while (S[pos]) { 55 if (S.charCodeAt(pos) == openBracketCC) { 56 if (S.charCodeAt(pos + 1) === slashCC) { 57 pos = S.indexOf(closeBracket, pos); 58 if (pos + 1) pos += 1 59 return children; 60 } else if (S.charCodeAt(pos + 1) === exclamationCC) { 61 if (S.charCodeAt(pos + 2) == minusCC) { 62 //comment support 63 while (pos !== -1 && !(S.charCodeAt(pos) === closeBracketCC && S.charCodeAt(pos - 1) == minusCC && S.charCodeAt(pos - 2) == minusCC && pos != -1)) { 64 pos = S.indexOf(closeBracket, pos + 1); 65 } 66 if (pos === -1) { 67 pos = S.length 68 } 69 } else { 70 // doctypesupport 71 pos += 2; 72 while (S.charCodeAt(pos) !== closeBracketCC && S[pos]) { 73 pos++; 74 } 75 } 76 pos++; 77 continue; 78 } 79 var node = parseNode(); 80 children.push(node); 81 } else { 82 var text = parseText() 83 if (text.trim().length > 0) 84 children.push(text); 85 pos++; 86 } 87 } 88 return children; 89 } 90 91 /** 92 * returns the text outside of texts until the first '<' 93 */ 94 function parseText() { 95 var start = pos; 96 pos = S.indexOf(openBracket, pos) - 1; 97 if (pos === -2) 98 pos = S.length; 99 return S.slice(start, pos + 1); 100 } 101 /** 102 * returns text until the first nonAlphebetic letter 103 */ 104 var nameSpacer = '\n\t>/= '; 105 106 function parseName() { 107 var start = pos; 108 while (nameSpacer.indexOf(S[pos]) === -1 && S[pos]) { 109 pos++; 110 } 111 return S.slice(start, pos); 112 } 113 /** 114 * is parsing a node, including tagName, Attributes and its children, 115 * to parse children it uses the parseChildren again, that makes the parsing recursive 116 */ 117 var NoChildNodes = ['img', 'br', 'input', 'meta', 'link']; 118 119 function parseNode() { 120 var node = {}; 121 pos++; 122 node.tagName = parseName(); 123 // parsing attributes 124 var attrFound = false; 125 while (S.charCodeAt(pos) !== closeBracketCC && S[pos]) { 126 var c = S.charCodeAt(pos); 127 if ((c > 64 && c < 91) || (c > 96 && c < 123)) { 128 //if('abcdefghijklmnopqrstuvwxyzABCDEFGHIJKLMNOPQRSTUVWXYZ'.indexOf(S[pos])!==-1 ){ 129 var name = parseName(); 130 // search beginning of the string
131 var code = S.charCodeAt(pos); 132 while (code && code !== singleQuoteCC && code !== doubleQuoteCC && !((code > 64 && code < 91) || (code > 96 && code < 123)) && code !== closeBracketCC) { 133 pos++; 134 code = S.charCodeAt(pos); 135 } 136 if (!attrFound) { 137 node.attributes = {}; 138 attrFound = true; 139 } 140 if (code === singleQuoteCC || code === doubleQuoteCC) { 141 var value = parseString(); 142 if (pos === -1) { 143 return node; 144 } 145 } else { 146 value = null; 147 pos--; 148 } 149 node.attributes[name] = value; 150 } 151 pos++; 152 153 } 154 // optional parsing of children 155 if (S.charCodeAt(pos - 1) !== slashCC) { 156 if (node.tagName == "script") { 157 var start = pos + 1; 158 pos = S.indexOf('</' + 'script>', pos); 159 node.children = [S.slice(start, pos - 1)]; 160 pos += 8; 161 } else if (node.tagName == "style") { 162 var start = pos + 1; 163 pos = S.indexOf('</style>', pos); 164 node.children = [S.slice(start, pos - 1)]; 165 pos += 7; 166 } else if (NoChildNodes.indexOf(node.tagName) == -1) { 167 pos++; 168 node.children = parseChildren(name); 169 } 170 } else { 171 pos++; 172 } 173 return node; 174 } 175 176 /** 177 * is parsing a string, that starts with a char and with the same usually ' or " 178 */ 179 180 function parseString() { 181 var startChar = S[pos]; 182 var startpos = ++pos; 183 pos = S.indexOf(startChar, startpos) 184 return S.slice(startpos, pos); 185 } 186 187 /** 188 * 189 */ 190 191 function findElements() { 192 var r = new RegExp('\\s' + options.attrName + '\\s*=[\'"]' + options.attrValue + '[\'"]').exec(S) 193 if (r) { 194 return r.index; 195 } else { 196 return -1; 197 } 198 } 199 200 var out = null; 201 if (options.attrValue !== undefined) { 202 options.attrName = options.attrName || 'id'; 203 var out = []; 204 205 while ((pos = findElements()) !== -1) { 206 pos = S.lastIndexOf('<', pos); 207 if (pos !== -1) { 208 out.push(parseNode()); 209 } 210 S = S.substr(pos); 211 pos = 0; 212 } 213 } else if (options.parseNode) { 214 out = parseNode() 215 } else { 216 out = parseChildren(); 217 } 218 219 if (options.filter) { 220 out = tXml.filter(out, options.filter); 221 } 222 223 if (options.simplify) { 224 out = tXml.simplify(out); 225 } 226 out.pos = pos; 227 return out; 228} 229 230/** 231 * transform the DomObject to an object that is like the object of PHPs simplexmp_load_*() methods. 232 * this format helps you to write that is more likely to keep your programm working, even if there a small changes in the XML schema. 233 * be aware, that it is not possible to reproduce the original xml from a simplified version, because the order of elements is not saved. 234 * therefore your programm will be more flexible and easyer to read. 235 * 236 * @param {tNode[]} children the childrenList 237 */ 238tXml.simplify = function simplify(children) { 239 var out = {}; 240 if (!children.length) { 241 return ''; 242 } 243 244 if (children.length === 1 && typeof children[0] == 'string') { 245 return children[0]; 246 } 247 // map each object 248 children.forEach(function(child) { 249 if (typeof child !== 'object') { 250 return; 251 } 252 if (!out[child.tagName]) 253 out[child.tagName] = []; 254 var kids = tXml.simplify(child.children||[]); 255 out[child.tagName].push(kids); 256 if (child.attributes) { 257 kids._attributes = child.attributes; 258 } 259 }); 260 261 for (var i in out) { 262 if (out[i].length == 1) { 263 out[i] = out[i][0]; 264 } 265 } 266 267 return out; 268}; 269 270/** 271 * behaves the same way as Array.filter, if the filter method return true, the element is in the resultList 272 * @params children{Array} the children of a node 273 * @param f{function} the filter method 274 */ 275tXml.filter = function(children, f) { 276 var out = [];
277 children.forEach(function(child) { 278 if (typeof(child) === 'object' && f(child)) out.push(child); 279 if (child.children) { 280 var kids = tXml.filter(child.children, f); 281 out = out.concat(kids); 282 } 283 }); 284 return out; 285}; 286 287/** 288 * stringify a previously parsed string object. 289 * this is useful, 290 * 1. to remove whitespaces 291 * 2. to recreate xml data, with some changed data. 292 * @param {tNode} O the object to Stringify 293 */ 294tXml.stringify = function TOMObjToXML(O) { 295 var out = ''; 296 297 function writeChildren(O) { 298 if (O) 299 for (var i = 0; i < O.length; i++) { 300 if (typeof O[i] == 'string') { 301 out += O[i].trim(); 302 } else { 303 writeNode(O[i]); 304 } 305 } 306 } 307 308 function writeNode(N) { 309 out += "<" + N.tagName; 310 for (var i in N.attributes) { 311 if (N.attributes[i] === null) { 312 out += ' ' + i; 313 } else if (N.attributes[i].indexOf('"') === -1) { 314 out += ' ' + i + '="' + N.attributes[i].trim() + '"'; 315 } else { 316 out += ' ' + i + "='" + N.attributes[i].trim() + "'"; 317 } 318 } 319 out += '>'; 320 writeChildren(N.children); 321 out += '</' + N.tagName + '>'; 322 } 323 writeChildren(O); 324 325 return out; 326}; 327 328 329/** 330 * use this method to read the textcontent, of some node. 331 * It is great if you have mixed content like: 332 * this text has some <b>big</b> text and a <a href=''>link</a> 333 * @return {string} 334 */ 335tXml.toContentString = function(tDom) { 336 if (Array.isArray(tDom)) { 337 var out = ''; 338 tDom.forEach(function(e) { 339 out += ' ' + tXml.toContentString(e); 340 out = out.trim(); 341 }); 342 return out; 343 } else if (typeof tDom === 'object') { 344 return tXml.toContentString(tDom.children) 345 } else { 346 return ' ' + tDom; 347 } 348}; 349 350tXml.getElementById = function(S, id, simplified) { 351 var out = tXml(S, { 352 attrValue: id, 353 simplify: simplified 354 }); 355 return simplified ? out : out[0]; 356}; 357/** 358 * A fast parsing method, that not realy finds by classname, 359 * more: the class attribute contains XXX 360 * @param 361 */ 362tXml.getElementsByClassName = function(S, classname, simplified) { 363 return tXml(S, { 364 attrName: 'class', 365 attrValue: '[a-zA-Z0-9\-\s ]*' + classname + '[a-zA-Z0-9\-\s ]*', 366 simplify: simplified 367 }); 368}; 369 370tXml.parseStream = function(stream, offset) { 371 if (typeof offset === 'function') { 372 cb = offset; 373 offset = 0; 374 } 375 if (typeof offset === 'string') { 376 offset = offset.length + 2; 377 } 378 if (typeof stream === 'string') { 379 var fs = require('fs'); 380 stream = fs.createReadStream(stream, { start: offset }); 381 offset = 0; 382 } 383 384 var position = offset; 385 var data = ''; 386 var cc = 0 387 stream.on('data', function(chunk) { 388 cc++; 389 data += chunk; 390 var lastpos = 0; 391 do { 392 position = data.indexOf('<', position) + 1 393 var res = tXml(data, { pos: position, parseNode: true }); 394 position = res.pos; 395 if (position > (data.length - 1) || position < lastpos) { 396 if (lastpos) { 397 data = data.slice(lastpos); 398 position = 0 399 lastpos = 0; 400 } 401 return; 402 } else { 403 stream.emit('xml', res); 404 lastpos = position; 405 } 406 } while (1) 407 data = data.slice(position); 408 position = 0; 409 }); 410 stream.on('end', function() { 411 console.log('end') 412 }); 413 return stream; 414} 415 416if ('object' === typeof module) { 417 module.exports = tXml; 418} 419//console.clear(); 420//console.log('here:',tXml.getElementById('<some><xml id="test">dada</xml><that id="test">value</that></some>','test')); 421//console.log('here:',tXml.getElementsByClassName('<some><xml id="test" class="sdf test jsalf">dada</xml><that id="test">value</that></some>','test')); 422 423/* 424console.clear(); 425tXml(d,'content'); 426 //some testCode 427var s = document.body.innerHTML.toLowerCase(); 428var start = new Date().getTime(); 429var o = tXml(s,'content'); 430var end = new Date().getTime(); 431//console.log(JSON.stringify(o,undefined,'\t')); 432console.log("MILLISECONDS",end-start); 433var nodeCount=document.querySelectorAll('*').length; 434console.log('node count',nodeCount); 435console.log("speed:",(1000/(end-start))*nodeCount,'Nodes / second') 436//console.log(JSON.stringify(tXml('<html><head><title>testPage</title></head><body><h1>TestPage</h1><p>this is a <b>test</b>page</p></body></html>'),undefined,'\t')); 437var p = new DOMParser(); 438var s2='<body>'+s+'</body>'
439var start2= new Date().getTime(); 440var o2 = p.parseFromString(s2,'text/html').querySelector('#content') 441var end2=new Date().getTime(); 442console.log("MILLISECONDS",end2-start2); 443// */
Line numbers count LF bytes from the start of the resource, as the search results do. Vendor segments are library code the classifier recognised; they are stored but not indexed. Bytes are shown as Latin1 characters, one per byte.