PageSourceSearch

https://prh.fi/stato/js/lib/tXml.js

js prh.fi collected 2026-09-24 08:46:56 UTC 13,608 bytes, 443 lines download raw bytes

1// ==ClosureCompiler==
2// @output_file_name default.js
3// @compilation_level SIMPLE_OPTIMIZATIONS
4// ==/ClosureCompiler==
5
6/**
7 * @author: Tobias Nickel
8 * @created: 06.04.2015
9 * I needed a small xmlparser chat can be used in a worker.
10 */
11
12/**
13 * @typedef tNode 
14 * @property {string} tagName 
15 * @property {object} [attributes] 
16 * @property {tNode|string|number[]} children 
17 **/
18
19/**
20 * parseXML / html into a DOM Object. with no validation and some failur tolerance
21 * @param {string} S your XML to parse
22 * @param options {object} all other options:
23 * searchId {string} the id of a single element, that should be returned. using this will increase the speed rapidly
24 * filter {function} filter method, as you know it from Array.filter. but is goes throw the DOM.
25 * simplify {bool} to use tXml.simplify.
26 * @return {tNode[]}
27 */
28function tXml(S, options) {
29    "use strict";
30    options = options || {};
31
32    var pos = options.pos || 0;
33
34    var openBracket = "<";
35    var openBracketCC = "<".charCodeAt(0);
36    var closeBracket = ">";
37    var closeBracketCC = ">".charCodeAt(0);
38    var minus = "-";
39    var minusCC = "-".charCodeAt(0);
40    var slash = "/";
41    var slashCC = "/".charCodeAt(0);
42    var exclamation = '!';
43    var exclamationCC = '!'.charCodeAt(0);
44    var singleQuote = "'";
45    var singleQuoteCC = "'".charCodeAt(0);
46    var doubleQuote = '"';
47    var doubleQuoteCC = '"'.charCodeAt(0);
48
49    /**
50     * parsing a list of entries
51     */
52    function parseChildren() {
53        var children = [];
54        while (S[pos]) {
55            if (S.charCodeAt(pos) == openBracketCC) {
56                if (S.charCodeAt(pos + 1) === slashCC) {
57                    pos = S.indexOf(closeBracket, pos);
58                    if (pos + 1) pos += 1
59                    return children;
60                } else if (S.charCodeAt(pos + 1) === exclamationCC) {
61                    if (S.charCodeAt(pos + 2) == minusCC) {
62                        //comment support
63                        while (pos !== -1 && !(S.charCodeAt(pos) === closeBracketCC && S.charCodeAt(pos - 1) == minusCC && S.charCodeAt(pos - 2) == minusCC && pos != -1)) {
64                            pos = S.indexOf(closeBracket, pos + 1);
65                        }
66                        if (pos === -1) {
67                            pos = S.length
68                        }
69                    } else {
70                        // doctypesupport
71                        pos += 2;
72                        while (S.charCodeAt(pos) !== closeBracketCC && S[pos]) {
73                            pos++;
74                        }
75                    }
76                    pos++;
77                    continue;
78                }
79                var node = parseNode();
80                children.push(node);
81            } else {
82                var text = parseText()
83                if (text.trim().length > 0)
84                    children.push(text);
85                pos++;
86            }
87        }
88        return children;
89    }
90
91    /**
92     *    returns the text outside of texts until the first '<'
93     */
94    function parseText() {
95        var start = pos;
96        pos = S.indexOf(openBracket, pos) - 1;
97        if (pos === -2)
98            pos = S.length;
99        return S.slice(start, pos + 1);
100    }
101    /**
102     *    returns text until the first nonAlphebetic letter
103     */
104    var nameSpacer = '\n\t>/= ';
105
106    function parseName() {
107        var start = pos;
108        while (nameSpacer.indexOf(S[pos]) === -1 && S[pos]) {
109            pos++;
110        }
111        return S.slice(start, pos);
112    }
113    /**
114     *    is parsing a node, including tagName, Attributes and its children,
115     * to parse children it uses the parseChildren again, that makes the parsing recursive
116     */
117    var NoChildNodes = ['img', 'br', 'input', 'meta', 'link'];
118
119    function parseNode() {
120        var node = {};
121        pos++;
122        node.tagName = parseName();
123        // parsing attributes
124        var attrFound = false;
125        while (S.charCodeAt(pos) !== closeBracketCC && S[pos]) {
126            var c = S.charCodeAt(pos);
127            if ((c > 64 && c < 91) || (c > 96 && c < 123)) {
128                //if('abcdefghijklmnopqrstuvwxyzABCDEFGHIJKLMNOPQRSTUVWXYZ'.indexOf(S[pos])!==-1 ){
129                var name = parseName();
130                // search beginning of the string
131                var code = S.charCodeAt(pos);
132                while (code && code !== singleQuoteCC && code !== doubleQuoteCC && !((code > 64 && code < 91) || (code > 96 && code < 123)) && code !== closeBracketCC) {
133                    pos++;
134                    code = S.charCodeAt(pos);
135                }
136                if (!attrFound) {
137                    node.attributes = {};
138                    attrFound = true;
139                }
140                if (code === singleQuoteCC || code === doubleQuoteCC) {
141                    var value = parseString();
142                    if (pos === -1) {
143                        return node;
144                    }
145                } else {
146                    value = null;
147                    pos--;
148                }
149                node.attributes[name] = value;
150            }
151            pos++;
152
153        }
154        // optional parsing of children
155        if (S.charCodeAt(pos - 1) !== slashCC) {
156            if (node.tagName == "script") {
157                var start = pos + 1;
158                pos = S.indexOf('</' + 'script>', pos);
159                node.children = [S.slice(start, pos - 1)];
160                pos += 8;
161            } else if (node.tagName == "style") {
162                var start = pos + 1;
163                pos = S.indexOf('</style>', pos);
164                node.children = [S.slice(start, pos - 1)];
165                pos += 7;
166            } else if (NoChildNodes.indexOf(node.tagName) == -1) {
167                pos++;
168                node.children = parseChildren(name);
169            }
170        } else {
171            pos++;
172        }
173        return node;
174    }
175
176    /**
177     *    is parsing a string, that starts with a char and with the same usually  ' or "
178     */
179
180    function parseString() {
181        var startChar = S[pos];
182        var startpos = ++pos;
183        pos = S.indexOf(startChar, startpos)
184        return S.slice(startpos, pos);
185    }
186
187    /**
188     *
189     */
190
191    function findElements() {
192        var r = new RegExp('\\s' + options.attrName + '\\s*=[\'"]' + options.attrValue + '[\'"]').exec(S)
193        if (r) {
194            return r.index;
195        } else {
196            return -1;
197        }
198    }
199
200    var out = null;
201    if (options.attrValue !== undefined) {
202        options.attrName = options.attrName || 'id';
203        var out = [];
204
205        while ((pos = findElements()) !== -1) {
206            pos = S.lastIndexOf('<', pos);
207            if (pos !== -1) {
208                out.push(parseNode());
209            }
210            S = S.substr(pos);
211            pos = 0;
212        }
213    } else if (options.parseNode) {
214        out = parseNode()
215    } else {
216        out = parseChildren();
217    }
218
219    if (options.filter) {
220        out = tXml.filter(out, options.filter);
221    }
222
223    if (options.simplify) {
224        out = tXml.simplify(out);
225    }
226    out.pos = pos;
227    return out;
228}
229
230/**
231 * transform the DomObject to an object that is like the object of PHPs simplexmp_load_*() methods.
232 * this format helps you to write that is more likely to keep your programm working, even if there a small changes in the XML schema.
233 * be aware, that it is not possible to reproduce the original xml from a simplified version, because the order of elements is not saved.
234 * therefore your programm will be more flexible and easyer to read.
235 *
236 * @param {tNode[]} children the childrenList
237 */
238tXml.simplify = function simplify(children) {
239    var out = {};
240    if (!children.length) {
241        return '';
242    }
243
244    if (children.length === 1 && typeof children[0] == 'string') {
245        return children[0];
246    }
247    // map each object
248    children.forEach(function(child) {
249        if (typeof child !== 'object') {
250            return;
251        }
252        if (!out[child.tagName])
253            out[child.tagName] = [];
254        var kids = tXml.simplify(child.children||[]);
255        out[child.tagName].push(kids);
256        if (child.attributes) {
257            kids._attributes = child.attributes;
258        }
259    });
260
261    for (var i in out) {
262        if (out[i].length == 1) {
263            out[i] = out[i][0];
264        }
265    }
266
267    return out;
268};
269
270/**
271 * behaves the same way as Array.filter, if the filter method return true, the element is in the resultList
272 * @params children{Array} the children of a node
273 * @param f{function} the filter method
274 */
275tXml.filter = function(children, f) {
276    var out = [];
277    children.forEach(function(child) {
278        if (typeof(child) === 'object' && f(child)) out.push(child);
279        if (child.children) {
280            var kids = tXml.filter(child.children, f);
281            out = out.concat(kids);
282        }
283    });
284    return out;
285};
286
287/**
288 * stringify a previously parsed string object.
289 * this is useful,
290 *  1. to remove whitespaces
291 * 2. to recreate xml data, with some changed data.
292 * @param {tNode} O the object to Stringify
293 */
294tXml.stringify = function TOMObjToXML(O) {
295    var out = '';
296
297    function writeChildren(O) {
298        if (O)
299            for (var i = 0; i < O.length; i++) {
300                if (typeof O[i] == 'string') {
301                    out += O[i].trim();
302                } else {
303                    writeNode(O[i]);
304                }
305            }
306    }
307
308    function writeNode(N) {
309        out += "<" + N.tagName;
310        for (var i in N.attributes) {
311            if (N.attributes[i] === null) {
312                out += ' ' + i;
313            } else if (N.attributes[i].indexOf('"') === -1) {
314                out += ' ' + i + '="' + N.attributes[i].trim() + '"';
315            } else {
316                out += ' ' + i + "='" + N.attributes[i].trim() + "'";
317            }
318        }
319        out += '>';
320        writeChildren(N.children);
321        out += '</' + N.tagName + '>';
322    }
323    writeChildren(O);
324
325    return out;
326};
327
328
329/**
330 * use this method to read the textcontent, of some node.
331 * It is great if you have mixed content like:
332 * this text has some <b>big</b> text and a <a href=''>link</a>
333 * @return {string}
334 */
335tXml.toContentString = function(tDom) {
336    if (Array.isArray(tDom)) {
337        var out = '';
338        tDom.forEach(function(e) {
339            out += ' ' + tXml.toContentString(e);
340            out = out.trim();
341        });
342        return out;
343    } else if (typeof tDom === 'object') {
344        return tXml.toContentString(tDom.children)
345    } else {
346        return ' ' + tDom;
347    }
348};
349
350tXml.getElementById = function(S, id, simplified) {
351    var out = tXml(S, {
352        attrValue: id,
353        simplify: simplified
354    });
355    return simplified ? out : out[0];
356};
357/**
358 * A fast parsing method, that not realy finds by classname,
359 * more: the class attribute contains XXX
360 * @param
361 */
362tXml.getElementsByClassName = function(S, classname, simplified) {
363    return tXml(S, {
364        attrName: 'class',
365        attrValue: '[a-zA-Z0-9\-\s ]*' + classname + '[a-zA-Z0-9\-\s ]*',
366        simplify: simplified
367    });
368};
369
370tXml.parseStream = function(stream, offset) {
371    if (typeof offset === 'function') {
372        cb = offset;
373        offset = 0;
374    }
375    if (typeof offset === 'string') {
376        offset = offset.length + 2;
377    }
378    if (typeof stream === 'string') {
379        var fs = require('fs');
380        stream = fs.createReadStream(stream, { start: offset });
381        offset = 0;
382    }
383
384    var position = offset;
385    var data = '';
386    var cc = 0
387    stream.on('data', function(chunk) {
388        cc++;
389        data += chunk;
390        var lastpos = 0;
391        do {
392            position = data.indexOf('<', position) + 1
393            var res = tXml(data, { pos: position, parseNode: true });
394            position = res.pos;
395            if (position > (data.length - 1) || position < lastpos) {
396                if (lastpos) {
397                    data = data.slice(lastpos);
398                    position = 0
399                    lastpos = 0;
400                }
401                return;
402            } else {
403                stream.emit('xml', res);
404                lastpos = position;
405            }
406        } while (1)
407        data = data.slice(position);
408        position = 0;
409    });
410    stream.on('end', function() {
411        console.log('end')
412    });
413    return stream;
414}
415
416if ('object' === typeof module) {
417    module.exports = tXml;
418}
419//console.clear();
420//console.log('here:',tXml.getElementById('<some><xml id="test">dada</xml><that id="test">value</that></some>','test'));
421//console.log('here:',tXml.getElementsByClassName('<some><xml id="test" class="sdf test jsalf">dada</xml><that id="test">value</that></some>','test'));
422
423/*
424console.clear();
425tXml(d,'content');
426 //some testCode
427var s = document.body.innerHTML.toLowerCase();
428var start = new Date().getTime();
429var o = tXml(s,'content');
430var end = new Date().getTime();
431//console.log(JSON.stringify(o,undefined,'\t'));
432console.log("MILLISECONDS",end-start);
433var nodeCount=document.querySelectorAll('*').length;
434console.log('node count',nodeCount);
435console.log("speed:",(1000/(end-start))*nodeCount,'Nodes / second')
436//console.log(JSON.stringify(tXml('<html><head><title>testPage</title></head><body><h1>TestPage</h1><p>this is a <b>test</b>page</p></body></html>'),undefined,'\t'));
437var p = new DOMParser();
438var s2='<body>'+s+'</body>'
439var start2= new Date().getTime();
440var o2 = p.parseFromString(s2,'text/html').querySelector('#content')
441var end2=new Date().getTime();
442console.log("MILLISECONDS",end2-start2);
443// */

Line numbers count LF bytes from the start of the resource, as the search results do. Vendor segments are library code the classifier recognised; they are stored but not indexed. Bytes are shown as Latin1 characters, one per byte.