1 2"use strict"; 3 4/** 5 * Dient dem "intelligenten" Beschneiden von html-Texten. 6 * 7 * Copyright (c) 2016 Arend van Beelen jr., Speakap B.V. 8 */ 9class HRTextClipper 10{ 11 // Void elements are elements without inner content, 12 // which close themselves regardless of trailing slash. 13 // E.g. both <br> and <br /> are self-closing. 14 static VOID_ELEMENTS = [ 15 "area", 16 "base", 17 "br", 18 "col", 19 "command", 20 "embed", 21 "hr", 22 "img", 23 "input", 24 "keygen", 25 "link", 26 "meta", 27 "param", 28 "source", 29 "track", 30 "wbr", 31 ]; 32 33 // Block elements trigger newlines where they're inserted, 34 // and are always safe places for truncation. 35 static BLOCK_ELEMENTS = [ 36 "address", 37 "article", 38 "aside", 39 "blockquote", 40 "canvas", 41 "dd", 42 "div", 43 "dl", 44 "dt", 45 "fieldset", 46 "figcaption", 47 "figure", 48 "footer", 49 "form", 50 "h1", 51 "h2", 52 "h3", 53 "h4", 54 "h5", 55 "h6", 56 "header", 57 "hgroup", 58 "hr", 59 "li", 60 "main", 61 "nav", 62 "noscript", 63 "ol", 64 "output", 65 "p", 66 "pre", 67 "section", 68 "table", 69 "tbody", 70 "tfoot", 71 "thead", 72 "tr", 73 "ul", 74 "video", 75 ]; 76 77 // Elements that are unbreakable: they are either included verbatim, or omitted entirely. 78 static UNBREAKABLE_ELEMENTS = ["audio", "math", "svg", "video"]; 79 static NEWLINE_CHAR_CODE = 10; // '\n' 80 static EXCLAMATION_CHAR_CODE = 33; // '!' 81 static DOUBLE_QUOTE_CHAR_CODE = 34; // '"' 82 static AMPERSAND_CHAR_CODE = 38; // '&' 83 static SINGLE_QUOTE_CHAR_CODE = 39; // '\'' 84 static FORWARD_SLASH_CHAR_CODE = 47; // '/' 85 static SEMICOLON_CHAR_CODE = 59; // ';' 86 static TAG_OPEN_CHAR_CODE = 60; // '<' 87 static EQUAL_SIGN_CHAR_CODE = 61; // '=' 88 static TAG_CLOSE_CHAR_CODE = 62; // '>' 89 static CHAR_OF_INTEREST_REGEX = /[<&\n\ud800-\udbff]/; 90 static CHAR_OF_INTEREST_NO_NEWLINE_REGEX = /[<&\ud800-\udbff]/; 91 static SIMPLIFY_WHITESPACE_REGEX = /\s+/g; 92 93 /** 94 * Clips a string to a maximum length. If the string exceeds the length, it is truncated and an 95 * indicator (an ellipsis, by default) is appended. 96 * 97 * In detail, the clipping rules are as follows: 98 * - The resulting clipped string may never contain more than maxLength characters. Examples: 99 * - clip("foo", 3) => "foo" 100 * - clip("foo", 2) => "f�" 101 * - The indicator is inserted if and only if the string is clipped at any place other than a 102 * newline. Examples: 103 * - clip("foo bar", 5) => "foo �" 104 * - clip("foo\nbar", 5) => "foo" 105 * - If the html option is true and valid HTML is inserted, the clipped output *must* also be valid 106 * HTML. If the input is not valid HTML, the result is undefined (not to be confused with JS' 107 * "undefined" type; some errors might be detected and result in an exception, but this is not 108 * guaranteed). 109 * 110 * @param string The string to clip. 111 * @param maxLength The maximum length of the clipped string in number of characters. 112 * @param options Optional options object. 113 * 114 * @return The clipped string. 115 */ 116 static clip(string, maxLength, options) 117 { 118 if (options === void 0) { options = {}; } 119 if (!string) { 120 return ""; 121 } 122 string = string.toString(); 123 return options.html 124 ? this.clipHtml(string, maxLength, options) 125 : this.clipPlainText(string, maxLength, options); 126 } 127 128 static clipHtml(string, maxLength, options) 129 { 130 var _a = options.imageWeight, imageWeight = _a === void 0 ? 2 : _a, _b = options.indicator, indicator = _b === void 0 ? "\u2026" : _b, _c = options.maxLines, maxLines = _c === void 0 ? Infinity : _c, _d = options.stripTags, stripTags = _d === void 0 ? false : _d; 131 var numChars = indicator.length; 132 var numLines = 1; 133 var shouldStrip = typeof stripTags === "boolean" 134 ? function () { return stripTags; } 135 : function (tagName) { return stripTags.includes(tagName); }; 136 var tagStack = []; // Stack of currently open HTML tags. 137 var popTagStack = function (result) { 138 var tagName; 139 while (((tagName = tagStack.pop()), tagName !== undefined)) { 140 if (!shouldStrip(tagName)) { 141 result += "</".concat(tagName, ">"); 142 } 143 } 144 return result; 145 }; 146 var i = 0; 147 var unbreakableElementIndex = -1; 148 var length = string.length; 149 for (; i < length; i++) { 150 var rest = i ? string.slice(i) : string; 151 var willSimplifyWhiteSpace = this.shouldSimplifyWhiteSpace(tagStack); 152 var regex = unbreakableElementIndex > -1 || willSimplifyWhiteSpace 153 ? this.CHAR_OF_INTEREST_NO_NEWLINE_REGEX 154 : this.CHAR_OF_INTEREST_REGEX; 155 var nextIndex = rest.search(regex); 156 var nextBlockSize = nextIndex > -1 ? nextIndex : rest.length; 157 if (unbreakableElementIndex === -1) { 158 if (willSimplifyWhiteSpace) { 159 var simplifiedBlock = this.simplifyWhiteSpace(nextBlockSize === rest.length ? rest : rest.slice(0, nextIndex)); 160 if (shouldStrip(tagStack[tagStack.length - 1])) { 161 // We want to strip whitespace, but we need to insert spaces if stripping the 162 // tags and whitespace together would otherwise inadvertently concatenate words: 163 var insertSpaceBefore = i > 0 && !this.isWhiteSpace(string.charCodeAt(i - 1)); 164 var insertSpaceAfter = !this.isWhiteSpace(string.charCodeAt(i + nextBlockSize)); 165 if (simplifiedBlock.length > 0) { 166 simplifiedBlock = 167 (insertSpaceBefore ? " " : "") + 168 simplifiedBlock + 169 (insertSpaceAfter ? " " : ""); 170 } 171 else if (insertSpaceBefore && insertSpaceAfter) { 172 simplifiedBlock = " "; 173 }
174 string = string.slice(0, i) + simplifiedBlock + string.slice(i + nextBlockSize); 175 nextBlockSize = simplifiedBlock.length; 176 } 177 numChars += simplifiedBlock.length; 178 if (numChars > maxLength) { 179 break; 180 } 181 } 182 else { 183 numChars += nextBlockSize; 184 if (numChars > maxLength) { 185 i = Math.max(i + nextBlockSize - numChars + maxLength, 0); 186 break; 187 } 188 } 189 } 190 i += nextBlockSize; 191 if (nextIndex === -1) { 192 break; 193 } 194 var charCode = string.charCodeAt(i); 195 if (charCode === this.TAG_OPEN_CHAR_CODE) { 196 var nextCharCode = string.charCodeAt(i + 1); 197 var isSpecialTag = nextCharCode === this.EXCLAMATION_CHAR_CODE; 198 if (isSpecialTag && string.substr(i + 2, 2) === "--") { 199 var commentEndIndex = string.indexOf("-->", i + 4) + 3; 200 i = commentEndIndex - 1; // - 1 because the outer for loop will increment it 201 } 202 else if (isSpecialTag && string.substr(i + 2, 7) === "[CDATA[") { 203 var cdataEndIndex = string.indexOf("]]>", i + 9) + 3; 204 i = cdataEndIndex - 1; // - 1 because the outer for loop will increment it 205 // note we don't count CDATA text for our character limit because it is only 206 // allowed within SVG and MathML content, both of which we don't clip 207 } 208 else { 209 // don't open new tags if we are currently at the limit 210 var isEndTag = nextCharCode === this.FORWARD_SLASH_CHAR_CODE; 211 if (numChars === maxLength && !isEndTag) { 212 numChars++; 213 break; 214 } 215 var attributeQuoteCharCode = 0; 216 var endIndex = i; 217 var isAttributeValue = false; 218 while (true /* eslint-disable-line */) { 219 endIndex++; 220 if (endIndex >= length) { 221 throw new Error("Invalid HTML: ".concat(string)); 222 } 223 var charCode_1 = string.charCodeAt(endIndex); 224 if (isAttributeValue) { 225 if (attributeQuoteCharCode) { 226 if (charCode_1 === attributeQuoteCharCode) { 227 isAttributeValue = false; 228 } 229 } 230 else { 231 if (this.isWhiteSpace(charCode_1)) { 232 isAttributeValue = false; 233 } 234 else if (charCode_1 === this.TAG_CLOSE_CHAR_CODE) { 235 isAttributeValue = false; 236 endIndex--; // re-evaluate this character 237 } 238 } 239 } 240 else if (charCode_1 === this.EQUAL_SIGN_CHAR_CODE) { 241 while (this.isWhiteSpace(string.charCodeAt(endIndex + 1))) { 242 endIndex++;
242 // skip whitespace 243 } 244 isAttributeValue = true; 245 var firstAttributeCharCode = string.charCodeAt(endIndex + 1); 246 if (firstAttributeCharCode === this.DOUBLE_QUOTE_CHAR_CODE || 247 firstAttributeCharCode === this.SINGLE_QUOTE_CHAR_CODE) { 248 attributeQuoteCharCode = firstAttributeCharCode; 249 endIndex++; 250 } 251 else { 252 attributeQuoteCharCode = 0; 253 } 254 } 255 else if (charCode_1 === this.TAG_CLOSE_CHAR_CODE) { 256 var tagNameStartIndex = i + (isEndTag ? 2 : 1); 257 var tagNameEndIndex = Math.min(this.indexOfWhiteSpace(string, tagNameStartIndex), endIndex); 258 var tagName = string 259 .slice(tagNameStartIndex, tagNameEndIndex) 260 .toLowerCase(); 261 if (tagName.charCodeAt(tagName.length - 1) === this.FORWARD_SLASH_CHAR_CODE) { 262 // Remove trailing slash for self-closing tag names like <br/> 263 tagName = tagName.slice(0, tagName.length - 1); 264 } 265 var strip = shouldStrip(tagName); 266 if (isEndTag) { 267 var currentTagName = tagStack.pop(); 268 if (currentTagName !== tagName) { 269 throw new Error("Invalid HTML: ".concat(string)); 270 } 271 if (this.UNBREAKABLE_ELEMENTS.includes(tagName)) { 272 if (this.UNBREAKABLE_ELEMENTS.some(function (tagName) { 273 return tagStack.includes(tagName); 274 })) { 275 // It's a nested unbreakable element. 276 } 277 else if (strip) { 278 i = unbreakableElementIndex; 279 unbreakableElementIndex = -1; 280 } 281 else { 282 unbreakableElementIndex = -1; 283 numChars += imageWeight; 284 if (numChars > maxLength) { 285 break; 286 } 287 } 288 } 289 // Block level elements should trigger a new line, unless stripped or 290 // part of unbreakable content. 291 var isBlockElement = this.BLOCK_ELEMENTS.includes(tagName); 292 if (isBlockElement && unbreakableElementIndex === -1 && !strip) { 293 numLines++; 294 if (numLines > maxLines) { 295 // If we exceed the max lines, push the tag back onto the 296 // stack so that it will be added back correctly after 297 // truncation. 298 tagStack.push(tagName); 299 break; 300 } 301 } 302 } 303 else if (this.VOID_ELEMENTS.includes(tagName) || 304 string.charCodeAt(endIndex - 1) === this.FORWARD_SLASH_CHAR_CODE) { 305 if (strip) { 306 // Stripped elements aren't counted towards anything. 307 } 308 else if (tagName === "br") { 309 numLines++; 310 if (numLines > maxLines) { 311 break; 312 } 313 } 314 else if (tagName === "img") { 315 numChars += imageWeight; 316 if (numChars > maxLength) { 317 break; 318 } 319 } 320 } 321 else { 322 if (this.UNBREAKABLE_ELEMENTS.some(function (tagName) { return tagStack.includes(tagName); })) { 323 // It's a nested unbreakable element. 324 } 325 else if (this.UNBREAKABLE_ELEMENTS.includes(tagName)) { 326 unbreakableElementIndex = i; 327 } 328 tagStack.push(tagName); 329 } 330 if (strip && unbreakableElementIndex === -1) {
331 string = string.slice(0, i) + string.slice(endIndex + 1); 332 i--; // Re-evaluate this index, because its contents changed. 333 } 334 else { 335 i = endIndex; 336 } 337 break; 338 } 339 } 340 if (numChars > maxLength || numLines > maxLines) { 341 break; 342 } 343 } 344 } 345 else if (charCode === this.AMPERSAND_CHAR_CODE) { 346 var endIndex = i + 1; 347 var isCharacterReference = true; 348 while (true /* eslint-disable-line */) { 349 var charCode_2 = string.charCodeAt(endIndex); 350 if (this.isCharacterReferenceCharacter(charCode_2)) { 351 endIndex++; 352 } 353 else if (charCode_2 === this.SEMICOLON_CHAR_CODE) { 354 break; 355 } 356 else { 357 isCharacterReference = false; 358 break; 359 } 360 } 361 if (unbreakableElementIndex === -1) { 362 numChars++; 363 if (numChars > maxLength) { 364 break; 365 } 366 } 367 if (isCharacterReference) { 368 i = endIndex; 369 } 370 } 371 else if (charCode === this.NEWLINE_CHAR_CODE) { 372 numChars++; 373 if (numChars > maxLength) { 374 break; 375 } 376 numLines++; 377 if (numLines > maxLines) { 378 break; 379 } 380 } 381 else { 382 if (unbreakableElementIndex === -1) { 383 numChars++; 384 if (numChars > maxLength) { 385 break; 386 } 387 } 388 if ((charCode & 0xfc00) === 0xd800) { 389 // high Unicode surrogate should never be separated from its matching low surrogate 390 var nextCharCode = string.charCodeAt(i + 1); 391 if ((nextCharCode & 0xfc00) === 0xdc00) { 392 i++; 393 } 394 } 395 } 396 } 397 if (numChars > maxLength) { 398 var nextChar = this.takeHtmlCharAt(string, i); 399 if (indicator) { 400 var peekIndex = i + nextChar.length; 401 while (string.charCodeAt(peekIndex) === this.TAG_OPEN_CHAR_CODE && 402 string.charCodeAt(peekIndex + 1) === this.FORWARD_SLASH_CHAR_CODE) { 403 var nextPeekIndex = string.indexOf(">", peekIndex + 2) + 1; 404 if (nextPeekIndex) { 405 peekIndex = nextPeekIndex; 406 } 407 else { 408 break; 409 } 410 } 411 if (peekIndex && (peekIndex === string.length || this.isLineBreak(string, peekIndex))) { 412 // if there's only a single character remaining in the input string, or the next 413 // character is followed by a line-break, we can include it instead of the clipping 414 // indicator (provided it's not a special HTML character) 415 i += nextChar.length; 416 nextChar = string.charAt(i); 417 } 418 } 419 // include closing tags before adding the clipping indicator if that's where they 420 // are in the input string 421 while (nextChar === "<" && string.charCodeAt(i + 1) === this.FORWARD_SLASH_CHAR_CODE) { 422 var tagName = tagStack.pop(); 423 if (!tagName) { 424 break; 425 } 426 var tagEndIndex = string.indexOf(">", i + 2); 427 if (tagEndIndex === -1 || string.slice(i + 2, tagEndIndex).trim() !== tagName) { 428 throw new Error("Invalid HTML: ".concat(string)); 429 } 430 if (shouldStrip(tagName)) {
431 string = string.slice(0, i) + string.slice(tagEndIndex + 1); 432 } 433 else { 434 i = tagEndIndex + 1; 435 } 436 nextChar = string.charAt(i); 437 } 438 if (i < string.length) { 439 if (!options.breakWords) { 440 // try to clip at word boundaries, if desired 441 for (var j = i - indicator.length; j >= 0; j--) { 442 var charCode = string.charCodeAt(j); 443 if (charCode === this.TAG_CLOSE_CHAR_CODE || charCode === this.SEMICOLON_CHAR_CODE) { 444 // these characters could be just regular characters, so if they occur in 445 // the middle of a word, they would "break" our attempt to prevent breaking 446 // of words, but given this seems highly unlikely and the alternative is 447 // doing another full parsing of the preceding text, this seems acceptable. 448 break; 449 } 450 else if (charCode === this.NEWLINE_CHAR_CODE || charCode === this.TAG_OPEN_CHAR_CODE) { 451 i = j; 452 break; 453 } 454 else if (this.isWhiteSpace(charCode)) { 455 i = j + (indicator ? 1 : 0); 456 break; 457 } 458 } 459 } 460 var result = string.slice(0, i); 461 if (!this.isLineBreak(string, i)) { 462 result += indicator; 463 } 464 return popTagStack(result); 465 } 466 } 467 else if (numLines > maxLines) { 468 return popTagStack(string.slice(0, i)); 469 } 470 return string; 471 }; 472 473 static clipPlainText(string, maxLength, options) 474 { 475 var _a = options.indicator, indicator = _a === void 0 ? "\u2026" : _a, _b = options.maxLines, maxLines = _b === void 0 ? Infinity : _b; 476 var numChars = indicator.length; 477 var numLines = 1; 478 var i = 0; 479 var length = string.length; 480 for (; i < length; i++) { 481 numChars++; 482 if (numChars > maxLength) { 483 break; 484 } 485 var charCode = string.charCodeAt(i); 486 if (charCode === this.NEWLINE_CHAR_CODE) { 487 numLines++; 488 if (numLines > maxLines) { 489 break; 490 } 491 } 492 else if ((charCode & 0xfc00) === 0xd800) { 493 // high Unicode surrogate should never be separated from its matching low surrogate 494 var nextCharCode = string.charCodeAt(i + 1); 495 if ((nextCharCode & 0xfc00) === 0xdc00) { 496 i++; 497 } 498 } 499 } 500 if (numChars > maxLength) { 501 var nextChar = this.takeCharAt(string, i); 502 if (indicator) { 503 var peekIndex = i + nextChar.length; 504 if (peekIndex === string.length) { 505 return string; 506 } 507 else if (string.charCodeAt(peekIndex) === this.NEWLINE_CHAR_CODE) { 508 return string.slice(0, i + nextChar.length); 509 } 510 } 511 if (!options.breakWords) { 512 // try to clip at word boundaries, if desired 513 for (var j = i - indicator.length; j >= 0; j--) { 514 var charCode = string.charCodeAt(j); 515 if (charCode === this.NEWLINE_CHAR_CODE) { 516 i = j; 517 nextChar = "\n"; 518 break; 519 } 520 else if (this.isWhiteSpace(charCode)) { 521 i = j + (indicator ? 1 : 0); 522 break; 523 } 524 } 525 } 526 return string.slice(0, i) + (nextChar === "\n" ? "" : indicator); 527 } 528 else if (numLines > maxLines) { 529 return string.slice(0, i); 530 } 531 return string; 532 }; 533 534 static indexOfWhiteSpace(string, fromIndex) 535 { 536 var length = string.length; 537 for (var i = fromIndex; i < length; i++) { 538 if (this.isWhiteSpace(string.charCodeAt(i))) { 539 return i; 540 } 541 } 542 // Rather than -1, this function returns the length of the string if no match is found, 543 // so it works well with the Math.min() usage above: 544 return length; 545 }; 546 547 static isCharacterReferenceCharacter(charCode) 548 { 549 return ((charCode >
549= 48 && charCode <= 57) || 550 (charCode >= 65 && charCode <= 90) || 551 (charCode >= 97 && charCode <= 122)); 552 }; 553 554 static isLineBreak(string, index) 555 { 556 var firstCharCode = string.charCodeAt(index); 557 if (firstCharCode === this.NEWLINE_CHAR_CODE) { 558 return true; 559 } 560 else if (firstCharCode === this.TAG_OPEN_CHAR_CODE) { 561 var newlineElements = "(".concat(this.BLOCK_ELEMENTS.join("|"), "|br)"); 562 var newlineRegExp = new RegExp("^<".concat(newlineElements, "[\t\n\f\r ]*" + "/?>"), "i"); 563 return newlineRegExp.test(string.slice(index)); 564 } 565 else { 566 return false; 567 } 568 } 569 570 static isWhiteSpace(charCode) 571 { 572 return (charCode === 9 || charCode === 10 || charCode === 12 || charCode === 13 || charCode === 32); 573 }; 574 575 /** 576 * Certain tags don't display their whitespace-only content. In such cases, we 577 * should simplify the whitespace before counting it. 578 */ 579 static shouldSimplifyWhiteSpace(tagStack) { 580 for (var i = tagStack.length - 1; i >= 0; i--) { 581 var tagName = tagStack[i]; 582 if (tagName === "li" || tagName === "td") { 583 return false; 584 } 585 if (tagName === "ol" || tagName === "table" || tagName === "ul") { 586 return true; 587 } 588 } 589 return false; 590 }; 591 592 static simplifyWhiteSpace(string) 593 { 594 return string.trim().replace(this.SIMPLIFY_WHITESPACE_REGEX, " "); 595 }; 596 597 static takeCharAt(string, index) 598 { 599 var charCode = string.charCodeAt(index); 600 if ((charCode & 0xfc00) === 0xd800) { 601 // high Unicode surrogate should never be separated from its matching low surrogate 602 var nextCharCode = string.charCodeAt(index + 1); 603 if ((nextCharCode & 0xfc00) === 0xdc00) { 604 return String.fromCharCode(charCode, nextCharCode); 605 } 606 } 607 return String.fromCharCode(charCode); 608 }; 609 610 static takeHtmlCharAt(string, index) 611 { 612 var char = this.takeCharAt(string, index); 613 if (char === "&") { 614 while (true /* eslint-disable-line */) { 615 index++; 616 var nextCharCode = string.charCodeAt(index); 617 if (this.isCharacterReferenceCharacter(nextCharCode)) { 618 char += String.fromCharCode(nextCharCode); 619 } 620 else if (nextCharCode === this.SEMICOLON_CHAR_CODE) { 621 char += String.fromCharCode(nextCharCode); 622 break; 623 } 624 else { 625 break; 626 } 627 } 628 } 629 return char; 630 }; 631}
Line numbers count LF bytes from the start of the resource, as the search results do. Vendor segments are library code the classifier recognised; they are stored but not indexed. Bytes are shown as Latin1 characters, one per byte.