1/*! 2 Papa Parse 3 v4.1.2 4 https://github.com/mholt/PapaParse 5*/ 6(function(global) 7{ 8 "use strict"; 9 10 var IS_WORKER = !global.document && !!global.postMessage, 11 IS_PAPA_WORKER = IS_WORKER && /(\?|&)papaworker(=|&|$)/.test(global.location.search), 12 LOADED_SYNC = false, AUTO_SCRIPT_PATH; 13 var workers = {}, workerIdCounter = 0; 14 15 var Papa = {}; 16 17 Papa.parse = CsvToJson; 18 Papa.unparse = JsonToCsv; 19 20 Papa.RECORD_SEP = String.fromCharCode(30); 21 Papa.UNIT_SEP = String.fromCharCode(31); 22 Papa.BYTE_ORDER_MARK = "\ufeff"; 23 Papa.BAD_DELIMITERS = ["\r", "\n", "\"", Papa.BYTE_ORDER_MARK]; 24 Papa.WORKERS_SUPPORTED = !IS_WORKER && !!global.Worker; 25 Papa.SCRIPT_PATH = null; // Must be set by your code if you use workers and this lib is loaded asynchronously 26 27 // Configurable chunk sizes for local and remote files, respectively 28 Papa.LocalChunkSize = 1024 * 1024 * 10; // 10 MB 29 Papa.RemoteChunkSize = 1024 * 1024 * 5; // 5 MB 30 Papa.DefaultDelimiter = ","; // Used if not specified and detection fails 31 32 // Exposed for testing and development only 33 Papa.Parser = Parser; 34 Papa.ParserHandle = ParserHandle; 35 Papa.NetworkStreamer = NetworkStreamer; 36 Papa.FileStreamer = FileStreamer; 37 Papa.StringStreamer = StringStreamer; 38 39 if (typeof module !== 'undefined' && module.exports) 40 { 41 // Export to Node... 42 module.exports = Papa; 43 } 44 else if (isFunction(global.define) && global.define.amd) 45 { 46 // Wireup with RequireJS 47 define(function() { return Papa; }); 48 } 49 else 50 { 51 // ...or as browser global 52 global.Papa = Papa; 53 } 54 55 if (global.jQuery) 56 { 57 var $ = global.jQuery; 58 $.fn.parse = function(options) 59 { 60 var config = options.config || {}; 61 var queue = []; 62 63 this.each(function(idx) 64 { 65 var supported = $(this).prop('tagName').toUpperCase() == "INPUT" 66 && $(this).attr('type').toLowerCase() == "file" 67 && global.FileReader; 68 69 if (!supported || !this.files || this.files.length == 0) 70 return true; // continue to next input element 71 72 for (var i = 0; i < this.files.length; i++) 73 { 74 queue.push({ 75 file: this.files[i], 76 inputElem: this, 77 instanceConfig: $.extend({}, config) 78 }); 79 } 80 }); 81 82 parseNextFile(); // begin parsing 83 return this; // maintains chainability 84 85 86 function parseNextFile() 87 { 88 if (queue.length == 0) 89 { 90 if (isFunction(options.complete)) 91 options.complete(); 92 return; 93 } 94 95 var f = queue[0]; 96 97 if (isFunction(options.before)) 98 { 99 var returned = options.before(f.file, f.inputElem); 100 101 if (typeof returned === 'object') 102 { 103 if (returned.action == "abort") 104 { 105 error("AbortError", f.file, f.inputElem, returned.reason); 106 return; // Aborts all queued files immediately 107 } 108 else if (returned.action == "skip") 109 { 110 fileComplete(); // parse the next file in the queue, if any 111 return; 112 } 113 else if (typeof returned.config === 'object') 114 f.instanceConfig = $.extend(f.instanceConfig, returned.config); 115 } 116 else if (returned == "skip") 117 { 118 fileComplete(); // parse the next file in the queue, if any 119 return; 120 } 121 } 122 123 // Wrap up the user's complete callback, if any, so that ours also gets executed 124 var userCompleteFunc = f.instanceConfig.complete; 125 f.instanceConfig.complete = function(results) 126 { 127 if (isFunction(userCompleteFunc)) 128 userCompleteFunc(results, f.file, f.inputElem); 129 fileComplete(); 130 }; 131 132 Papa.parse(f.file, f.instanceConfig); 133 } 134 135 function error(name, file, elem, reason) 136 { 137 if (isFunction(options.error))
138 options.error({name: name}, file, elem, reason); 139 } 140 141 function fileComplete() 142 { 143 queue.splice(0, 1); 144 parseNextFile(); 145 } 146 } 147 } 148 149 150 if (IS_PAPA_WORKER) 151 { 152 global.onmessage = workerThreadReceivedMessage; 153 } 154 else if (Papa.WORKERS_SUPPORTED) 155 { 156 AUTO_SCRIPT_PATH = getScriptPath(); 157 158 // Check if the script was loaded synchronously 159 if (!document.body) 160 { 161 // Body doesn't exist yet, must be synchronous 162 LOADED_SYNC = true; 163 } 164 else 165 { 166 document.addEventListener('DOMContentLoaded', function () { 167 LOADED_SYNC = true; 168 }, true); 169 } 170 } 171 172 173 174 175 function CsvToJson(_input, _config) 176 { 177 _config = _config || {}; 178 179 if (_config.worker && Papa.WORKERS_SUPPORTED) 180 { 181 var w = newWorker(); 182 183 w.userStep = _config.step; 184 w.userChunk = _config.chunk; 185 w.userComplete = _config.complete; 186 w.userError = _config.error; 187 188 _config.step = isFunction(_config.step); 189 _config.chunk = isFunction(_config.chunk); 190 _config.complete = isFunction(_config.complete); 191 _config.error = isFunction(_config.error); 192 delete _config.worker; // prevent infinite loop 193 194 w.postMessage({ 195 input: _input, 196 config: _config, 197 workerId: w.id 198 }); 199 200 return; 201 } 202 203 var streamer = null; 204 if (typeof _input === 'string') 205 { 206 if (_config.download) 207 streamer = new NetworkStreamer(_config); 208 else 209 streamer = new StringStreamer(_config); 210 } 211 else if ((global.File && _input instanceof File) || _input instanceof Object) // ...Safari. (see issue #106) 212 streamer = new FileStreamer(_config); 213 214 return streamer.stream(_input); 215 } 216 217 218 219 220 221 222 function JsonToCsv(_input, _config) 223 { 224 var _output = ""; 225 var _fields = []; 226 227 // Default configuration 228 229 /** whether to surround every datum with quotes */ 230 var _quotes = false; 231 232 /** delimiting character */ 233 var _delimiter = ","; 234 235 /** newline character(s) */ 236 var _newline = "\r\n"; 237 238 unpackConfig(); 239 240 if (typeof _input === 'string') 241 _input = JSON.parse(_input); 242 243 if (_input instanceof Array) 244 { 245 if (!_input.length || _input[0] instanceof Array) 246 return serialize(null, _input); 247 else if (typeof _input[0] === 'object') 248 return serialize(objectKeys(_input[0]), _input); 249 } 250 else if (typeof _input === 'object') 251 { 252 if (typeof _input.data === 'string') 253 _input.data = JSON.parse(_input.data); 254 255 if (_input.data instanceof Array) 256 { 257 if (!_input.fields) 258 _input.fields = _input.data[0] instanceof Array 259 ? _input.fields 260 : objectKeys(_input.data[0]); 261 262 if (!(_input.data[0] instanceof Array) && typeof _input.data[0] !== 'object') 263 _input.data = [_input.data]; // handles input like [1,2,3] or ["asdf"] 264 } 265 266 return serialize(_input.fields || [], _input.data || []); 267 } 268 269 // Default (any valid paths should return before this) 270 throw "exception: Unable to serialize unrecognized input"; 271 272 273 function unpackConfig() 274 { 275 if (typeof _config !== 'object') 276 return; 277 278 if (typeof _config.delimiter === 'string' 279 && _config.delimiter.length == 1 280 && Papa.BAD_DELIMITERS.indexOf(_config.delimiter) == -1) 281 { 282 _delimiter = _config.delimiter; 283 } 284 285 if (typeof _config.quotes === 'boolean' 286 || _config.quotes instanceof Array) 287 _quotes = _config.quotes; 288 289 if (typeof _config.newline === 'string') 290 _newline = _config.newline; 291 } 292 293 294 /** Turns an object's keys into an array */ 295 function objectKeys(obj) 296 { 297 if (typeof obj !== 'object') 298 return []; 299 var keys = []; 300 for (var key in obj) 301 keys.push(key); 302 return keys; 303 } 304 305 /** The double for loop that iterates the data and writes out a CSV string including header row */ 306 function serialize(fields, data) 307 { 308 var csv = ""; 309 310 if (typeof fields === 'string') 311 fields = JSON.parse(fields); 312 if (typeof data === 'string') 313 data = JSON.parse(data); 314 315 var hasHeader = fields instanceof Array && fields.length > 0; 316 var dataKeyedByField = !(data[0] instanceof Array); 317 318 // If there a header row, write it first 319 if (hasHeader) 320 { 321 for (var i = 0; i < fields.length; i++) 322 { 323 if (i > 0) 324 csv += _delimiter; 325 csv += safe(fields[i], i); 326 } 327 if (data.length > 0) 328 csv += _newline; 329 } 330 331 // Then write out the data 332 for (var row = 0; row < data.length; row++) 333 { 334 var maxCol = hasHeader ? fields.length : data[row].length; 335 336 for (var col = 0; col < maxCol; col++) 337 { 338 if (col > 0) 339 csv += _delimiter; 340 var colIdx = hasHeader && dataKeyedByField ? fields[col] : col; 341 csv += safe(data[row][colIdx], col); 342 } 343 344 if (row < data.length - 1) 345 csv += _newline; 346 } 347 348 return csv; 349 } 350 351 /** Encloses a value around quotes if needed (makes a value safe for CSV insertion) */ 352 function safe(str, col) 353 { 354 if (typeof str === "undefined" || str === null) 355 return ""; 356 357 str = str.toString().replace(/"/g, '""'); 358 359 var needsQuotes = (typeof _quotes === 'boolean' && _quotes) 360 || (_quotes instanceof Array && _quotes[col]) 361 || hasAny(str, Papa.BAD_DELIMITERS) 362 || str.indexOf(_delimiter) > -1 363 || str.charAt(0) == ' ' 364 || str.charAt(str.length - 1) == ' '; 365 366 return needsQuotes ? '"' + str + '"' : str; 367 } 368 369 function hasAny(str, substrings) 370 { 371 for (var i = 0; i < substrings.length; i++) 372 if (str.indexOf(substrings[i]) > -1) 373 return true; 374 return false;
375 } 376 } 377 378 /** ChunkStreamer is the base prototype for various streamer implementations. */ 379 function ChunkStreamer(config) 380 { 381 this._handle = null; 382 this._paused = false; 383 this._finished = false; 384 this._input = null; 385 this._baseIndex = 0; 386 this._partialLine = ""; 387 this._rowCount = 0; 388 this._start = 0; 389 this._nextChunk = null; 390 this.isFirstChunk = true; 391 this._completeResults = { 392 data: [], 393 errors: [], 394 meta: {} 395 }; 396 replaceConfig.call(this, config); 397 398 this.parseChunk = function(chunk) 399 { 400 // First chunk pre-processing 401 if (this.isFirstChunk && isFunction(this._config.beforeFirstChunk)) 402 { 403 var modifiedChunk = this._config.beforeFirstChunk(chunk); 404 if (modifiedChunk !== undefined) 405 chunk = modifiedChunk; 406 } 407 this.isFirstChunk = false; 408 409 // Rejoin the line we likely just split in two by chunking the file 410 var aggregate = this._partialLine + chunk; 411 this._partialLine = ""; 412 413 var results = this._handle.parse(aggregate, this._baseIndex, !this._finished); 414 415 if (this._handle.paused() || this._handle.aborted()) 416 return; 417 418 var lastIndex = results.meta.cursor; 419 420 if (!this._finished) 421 { 422 this._partialLine = aggregate.substring(lastIndex - this._baseIndex); 423 this._baseIndex = lastIndex; 424 } 425 426 if (results && results.data) 427 this._rowCount += results.data.length; 428 429 var finishedIncludingPreview = this._finished || (this._config.preview && this._rowCount >= this._config.preview); 430 431 if (IS_PAPA_WORKER) 432 { 433 global.postMessage({ 434 results: results, 435 workerId: Papa.WORKER_ID, 436 finished: finishedIncludingPreview 437 }); 438 } 439 else if (isFunction(this._config.chunk)) 440 { 441 this._config.chunk(results, this._handle); 442 if (this._paused) 443 return; 444 results = undefined; 445 this._completeResults = undefined; 446 } 447 448 if (!this._config.step && !this._config.chunk) { 449 this._completeResults.data = this._completeResults.data.concat(results.data); 450 this._completeResults.errors = this._completeResults.errors.concat(results.errors); 451 this._completeResults.meta = results.meta; 452 } 453 454 if (finishedIncludingPreview && isFunction(this._config.complete) && (!results || !results.meta.aborted)) 455 this._config.complete(this._completeResults); 456 457 if (!finishedIncludingPreview && (!results || !results.meta.paused)) 458 this._nextChunk(); 459 460 return results; 461 }; 462 463 this._sendError = function(error) 464 { 465 if (isFunction(this._config.error)) 466 this._config.error(error); 467 else if (IS_PAPA_WORKER && this._config.error) 468 { 469 global.postMessage({ 470 workerId: Papa.WORKER_ID, 471 error: error, 472 finished: false 473 }); 474 } 475 }; 476 477 function replaceConfig(config) 478 { 479 // Deep-copy the config so we can edit it 480 var configCopy = copy(config); 481 configCopy.chunkSize = parseInt(configCopy.chunkSize); // parseInt VERY important so we don't concatenate strings! 482 if (!config.step && !config.chunk) 483 configCopy.chunkSize = null; // disable Range header if not streaming; bad values break IIS - see issue #196 484 this._handle = new ParserHandle(configCopy); 485 this._handle.streamer = this; 486 this._config = configCopy; // persist the copy to the caller 487 } 488 } 489 490 491 function NetworkStreamer(config) 492 { 493 config = config || {}; 494 if (!config.chunkSize) 495 config.chunkSize = Papa.RemoteChunkSize; 496 ChunkStreamer.call(this, config); 497 498 var xhr; 499 500 if (IS_WORKER) 501 { 502 this._nextChunk = function() 503 { 504 this._readChunk(); 505 this._chunkLoaded(); 506 }; 507 } 508 else 509 { 510 this._nextChunk = function() 511 { 512 this._readChunk(); 513 }; 514 } 515 516 this.stream = function(url) 517 { 518 this._input = url; 519 this._nextChunk(); // Starts streaming 520 }; 521 522 this._readChunk = function() 523 { 524 if (this._finished) 525 { 526 this._chunkLoaded(); 527 return; 528 } 529 530 xhr = new XMLHttpRequest(); 531 532 if (!IS_WORKER) 533 { 534 xhr.onload = bindFunction(this._chunkLoaded, this); 535 xhr.onerror = bindFunction(this._chunkError, this); 536 } 537 538 xhr.open("GET", this._input, !IS_WORKER); 539 540 if (this._config.chunkSize) 541 { 542 var end = this._start + this._config.chunkSize - 1; // minus one because byte range is inclusive 543 xhr.setRequestHeader("Range", "bytes="+this._start+"-"+end); 544 xhr.setRequestHeader("If-None-Match", "webkit-no-cache"); // https://bugs.webkit.org/show_bug.cgi?id=82672 545 } 546 547 try { 548 xhr.send(); 549 } 550 catch (err) { 551 this._chunkError(err.message); 552 } 553 554 if (IS_WORKER && xhr.status == 0) 555 this._chunkError(); 556 else 557 this._start += this._config.chunkSize; 558 } 559 560 this._chunkLoaded = function() 561 { 562 if (xhr.readyState != 4) 563 return; 564 565 if (xhr.status < 200 || xhr.status >= 400) 566 { 567 this._chunkError(); 568 return; 569 } 570 571 this._finished = !this._config.chunkSize || this._start > getFileSize(xhr); 572 this.parseChunk(xhr.responseText); 573 } 574 575 this._chunkError = function(errorMessage) 576 { 577 var errorText = xhr.statusText || errorMessage; 578 this._sendError(errorText); 579 } 580 581 function getFileSize(xhr) 582 { 583 var contentRange = xhr.getResponseHeader("Content-Range"); 584 return parseInt(contentRange.substr(contentRange.lastIndexOf("/") + 1)); 585 } 586 } 587 NetworkStreamer.prototype = Object.create(ChunkStreamer.prototype); 588 NetworkStreamer.prototype.constructor = NetworkStreamer; 589 590 591 function FileStreamer(config) 592 { 593 config = config || {}; 594 if (!config.chunkSize) 595 config.chunkSize = Papa.LocalChunkSize; 596 ChunkStreamer.call(this, config); 597 598 var reader, slice; 599 600 // FileReader is better than FileReaderSync (even in worker) - see http://stackoverflow.com/q/24708649/1048862 601 // But Firefox is a pill, too - see issue #76: https://github.com/mholt/PapaParse/issues/76 602 var usingAsyncReader = typeof FileReader !== 'undefined'; // Safari doesn't consider it a function - see issue #105 603 604 this.stream = function(file) 605 { 606 this._input = file; 607 slice = file.slice || file.webkitSlice || file.mozSlice; 608 609 if (usingAsyncReader) 610 { 611 reader = new FileReader(); // Preferred method of reading files, even in workers 612 reader.onload = bindFunction(this._chunkLoaded, this); 613 reader.onerror = bindFunction(this._chunkError, this); 614 } 615 else 616 reader = new FileReaderSync(); // Hack for running in a web worker in Firefox 617 618 this._nextChunk(); // Starts streaming 619 }; 620 621 this._nextChunk = function() 622 { 623 if (!this._finished && (!this._config.preview || this._rowCount < this._config.preview)) 624 this._readChunk(); 625 } 626 627 this._readChunk = function() 628 { 629 var input = this._input; 630 if (this._config.chunkSize) 631 { 632 var end = Math.min(this._start + this._config.chunkSize, this._input.size); 633 input = slice.call(input, this._start, end); 634 } 635 var txt = reader.readAsText(input, this._config.encoding); 636 if (!usingAsyncReader) 637 this._chunkLoaded({ target: { result: txt } }); // mimic the async signature 638 } 639 640 this._chunkLoaded = function(event) 641 { 642 // Very important to increment start each time before handling results 643 this._start += this._config.chunkSize; 644 this._finished = !this._config.chunkSize || this._start >= this._input.size; 645 this.parseChunk(event.target.result); 646 } 647 648 this._chunkError = function() 649 { 650 this._sendError(reader.error); 651 } 652 653 } 654 FileStreamer.prototype = Object.create(ChunkStreamer.prototype); 655 FileStreamer.prototype.constructor = FileStreamer; 656 657 658 function StringStreamer(config) 659 { 660 config = config || {}; 661 ChunkStreamer.call(this, config); 662 663 var string; 664 var remaining; 665 this.stream = function(s) 666 { 667 string = s; 668 remaining = s; 669 return this._nextChunk(); 670 } 671 this._nextChunk = function() 672 { 673 if (this._finished) return; 674 var size = this._config.chunkSize; 675 var chunk = size ? remaining.substr(0, size) : remaining; 676 remaining = size ? remaining.substr(size) : ''; 677 this._finished = !remaining; 678 return this.parseChunk(chunk); 679 } 680 } 681 StringStreamer.prototype = Object.create(StringStreamer.prototype); 682 StringStreamer.prototype.constructor = StringStreamer; 683 684 685 686 // Use one ParserHandle per entire CSV file or string 687 function ParserHandle(_config) 688 { 689 // One goal is to minimize the use of regular expressions... 690 var FLOAT = /^\s*-?(\d*\.?\d+|\d+\.?\d*)(e[-+]?\d+)?\s*$/i; 691 692 var self = this; 693 var _stepCounter = 0; // Number of times step was called (number of rows parsed) 694 var _input; // The input being parsed 695 var _parser; // The core parser being used 696 var _paused = false; // Whether we are paused or not 697 var _aborted = false; // Whether the parser has aborted or not 698 var _delimiterError; // Temporary state between delimiter detection and processing results 699 var _fields = []; // Fields are from the header row of the input, if there is one 700 var _results = { // The last results returned from the parser 701 data: [], 702 errors: [], 703 meta: {} 704 }; 705 706 if (isFunction(_config.step)) 707 { 708 var userStep = _config.step; 709 _config.step = function(results) 710 { 711 _results = results; 712 713 if (needsHeaderRow()) 714 processResults(); 715 else // only call user's step function after header row 716 { 717 processResults(); 718 719 // It's possbile that this line was empty and there's no row here after all 720 if (_results.data.length == 0) 721 return; 722 723 _stepCounter += results.data.length; 724 if (_config.preview && _stepCounter > _config.preview) 725 _parser.abort(); 726 else 727 userStep(_results, self); 728 } 729 }; 730 } 731 732 /** 733 * Parses input. Most users won't need, and shouldn't mess with, the baseIndex 734 * and ignoreLastRow parameters. They are used by streamers (wrapper functions) 735 * when an input comes in multiple chunks, like from a file. 736 */ 737 this.parse = function(input, baseIndex, ignoreLastRow) 738 { 739 if (!_config.newline) 740 _config.newline = guessLineEndings(input); 741 742 _delimiterError = false;
743 if (!_config.delimiter) 744 { 745 var delimGuess = guessDelimiter(input); 746 if (delimGuess.successful) 747 _config.delimiter = delimGuess.bestDelimiter; 748 else 749 { 750 _delimiterError = true; // add error after parsing (otherwise it would be overwritten) 751 _config.delimiter = Papa.DefaultDelimiter; 752 } 753 _results.meta.delimiter = _config.delimiter; 754 } 755 756 var parserConfig = copy(_config); 757 if (_config.preview && _config.header) 758 parserConfig.preview++; // to compensate for header row 759 760 _input = input; 761 _parser = new Parser(parserConfig); 762 _results = _parser.parse(_input, baseIndex, ignoreLastRow); 763 processResults(); 764 return _paused ? { meta: { paused: true } } : (_results || { meta: { paused: false } }); 765 }; 766 767 this.paused = function() 768 { 769 return _paused; 770 }; 771 772 this.pause = function() 773 { 774 _paused = true; 775 _parser.abort(); 776 _input = _input.substr(_parser.getCharIndex()); 777 }; 778 779 this.resume = function() 780 { 781 _paused = false; 782 self.streamer.parseChunk(_input); 783 }; 784 785 this.aborted = function () { 786 return _aborted; 787 } 788 789 this.abort = function() 790 { 791 _aborted = true; 792 _parser.abort(); 793 _results.meta.aborted = true; 794 if (isFunction(_config.complete)) 795 _config.complete(_results); 796 _input = ""; 797 }; 798 799 function processResults() 800 { 801 if (_results && _delimiterError) 802 { 803 addError("Delimiter", "UndetectableDelimiter", "Unable to auto-detect delimiting character; defaulted to '"+Papa.DefaultDelimiter+"'"); 804 _delimiterError = false; 805 } 806 807 if (_config.skipEmptyLines) 808 { 809 for (var i = 0; i < _results.data.length; i++) 810 if (_results.data[i].length == 1 && _results.data[i][0] == "") 811 _results.data.splice(i--, 1); 812 } 813 814 if (needsHeaderRow()) 815 fillHeaderFields(); 816 817 return applyHeaderAndDynamicTyping(); 818 } 819 820 function needsHeaderRow() 821 { 822 return _config.header && _fields.length == 0; 823 } 824 825 function fillHeaderFields() 826 { 827 if (!_results) 828 return; 829 for (var i = 0; needsHeaderRow() && i < _results.data.length; i++) 830 for (var j = 0; j < _results.data[i].length; j++) 831 _fields.push(_results.data[i][j]); 832 _results.data.splice(0, 1); 833 } 834 835 function applyHeaderAndDynamicTyping() 836 { 837 if (!_results || (!_config.header && !_config.dynamicTyping)) 838 return _results; 839 840 for (var i = 0; i < _results.data.length; i++) 841 { 842 var row = {}; 843 844 for (var j = 0; j < _results.data[i].length; j++) 845 { 846 if (_config.dynamicTyping) 847 { 848 var value = _results.data[i][j]; 849 if (value == "true" || value == "TRUE") 850 _results.data[i][j] = true; 851 else if (value == "false" || value == "FALSE") 852 _results.data[i][j] = false; 853 else 854 _results.data[i][j] = tryParseFloat(value); 855 } 856 857 if (_config.header) 858 { 859 if (j >= _fields.length) 860 { 861 if (!row["__parsed_extra"]) 862 row["__parsed_extra"] = []; 863 row["__parsed_extra"].push(_results.data[i][j]); 864 } 865 else 866 row[_fields[j]] = _results.data[i][j]; 867 } 868 } 869 870 if (_config.header) 871 { 872 _results.data[i] = row; 873 if (j > _fields.length) 874 addError("FieldMismatch", "TooManyFields", "Too many fields: expected " + _fields.length + " fields but parsed " + j, i); 875 else if (j < _fields.length) 876 addError("FieldMismatch", "TooFewFields", "Too few fields: expected " + _fields.length + " fields but parsed " + j, i); 877 } 878 } 879 880 if (_config.header && _results.meta) 881 _results.meta.fields = _fields; 882 return _results; 883 } 884 885 function guessDelimiter(input) 886 { 887 var delimChoices = [",", "\t", "|", ";", Papa.RECORD_SEP, Papa.UNIT_SEP]; 888 var bestDelim, bestDelta, fieldCountPrevRow; 889 890 for (var i = 0; i < delimChoices.length; i++) 891 { 892 var delim = delimChoices[i]; 893 var delta = 0, avgFieldCount = 0; 894 fieldCountPrevRow = undefined; 895 896 var preview = new Parser({ 897 delimiter: delim, 898 preview: 10 899 }).parse(input); 900 901 for (var j = 0; j < preview.data.length; j++) 902 { 903 var fieldCount = preview.data[j].length; 904 avgFieldCount += fieldCount; 905 906 if (typeof fieldCountPrevRow === 'undefined') 907 { 908 fieldCountPrevRow = fieldCount; 909 continue; 910 } 911 else if (fieldCount > 1) 912 { 913 delta += Math.abs(fieldCount - fieldCountPrevRow); 914 fieldCountPrevRow = fieldCount; 915 } 916 } 917 918 if (preview.data.length > 0) 919 avgFieldCount /= preview.data.length; 920 921 if ((typeof bestDelta === 'undefined' || delta < bestDelta) 922 && avgFieldCount > 1.99) 923 { 924 bestDelta = delta; 925 bestDelim = delim; 926 } 927 } 928 929 _config.delimiter = bestDelim; 930 931 return { 932 successful: !!bestDelim, 933 bestDelimiter: bestDelim 934 } 935 } 936 937 function guessLineEndings(input) 938 { 939 input = input.substr(0, 1024*1024); // max length 1 MB 940 941 var r = input.split('\r'); 942 943 if (r.length == 1) 944 return '\n'; 945 946 var numWithN = 0; 947 for (var i = 0; i < r.length; i++) 948 { 949 if (r[i][0] == '\n') 950 numWithN++; 951 } 952 953 return numWithN >= r.length / 2 ? '\r\n' : '\r'; 954 } 955 956 function tryParseFloat(val) 957 { 958 var isNumber = FLOAT.test(val); 959 return isNumber ? parseFloat(val) : val; 960 } 961
962 function addError(type, code, msg, row) 963 { 964 _results.errors.push({ 965 type: type, 966 code: code, 967 message: msg, 968 row: row 969 }); 970 } 971 } 972 973 974 975 976 977 /** The core parser implements speedy and correct CSV parsing */ 978 function Parser(config) 979 { 980 // Unpack the config object 981 config = config || {}; 982 var delim = config.delimiter; 983 var newline = config.newline; 984 var comments = config.comments; 985 var step = config.step; 986 var preview = config.preview; 987 var fastMode = config.fastMode; 988 989 // Delimiter must be valid 990 if (typeof delim !== 'string' 991 || Papa.BAD_DELIMITERS.indexOf(delim) > -1) 992 delim = ","; 993 994 // Comment character must be valid 995 if (comments === delim) 996 throw "Comment character same as delimiter"; 997 else if (comments === true) 998 comments = "#"; 999 else if (typeof comments !== 'string' 1000 || Papa.BAD_DELIMITERS.indexOf(comments) > -1) 1001 comments = false; 1002 1003 // Newline must be valid: \r, \n, or \r\n 1004 if (newline != '\n' && newline != '\r' && newline != '\r\n') 1005 newline = '\n'; 1006 1007 // We're gonna need these at the Parser scope 1008 var cursor = 0; 1009 var aborted = false; 1010 1011 this.parse = function(input, baseIndex, ignoreLastRow) 1012 { 1013 // For some reason, in Chrome, this speeds things up (!?) 1014 if (typeof input !== 'string') 1015 throw "Input must be a string"; 1016 1017 // We don't need to compute some of these every time parse() is called, 1018 // but having them in a more local scope seems to perform better 1019 var inputLen = input.length, 1020 delimLen = delim.length, 1021 newlineLen = newline.length, 1022 commentsLen = comments.length; 1023 var stepIsFunction = typeof step === 'function'; 1024 1025 // Establish starting state 1026 cursor = 0; 1027 var data = [], errors = [], row = [], lastCursor = 0; 1028 1029 if (!input) 1030 return returnable(); 1031 1032 if (fastMode || (fastMode !== false && input.indexOf('"') === -1)) 1033 { 1034 var rows = input.split(newline); 1035 for (var i = 0; i < rows.length; i++) 1036 { 1037 var row = rows[i]; 1038 cursor += row.length; 1039 if (i !== rows.length - 1) 1040 cursor += newline.length; 1041 else if (ignoreLastRow) 1042 return returnable(); 1043 if (comments && row.substr(0, commentsLen) == comments) 1044 continue; 1045 if (stepIsFunction) 1046 { 1047 data = []; 1048 pushRow(row.split(delim)); 1049 doStep(); 1050 if (aborted) 1051 return returnable(); 1052 } 1053 else 1054 pushRow(row.split(delim)); 1055 if (preview && i >= preview) 1056 { 1057 data = data.slice(0, preview); 1058 return returnable(true); 1059 } 1060 } 1061 return returnable(); 1062 } 1063 1064 var nextDelim = input.indexOf(delim, cursor); 1065 var nextNewline = input.indexOf(newline, cursor); 1066 1067 // Parser loop 1068 for (;;) 1069 { 1070 // Field has opening quote 1071 if (input[cursor] == '"') 1072 { 1073 // Start our search for the closing quote where the cursor is 1074 var quoteSearch = cursor; 1075 1076 // Skip the opening quote 1077 cursor++; 1078 1079 for (;;) 1080 { 1081 // Find closing quote 1082 var quoteSearch = input.indexOf('"', quoteSearch+1); 1083 1084 if (quoteSearch === -1) 1085 { 1086 if (!ignoreLastRow) { 1087 // No closing quote... what a pity 1088 errors.push({ 1089 type: "Quotes", 1090 code: "MissingQuotes", 1091 message: "Quoted field unterminated", 1092 row: data.length, // row has yet to be inserted 1093 index: cursor 1094 }); 1095 } 1096 return finish(); 1097 } 1098 1099 if (quoteSearch === inputLen-1) 1100 { 1101 // Closing quote at EOF 1102 var value = input.substring(cursor, quoteSearch).replace(/""/g, '"'); 1103 return finish(value); 1104 } 1105 1106 // If this quote is escaped, it's part of the data; skip it 1107 if (input[quoteSearch+1] == '"') 1108 { 1109 quoteSearch++; 1110 continue; 1111 } 1112 1113 if (input[quoteSearch+1] == delim) 1114 { 1115 // Closing quote followed by delimiter 1116 row.push(input.substring(cursor, quoteSearch).replace(/""/g, '"')); 1117 cursor = quoteSearch + 1 + delimLen; 1118 nextDelim = input.indexOf(delim, cursor); 1119 nextNewline = input.indexOf(newline, cursor); 1120 break; 1121 } 1122 1123 if (input.substr(quoteSearch+1, newlineLen) === newline) 1124 { 1125 // Closing quote followed by newline 1126 row.push(input.substring(cursor, quoteSearch).replace(/""/g, '"')); 1127 saveRow(quoteSearch + 1 + newlineLen); 1128 nextDelim = input.indexOf(delim, cursor); // because we may have skipped the nextDelim in the quoted field 1129 1130 if (stepIsFunction) 1131 { 1132 doStep(); 1133 if (aborted) 1134 return returnable(); 1135 } 1136 1137 if (preview && data.length >= preview) 1138 return returnable(true); 1139 1140 break; 1141 } 1142 } 1143 1144 continue; 1145 } 1146 1147 // Comment found at start of new line 1148 if (comments && row.length === 0 && input.substr(cursor, commentsLen) === comments) 1149 { 1150 if (nextNewline == -1) // Comment ends at EOF 1151 return returnable(); 1152 cursor = nextNewline + newlineLen; 1153 nextNewline = input.indexOf(newline, cursor); 1154 nextDelim = input.indexOf(delim, cursor); 1155 continue; 1156 } 1157 1158 // Next delimiter comes before next newline, so we've reached end of field 1159 if (nextDelim !== -1 && (nextDelim < nextNewline || nextNewline === -1)) 1160 { 1161 row.push(input.substring(cursor, nextDelim)); 1162 cursor = nextDelim + delimLen; 1163 nextDelim = input.indexOf(delim, cursor); 1164 continue; 1165 } 1166 1167 // End of row 1168 if (nextNewline !== -1) 1169 { 1170 row.push(input.substring(cursor, nextNewline)); 1171 saveRow(nextNewline + newlineLen); 1172 1173 if (stepIsFunction) 1174 { 1175 doStep(); 1176 if (aborted) 1177 return returnable(); 1178 } 1179 1180 if (preview && data.length >= preview) 1181 return returnable(true); 1182 1183 continue; 1184 } 1185 1186 break; 1187 } 1188 1189 1190 return finish(); 1191 1192 1193 function pushRow(row) 1194 { 1195 data.push(row);
1196 lastCursor = cursor; 1197 } 1198 1199 /** 1200 * Appends the remaining input from cursor to the end into 1201 * row, saves the row, calls step, and returns the results. 1202 */ 1203 function finish(value) 1204 { 1205 if (ignoreLastRow) 1206 return returnable(); 1207 if (typeof value === 'undefined') 1208 value = input.substr(cursor); 1209 row.push(value); 1210 cursor = inputLen; // important in case parsing is paused 1211 pushRow(row); 1212 if (stepIsFunction) 1213 doStep(); 1214 return returnable(); 1215 } 1216 1217 /** 1218 * Appends the current row to the results. It sets the cursor 1219 * to newCursor and finds the nextNewline. The caller should 1220 * take care to execute user's step function and check for 1221 * preview and end parsing if necessary. 1222 */ 1223 function saveRow(newCursor) 1224 { 1225 cursor = newCursor; 1226 pushRow(row); 1227 row = []; 1228 nextNewline = input.indexOf(newline, cursor); 1229 } 1230 1231 /** Returns an object with the results, errors, and meta. */ 1232 function returnable(stopped) 1233 { 1234 return { 1235 data: data, 1236 errors: errors, 1237 meta: { 1238 delimiter: delim, 1239 linebreak: newline, 1240 aborted: aborted, 1241 truncated: !!stopped, 1242 cursor: lastCursor + (baseIndex || 0) 1243 } 1244 }; 1245 } 1246 1247 /** Executes the user's step function and resets data & errors. */ 1248 function doStep() 1249 { 1250 step(returnable()); 1251 data = [], errors = []; 1252 } 1253 }; 1254 1255 /** Sets the abort flag */ 1256 this.abort = function() 1257 { 1258 aborted = true; 1259 }; 1260 1261 /** Gets the cursor position */ 1262 this.getCharIndex = function() 1263 { 1264 return cursor; 1265 }; 1266 } 1267 1268 1269 // If you need to load Papa Parse asynchronously and you also need worker threads, hard-code 1270 // the script path here. See: https://github.com/mholt/PapaParse/issues/87#issuecomment-57885358 1271 function getScriptPath() 1272 { 1273 var scripts = document.getElementsByTagName('script'); 1274 return scripts.length ? scripts[scripts.length - 1].src : ''; 1275 } 1276 1277 function newWorker() 1278 { 1279 if (!Papa.WORKERS_SUPPORTED) 1280 return false; 1281 if (!LOADED_SYNC && Papa.SCRIPT_PATH === null) 1282 throw new Error( 1283 'Script path cannot be determined automatically when Papa Parse is loaded asynchronously. ' + 1284 'You need to set Papa.SCRIPT_PATH manually.' 1285 ); 1286 var workerUrl = Papa.SCRIPT_PATH || AUTO_SCRIPT_PATH; 1287 // Append "papaworker" to the search string to tell papaparse that this is our worker. 1288 workerUrl += (workerUrl.indexOf('?') !== -1 ? '&' : '?') + 'papaworker'; 1289 var w = new global.Worker(workerUrl); 1290 w.onmessage = mainThreadReceivedMessage; 1291 w.id = workerIdCounter++; 1292 workers[w.id] = w; 1293 return w; 1294 } 1295 1296 /** Callback when main thread receives a message */ 1297 function mainThreadReceivedMessage(e) 1298 { 1299 var msg = e.data; 1300 var worker = workers[msg.workerId]; 1301 var aborted = false; 1302 1303 if (msg.error) 1304 worker.userError(msg.error, msg.file); 1305 else if (msg.results && msg.results.data) 1306 { 1307 var abort = function() { 1308 aborted = true; 1309 completeWorker(msg.workerId, { data: [], errors: [], meta: { aborted: true } }); 1310 }; 1311 1312 var handle = { 1313 abort: abort, 1314 pause: notImplemented, 1315 resume: notImplemented 1316 }; 1317 1318 if (isFunction(worker.userStep)) 1319 { 1320 for (var i = 0; i < msg.results.data.length; i++) 1321 { 1322 worker.userStep({ 1323 data: [msg.results.data[i]], 1324 errors: msg.results.errors, 1325 meta: msg.results.meta 1326 }, handle); 1327 if (aborted) 1328 break; 1329 } 1330 delete msg.results; // free memory ASAP 1331 } 1332 else if (isFunction(worker.userChunk)) 1333 { 1334 worker.userChunk(msg.results, handle, msg.file); 1335 delete msg.results; 1336 } 1337 } 1338 1339 if (msg.finished && !aborted) 1340 completeWorker(msg.workerId, msg.results); 1341 } 1342 1343 function completeWorker(workerId, results) { 1344 var worker = workers[workerId]; 1345 if (isFunction(worker.userComplete)) 1346 worker.userComplete(results); 1347 worker.terminate(); 1348 delete workers[workerId]; 1349 } 1350 1351 function notImplemented() { 1352 throw "Not implemented."; 1353 } 1354 1355 /** Callback when worker thread receives a message */ 1356 function workerThreadReceivedMessage(e) 1357 { 1358 var msg = e.data; 1359 1360 if (typeof Papa.WORKER_ID === 'undefined' && msg) 1361 Papa.WORKER_ID = msg.workerId; 1362 1363 if (typeof msg.input === 'string') 1364 { 1365 global.postMessage({ 1366 workerId: Papa.WORKER_ID, 1367 results: Papa.parse(msg.input, msg.config), 1368 finished: true 1369 }); 1370 } 1371 else if ((global.File && msg.input instanceof File) || msg.input instanceof Object) // thank you, Safari (see issue #106) 1372 { 1373 var results = Papa.parse(msg.input, msg.config); 1374 if (results) 1375 global.postMessage({ 1376 workerId: Papa.WORKER_ID, 1377 results: results, 1378 finished: true 1379 }); 1380 } 1381 } 1382
1383 /** Makes a deep copy of an array or object (mostly) */ 1384 function copy(obj) 1385 { 1386 if (typeof obj !== 'object') 1387 return obj; 1388 var cpy = obj instanceof Array ? [] : {}; 1389 for (var key in obj) 1390 cpy[key] = copy(obj[key]); 1391 return cpy; 1392 } 1393 1394 function bindFunction(f, self) 1395 { 1396 return function() { f.apply(self, arguments); }; 1397 } 1398 1399 function isFunction(func) 1400 { 1401 return typeof func === 'function'; 1402 } 1403})(typeof window !== 'undefined' ? window : this);
Line numbers count LF bytes from the start of the resource, as the search results do. Vendor segments are library code the classifier recognised; they are stored but not indexed. Bytes are shown as Latin1 characters, one per byte.