1(self.webpackChunk_N_E=self.webpackChunk_N_E||[]).push([[704],{6334:function(n,e,t){(window.__NEXT_P=window.__NEXT_P||[]).push(["/posts/js-to-html",function(){return t(919)}])},919:function(n,e,t){"use strict";t.r(e),t.d(e,{__N_SSG:function(){return j},default:function(){return L},meta:function(){return N}});var i=t(5893),r=t(1151),a=t(7294);function o(n){if("undefined"!==typeof Symbol&&null!=n[Symbol.iterator]||null!=n["@@iterator"])return Array.from(n)}function s(n,e){(null==e||e>n.length)&&(e=n.length);for(var t=0,i=new Array(e);t<e;t++)i[t]=n[t];return i}function c(n,e){if(n){if("string"===typeof n)return s(n,e);var t=Object.prototype.toString.call(n).slice(8,-1);return"Object"===t&&n.constructor&&(t=n.constructor.name),"Map"===t||"Set"===t?Array.from(t):"Arguments"===t||/^(?:Ui|I)nt(?:8|16|32)(?:Clamped)?Array$/.test(t)?s(n,e):void 0}}var l,u,h=(l=["Identifier","Keyword","Punctuator","Numeric","String","Whitespace","LineBreak","SingleLineComment","MultiLineComment"].map((function(n){return n})),u=9,function(n){if(Array.isArray(n))return n}(l)||o(l)||c(l,u)||function(){throw new TypeError("Invalid attempt to destructure non-iterable instance.\\nIn order to be iterable, non-array objects must have a [Symbol.iterator]() method.")}()),d=h[0],p=h[1],f=h[2],m=h[3],y=h[4],v=h[5],_=h[6],T=h[7],x=h[8],g=new Set(["for","do","while","if","else","return","function","var","let","const","true","false","undefined","this","new","delete","typeof","in","instanceof","void","break","continue","switch","case","default","throw","try","catch","finally","debugger","with","yield","async","await","class","extends","super","import","export","from","static","null"]),S=new Set(["_","$","a","b","c","d","e","f","g","h","i","j","k","l","m","n","o","p","q","r","s","t","u","v","w","x","y","z","A","B","C","D","E","F","G","H","I","J","K","L","M","N","O","P","Q","R","S","T","U","V","W","X","Y","Z"]),k=new Set(["0","1","2","3","4","5","6","7","8","9"].concat(function(n){return function(n){if(Array.isArray(n))return s(n)}(n)||o(n)||c(n)||function(){throw new TypeError("Invalid attempt to spread non-iterable instance.\\nIn order to be iterable, non-array objects must have a [Symbol.iterator]() method.")}()}(S))),b=new Set(["+","-","*","/","%","=","!","&","|","^","~","!","?",":",".",",",";","'",'"',".","(",")","[","]","#","@","\\","$","{","}"]),w=new Set([" ","\t"]);var I={Keyword:function(n){var e=n.token;return(0,i.jsx)("span",{className:"text-fuchsia-400","data-type":e.type,children:e.value})},Identifier:function(n){var e=n.token;return(0,i.jsx)("span",{className:"text-sky-400","data-type":e.type,children:e.value})},String:function(n){var e=n.token;return(0,i.jsx)("span",{className:"text-orange-400","data-type":e.type,children:e.value})},SingleLineComment:function(n){var e=n.token;return(0,i.jsx)("span",{className:"text-lime-400","data-type":e.type,children:e.value})},MultiLineComment:function(n){var e=n.token;return(0,i.jsx)("span",{className:"text-green-400","data-type":e.type,children:e.value})},Numeric:function(n){var e=n.token;return(0,i.jsx)("span",{className:"text-rose-400","data-type":e.type,children:e.value})},Default:function(n){var e=n.token;return(0,i.jsx)("span",{"data-type":e.type,children:e.value})}},E=function(n){var e=n.code,t=(0,a.useMemo)((function(){return function(n){for(var e,t=[],i=0,r=0,a=0,o=function(){for(var e=i,o=i+1;;){if("$"===n[o]&&"{"===n[o+1]){r++,o--;break}if("`"===n[o]){a--;break}o++}t.push({type:y,value:n.slice(e,o+1)}),i=o+1};i<n.length;){var s=n[i];if(a>r){if("`"===s){a--,t.push({type:y,value:"`"}),i++;continue}o()}else if("`"!==s)if(w.has(s)){for(var c=i,l=i+1;w.has(n[l]);)l++;t.push({type:v,value:n.slice(c,l)}),i=l}else if("\n"!==s){if("/"===s){var u=n[i+1];if("/"===u){for(var h=i,I=i+1;"\n"!==n[I];)I++;t.push({type:T,value:n.slice(h,I)}),i=I;continue}if("*"===u){for(var E=i,j=i+1;"*"!==n[j]||"/"!==n[j+1];)j++;t.push({type:x,value:n.slice(E,j+2)}),i=j+2;continue}}if("'"!==s&&'"'!==s)if(b.has(s))a>0&&a===r&&"}"===s&&r--,t.push({type:f,value:s}),i++;else if(S.has(s)){for(var N=i,C=i+1;e=n[C],k.has(e)||!(e<127)&&(8204==e||8205==e);)C++;var L=n.slice(N,C),A=g.has(L)?p:d;t.push({type:A,value:L}),i=C}else if(/\d/.test(s)){for(var M=i,R=i+1;/[1-9a-z_]/.test(n[R]);)R++;t.push({type:m,value:n.slice(M,R)}),i=R}else i++;else{for(var O=i,P=i+1;n[P]!==s;)P++;t.push({type:y,value:n.slice(O,P+1)}),i=P+1}}else t.push({type:_,value:s}),i++;else a++,o()}return t}(e)}),[e]);return(0,i.jsx)("pre",{className:"overflow-x-auto text-xs bg-gray-800 text-slate-50 p-2 rounded",children:t.map((function(n,e){var t=I[n.type]||I.Default;return(0,i.jsx)(t,{token:n},e)}))})},j=!0,N={path:"js-to-html",title:"JS-to-HTML Syntax highlighter",date:"24 October 2022"};function C(n){var e=Object.assign({h1:"h1",p:"p",a:"a",h2:"h2"},(0,r.ah)(),n.components);return(0,i.jsxs)(i.Fragment,{children:[(0,i.jsx)(e.h1,{children:"JS-to-HTML Syntax highlighter"}),"\n",(0,i.jsxs)(e.p,{children:["I stumbled upon ",(0,i.jsx)(e.a,{href:"https://github.com/huozhi/sugar-high",children:"sugar-high"}),", a lightweight JSX syntax highlighter, project on github.\nSince I have time on my hands, I thought why not go through the source code and try to implement a similar syntax highlighter myself.\nI wrote the tokenizer and implemented a simple JSHighlighter ",(0,i.jsx)(e.a,{href:"https://github.com/nibmz7/portfolio/blob/main/lib/JSHighlighter.jsx",children:"component"}),".\nThe final outcome is far from perfect and I can already imagine how much work it would be to implement a full fledged one like in a code editor."]}),"\n",(0,i.jsx)(e.h2,{children:"Result 1"}),"\n",(0,i.jsxs)(e.p,{children:["This is an example of the syntax highlighting on the tokenizer ",(0,i.jsx)(e.a,{href:"https://github.com/nibmz7/portfolio/blob/main/lib/tokenizer.js",children:"code"})," itself."]}),"\n",(0,i.jsx)(E,{code:'import {\n ValidIdentifierContinue,\n ValidIdentifierStart,\n Keywords,\n T_IDENTIFIER,\n Whitespace,\n T_WHITESPACE,\n T_LINEBREAK,\n T_SINGLE_LINE_COMMENT,\n T_MULTI_LINE_COMMENT,\n Puncuators,\n T_PUNCTUATOR,\n T_KEYWORD,\n T_STRING,\n T_NUMERIC,\n} from "./tokenizer.constants";\n\nfunction isIdentifierContinue(codePoint) {\n if (ValidIdentifierContinue.has(codePoint)) {\n return true;\n }\n\n // All ASCII identifier start code points are listed above\n if (codePoint < 0x7f) {\n return false;\n }\n\n // ZWNJ and ZWJ are allowed in identifiers\n if (codePoint == 0
1x200c || codePoint == 0x200d) {\n return true;\n }\n\n return false;\n}\n\nexport function tokenize(code) {\n const tokens = [];\n\n let i = 0;\n let __strTemplateExprStack = 0;\n let __strTemplateQuoteStack = 0;\n const inStrTemplateLiterals = () =>\n __strTemplateQuoteStack > __strTemplateExprStack;\n const inStrTemplateExpr = () =>\n __strTemplateQuoteStack > 0 &&\n __strTemplateQuoteStack === __strTemplateExprStack;\n const scanTemplateString = () => {\n const start = i;\n let end = i + 1;\n while (true) {\n if (code[end] === "$" && code[end + 1] === "{") {\n __strTemplateExprStack++;\n end--;\n break;\n }\n if (code[end] === "`") {\n __strTemplateQuoteStack--;\n break;\n }\n end++;\n }\n tokens.push({\n type: T_STRING,\n value: code.slice(start, end + 1),\n });\n i = end + 1;\n };\n\n while (i < code.length) {\n const char = code[i];\n\n if (inStrTemplateLiterals()) {\n if (char === "`") {\n __strTemplateQuoteStack--;\n tokens.push({\n type: T_STRING,\n value: "`",\n });\n i++;\n continue;\n }\n scanTemplateString();\n continue;\n }\n\n if (char === "`") {\n __strTemplateQuoteStack++;\n scanTemplateString();\n continue;\n }\n\n if (Whitespace.has(char)) {\n const start = i;\n let end = i + 1;\n while (Whitespace.has(code[end])) {\n end++;\n }\n tokens.push({\n type: T_WHITESPACE,\n value: code.slice(start, end),\n });\n i = end;\n continue;\n }\n\n if (char === "\\n") {\n tokens.push({\n type: T_LINEBREAK,\n value: char,\n });\n i++;\n continue;\n }\n\n if (char === "/") {\n const nextChar = code[i + 1];\n if (nextChar === "/") {\n const start = i;\n let end = i + 1;\n while (code[end] !== "\\n") {\n end++;\n }\n tokens.push({\n type: T_SINGLE_LINE_COMMENT,\n value: code.slice(start, end),\n });\n i = end;\n continue;\n }\n\n if (nextChar === "*") {\n const start = i;\n let end = i + 1;\n while (code[end] !== "*" || code[end + 1] !== "/") {\n end++;\n }\n tokens.push({\n type: T_MULTI_LINE_COMMENT,\n value: code.slice(start, end + 2),\n });\n i = end + 2;\n continue;\n }\n }\n\n if (char === "\'" || char === \'"\') {\n const start = i;\n let end = i + 1;\n while (code[end] !== char) {\n end++;\n }\n tokens.push({\n type: T_STRING,\n value: code.slice(start, end + 1),\n });\n i = end + 1;\n continue;\n }\n\n if (Puncuators.has(char)) {\n if (inStrTemplateExpr() && char === "}") {\n __strTemplateExprStack--;\n }\n\n tokens.push({\n type: T_PUNCTUATOR,\n value: char,\n });\n i++;\n continue;\n }\n\n if (ValidIdentifierStart.has(char)) {\n const start = i;\n let end = i + 1;\n while (isIdentifierContinue(code[end])) {\n end++;\n }\n const identifier = code.slice(start, end);\n const type = Keywords.has(identifier) ? T_KEYWORD : T_IDENTIFIER;\n tokens.push({ type, value: identifier });\n i = end;\n continue;\n }\n\n if (/\\d/.test(char)) {\n const start = i;\n let end = i + 1;\n while (/[1-9a-z_]/.test(code[end])) {\n end++;\n }\n tokens.push({\n type: T_NUMERIC,\n value: code.slice(start, end),\n });\n i = end;\n continue;\n }\n\n i++;\n }\n\n return tokens;\n}\n'}),"\n",(0,i.jsx)(e.h2,{children:"Result 2"}),"\n",(0,i.jsx)(e.p,{children:"Another example with string templates."}),"\n",(0,i.jsx)(E,{code:"`Hello ${lol + 'banana'} 7 ${`bye bye` + \"pineapple\"}`;"}),"\n",(0,i.jsx)(e.h2,{children:"Result 3"}),"\n",(0,i.jsx)(e.p,{children:"Last example with multiline comments."}),"\n",(0,i.jsx)(E,{code:"/**\n * This is a multi-line JavaScript comment\n */"})]})}var L=function(){var n=arguments.length>0&&void 0!==arguments[0]?arguments[0]:{},e=Object.assign({},(0,r.ah)(),n.components),t=e.wrapper;return t?(0,i.jsx)(t,Object.assign({},n,{children:(0,i.jsx)(C,n)})):C(n)}}},function(n){n.O(0,[774,888,179],(function(){return e=6334,n(n.s=e);var e}));var e=n.O();_N_E=e}]);
Line numbers count LF bytes from the start of the resource, as the search results do. Vendor segments are library code the classifier recognised; they are stored but not indexed. Bytes are shown as Latin1 characters, one per byte.