PageSourceSearch

https://tim-paik.github.io/js/search.js

js tim-paik.github.io collected 2026-10-03 10:05:13 UTC 8,475 bytes, 234 lines download raw bytes

1/*
2 * A local search script for [hexo-generator-search](https://github.com/wzpan/hexo-generator-search)
3 * CopyLeft (C) 2015-2018
4 * Joseph Pan <http://github.com/wzpan>
5 * Shuhao Mao <http://github.com/maoshuhao>
6 * Edited by MOxFIVE <http://github.com/MOxFIVE>
7 * Rewrited by AlynxZhou <https://alynx.xyz/>
8 *   Cleaned: Use native JavaScript instead of jQuert. Split functions.
9 *   Fixed: Mark all keywords found in content and title.
10 *   Optimized: Sort result by the number of keyword found.
11 */
12
13"use strict";
14
15var SUBSTRING_OFFSET = 15;
16var MAX_KEYWORDS = 30;
17var MAX_DISPLAY_SLICES = 5;
18
19// Calculate how many keywords a page contains.
20function findKeywords(keywords, prop) {
21  for (var i = 0; i < keywords.length; ++i) {
22    var indexContent = prop["dataContent"].toLowerCase().indexOf(keywords[i].toLowerCase());
23    // Find all keyword indices.
24    while (indexContent >= 0) {
25      prop["matchedContentKeywords"].push({
26        "keyword": prop["dataContent"].substring(indexContent, indexContent + keywords[i].length),
27        "index": indexContent
28      });
29      indexContent = prop["dataContent"].toLowerCase().indexOf(keywords[i].toLowerCase(), indexContent + keywords[i].length);
30    }
31    var indexTitle = prop["dataTitle"].toLowerCase().indexOf(keywords[i].toLowerCase());
32    while (indexTitle >= 0) {
33      prop["matchedTitleKeywords"].push({
34        "keyword": prop["dataTitle"].substring(indexTitle, indexTitle + keywords[i].length),
35        "index": indexTitle
36      });
37      indexTitle = prop["dataTitle"].toLowerCase().indexOf(keywords[i].toLowerCase(), indexTitle + keywords[i].length);
38    }
39  }
40}
41
42function buildSortedMatchedDataProps(datas, keywords) {
43  var matchedDataProps = [];
44  for (var i = 0; i < datas.length; ++i) {
45    var prop = {
46      "matchedContentKeywords": [],
47      "matchedTitleKeywords": [],
48      "dataTitle": datas[i]["title"].trim(),
49      "dataContent": datas[i]["content"].trim().replace(/<[^>]+>/g, ""),
50      "dataURL": datas[i]["url"]
51    };
52    // Only match articles with valid titles and contents.
53    if (prop["dataTitle"].length + prop["dataContent"].length > 0) {
54      findKeywords(keywords, prop);
55    }
56    if (prop["matchedContentKeywords"].length + prop["matchedTitleKeywords"].length > 0) {
57      matchedDataProps.push(prop);
58    }
59  }
60  // The more keywords a page contains, the higher this page ranks.
61  matchedDataProps.sort(function (a, b) {
62    return -((a["matchedContentKeywords"].length + a["matchedTitleKeywords"].length) - (b["matchedContentKeywords"].length + b["matchedTitleKeywords"].length));
63  });
64  return matchedDataProps;
65}
66
67function buildSortedSliceArray(prop) {
68  var sliceArray = [];
69  // Sorting slice array is hard so sort index array instead.
70  prop["matchedContentKeywords"].sort(function (a, b) {
71    return a["index"] - b["index"]
72  });
73  // Get content slice position.
74  for (var i = 0; i < prop["matchedContentKeywords"].length && i < MAX_DISPLAY_SLICES; ++i) {
75    var start = prop["matchedContentKeywords"][i]["index"] - SUBSTRING_OFFSET;
76    var end = prop["matchedContentKeywords"][i]["index"] + prop["matchedContentKeywords"][i]["keyword"].length + SUBSTRING_OFFSET;
77    if (start < 0) {
78      start = 0;
79    }
80    if (start === 0) {
81      end = SUBSTRING_OFFSET + prop["matchedContentKeywords"][i]["keyword"].length + SUBSTRING_OFFSET;
82    }
83    if (end > prop["dataContent"].length) {
84      end = prop["dataContent"].length;
85    }
86    sliceArray.push({ "start": start, "end": end });
87  }
88  return sliceArray;
89}
90
91function mergeSliceArray(sliceArray) {
92  var mergedSliceArray = [];
93  if (sliceArray.length === 0) {
94    return mergedSliceArray;
95  }
96  mergedSliceArray.push(sliceArray[0])
97  for (var i = 1; i < sliceArray.length; ++i) {
98    // If two slice have common part, merge them.
99    if (mergedSliceArray[mergedSliceArray.length - 1]["end"] >= sliceArray[i]["start"]) {
100      if (sliceArray[i]["end"] > mergedSliceArray[mergedSliceArray.length - 1]["end"]) {
101        mergedSliceArray[mergedSliceArray.length - 1]["end"] = sliceArray[i]["end"];
102      }
103    } else {
104      mergedSliceArray.push(sliceArray[i]);
105    }
106  }
107  return mergedSliceArray;
108}
109
110function buildHighlightedTitle(prop) {
111  var matchedTitle = prop["dataTitle"];
112  var reArray = [];
113  for (var i = 0; i < prop["matchedTitleKeywords"].length; ++i) {
114    if (prop["matchedTitleKeywords"][i]["keyword"].length > 0) {
115      reArray.push(prop["matchedTitleKeywords"][i]["keyword"])
116    }
117  }
118  // Replace all in one time to prevent it from matching <strong> tag.
119  var re = new RegExp(reArray.join("|"), "gi");
120  // `$&` is the matched part of RegExp.
121  matchedTitle = matchedTitle.replace(re, "<strong class=\"search-keyword\">$&</strong>");
122  return matchedTitle;
123}
124
125function buildHighlightedContent(prop, mergedSliceArray) {
126  var matchedContentArray = [];
127  for (var i = 0; i < mergedSliceArray.length; ++i) {
128    matchedContentArray.push(prop["dataContent"].substring(mergedSliceArray[i]["start"], mergedSliceArray[i]["end"]));
129  }
130  var reArray = [];
131  for (var i = 0; i < prop["matchedContentKeywords"].length; ++i) {
132    if (prop["matchedContentKeywords"][i]["keyword"].length > 0) {
133      reArray.push(prop["matchedContentKeywords"][i]["keyword"]);
134    }
135  }
136  var re = new RegExp(reArray.join("|"), "gi");
137  for (var i = 0; i < matchedContentArray.length; i++) {
138    matchedContentArray[i] = matchedContentArray[i].replace(re, "<strong class=\"search-keyword\">$&</strong>");
139  }
140  return matchedContentArray.join("...");
141}
142
143function onInput(input, resultContent, datas) {
144  resultContent.innerHTML = "";
145  if (input.value.trim().length <= 0) {
146    return;
147  }
148  var keywords = input.value.trim().split(/[\s-\+]+/);
149  var li = [];
150  if (keywords.length > MAX_KEYWORDS) {
151    keywords = keywords.slice(0, MAX_KEYWORDS);
152    // li.push("<span>Keywords more than ");
153    // li.push(MAX_KEYWORDS);
154    // li.push(" are sliced.</span>");
155  }
156  var matchedDataProps = buildSortedMatchedDataProps(datas, keywords);
157  if (matchedDataProps.length === 0) {
158    return;
159  }
160  li.push("<ul class=\"search-result-list\">");
161  for (var i = 0; i < matchedDataProps.length; ++i) {
162    // Show search results
163    li.push("<li><a href=\"")
164    li.push(matchedDataProps[i]["dataURL"]);
165    li.push("\" class=\"search-result-title\">");
166    li.push(buildHighlightedTitle(matchedDataProps[i]));
167    li.push("</a>");
168    li.push("<p class=\"search-result-content\">...");
169    var sliceArray = buildSortedSliceArray(matchedDataProps[i])
170    var mergedSliceArray = mergeSliceArray(sliceArray);
171    // Highlight keyword.
172    li.push(buildHighlightedContent(matchedDataProps[i], mergedSliceArray));
173    li.push("...</p>");
174  }
175  li.push("</ul>");
176  resultContent.innerHTML = li.join("");
177}
178
179function ajax(url, callback) {
180  var xhr = null;
181  if (window.XMLHttpRequest) {
182    xhr = new XMLHttpRequest();
183  } else if (window.ActiveXObject) {
184    xhr = new ActiveXObject("Microsoft.XMLHTTP");
185  }
186  if (xhr == null) {
187    console.error("Your broswer does not support XMLHttpRequest!");
188    return;
189  }
190  xhr.onreadystatechange = function () {
191    // 4 is ready.
192    if (xhr.readyState !== 4) {
193      return;
194    }
195    if (xhr.status !== 200) {
196      console.error("XMLHttpRequest failed!");
197      return;
198    }
199    callback(xhr);
200  };
201  xhr.open("GET", url, true);
202  xhr.send(null);
203}
204
205var searchFunc = function (path, searchID, contentID) {
206  ajax(path, function (xhr) {
207    var datas = [];
208    if (xhr.responseXML) {
209      var xmlDoc = xhr.responseXML;
210      var entries = xmlDoc.getElementsByTagName("entry");
211      for (var i = 0; i < entries.length; i++) {
212        datas.push({
213          "title": entries[i].getElementsByTagName("title")[0].innerHTML || "",
214          "content": entries[i].getElementsByTagName("content")[0].innerHTML || "",
215          "url": entries[i].getElementsByTagName("url")[0].innerHTML || ""
216        });
217      }
218    } else {
219      var xhrJSON = JSON.parse(xhr.response);
220      for (var i = 0; i < xhrJSON.length; i++) {
221        datas.push({
222          "title": xhrJSON[i]["title"] || "",
223          "content": xhrJSON[i]["content"] || "", // Hexo Generator Search does not fill a key when a page is blank.
224          "url": xhrJSON[i]["url"] || ""
225        });
226      }
227    }
228    var input = document.getElementById(searchID);
229    var resultContent = document.getElementById(contentID);
230    input.addEventListener("input", function () {
231      onInput(input, resultContent, datas);
232    });
233  });
234}

Line numbers count LF bytes from the start of the resource, as the search results do. Vendor segments are library code the classifier recognised; they are stored but not indexed. Bytes are shown as Latin1 characters, one per byte.