PageSourceSearch

https://smalldocs.org/public/sdocs-comments.js?v=0ef0401827

js smalldocs.org collected 2026-09-29 16:36:38 UTC 33,235 bytes, 793 lines download raw bytes

1/**
2 * sdocs-comments.js - sidecar comment storage for comment mode.
3 *
4 * Comments live in the document's YAML front matter under `comments:` - a
5 * list of objects. The markdown body is NEVER touched; this removes every
6 * class of "marked parses our marker wrong" bug the old inline-HTML-comment
7 * format suffered from (line-start paragraph loss, inline-code backtick
8 * unbalancing, cross-block wrapping, etc.).
9 *
10 * Comment shape (all fields flat; easier to YAML-serialize):
11 *
12 *   Selection-anchored:
13 *     { id, kind: 'inline', quote, prefix, suffix, block, author, color, at, text }
14 *
15 *   Block-anchored:
16 *     { id, kind: 'block', block, author, color, at, text }
17 *
18 *   Nested-element-anchored:
19 *     { id, kind: 'element', block, element, element_text, element_scope,
20 *       author, color, at, text }
21 *
22 *   Table-anchored:
23 *     { id, kind: 'table', block, table_scope, table_row?, table_column?,
24 *       row_text?, column_text?, cell_text?, author, color, at, text }
25 *
26 *   Slide-anchored (a ```slide block, optionally a single shape within it):
27 *     { id, kind: 'slide', slide, shape?, slide_text?, author, color, at, text }
28 *
29 * - `quote` is the exact rendered-text phrase to highlight.
30 * - `prefix` / `suffix` (0-60 chars) disambiguate when `quote` appears
31 *   multiple times in the target block.
32 * - `block` is a "tagname:index-among-siblings-of-that-tagname" hint, e.g.
33 *   "p:3" = 4th <p> in render order. Per-type indexing is more resilient to
34 *   block reordering than a single global ordinal.
35 * - `element` is a structural path below `block`, e.g.
36 *   "li:1/ul:0/li:0". Each segment is the tag plus its index among direct
37 *   siblings of the same tag. `element_text` is the element's leading text
38 *   and lets the renderer recover when an item is inserted before it.
39 *   `element_scope` is "self" for one item or "branch" for the item and its
40 *   nested descendants.
41 * - `table_scope` is "table", "row", "column", or "cell". Row and column
42 *   indices are 0-based. The text fields let comments follow table content
43 *   when rows or columns move.
44 * - `slide` is the 0-based index of the ```slide block in document order.
45 *   `shape` (optional) is the 0-based index of a shape within that slide's
46 *   resolved DSL - it matches the `data-shape-idx` the renderer stamps on
47 *   each shape, and lines up with a shape line the model can read in the
48 *   slide source. Absent `shape` = a comment on the whole slide.
49 *   `slide_text` is the visible text of the targeted shape (or the slide's
50 *   leading text for a whole-slide note), kept as a human/AI-readable hint.
51 *
52 * Anchor resolution at render time uses a three-tier fallback:
53 *   1. Block-scoped:  find block → locate (prefix+quote+suffix) in its text.
54 *   2. Global:        search the whole rendered body for (prefix+quote+suffix).
55 *   3. Quote-only:    last-resort global search for just the quote.
56 *
57 * UMD module so Node tests can call it directly.
58 */
59(function (exports) {
60'use strict';
61
62// ── Helpers ─────────────────────────────────────────────────────────────
63
64var DEFAULT_COLOR = '#ffbb00';
65var HEX_COLOR = /^#(?:[0-9a-f]{3,4}|[0-9a-f]{6}|[0-9a-f]{8})$/i;
66var ID_FORMAT = /^c\d+$/;
67var ELEMENT_PATH = /^(?:li|ul|ol):\d+(?:\/(?:li|ul|ol):\d+)*$/;
68var TABLE_BLOCK = /^table:\d+$/;
69
70// Comment ids are written into a CSS selector
71// (`.sdoc-card[data-c="..."]`). Crafted ids with quotes or brackets
72// would break querySelector, or worse, match unrelated nodes. The
73// writer always produces `cN`, so reject anything else.
74function isValidId(id) {
75  return typeof id === 'string' && ID_FORMAT.test(id);
76}
77
78// Comment colours flow into `style.setProperty('--sdoc-...-color', c)`,
79// which is substituted directly into `background: var(--..., url())`-shaped
80// CSS. setProperty accepts arbitrary token sequences, so without a gate a
81// crafted shared URL could ship `url(https://attacker/p.gif)` as a colour
82// and every viewer would GET that on render. The colour <input> in the UI
83// emits #rrggbb only, so restrict to hex.
84function sanitizeColor(c) {
85  return (typeof c === 'string' && HEX_COLOR.test(c)) ? c : DEFAULT_COLOR;
86}
87
88// "Copy with comments" output gets pasted into agents, Slack, etc.
89// A crafted shared URL whose comment text contains an embedded newline
90// can forge additional [^cN]: footnote definitions in the copied bytes;
91// bidi format characters can make the rendered card look one way while
92// the copied bytes carry another. Strip both at serialization time so
93// the clipboard contents match what the user saw on screen.
94//   - C0 controls (0x00-0x1F): stripped except \t
95//   - C1 controls (0x80-0x9F): stripped
96//   - Bidi format chars: U+202A-U+202E, U+2066-U+2069
97//   - Embedded \n collapses to a single space (footnote labels are
98//     single-line; preserving the bytes would forge new footnotes)
99var BIDI_OR_CTRL = /[\u0000-\u0008\u000B-\u001F\u007F-\u009F\u202A-\u202E\u2066-\u2069]/g;
100function sanitizeText(s) {
101  if (typeof s !== 'string') return '';
102  return s.replace(/\n+/g, ' ').replace(BIDI_OR_CTRL, '');
103}
104
105function getComments(meta) {
106  if (!meta || typeof meta !== 'object') return [];
107  var list = meta.comments;
108  if (!Array.isArray(list)) return [];
109  return list.slice();
110}
111
112function setComments(meta, list) {
113  var out = Object.assign({}, meta || {});
114  if (!list || list.length === 0) delete out.comments;
115  else out.comments = list.slice();
116  return out;
117}
118
119function nextId(meta) {
120  var max = 0;
121  var list = getComments(meta);
122  for (var i = 0; i < list.length; i++) {
123    var m = /^c(\d+)$/.exec(list[i].id || '');
124    if (m) {
125      var n = parseInt(m[1], 10);
126      if (!isNaN(n) && n > max) max = n;
127    }
128  }
129  return 'c' + (max + 1);
130}
131
132// Coerce a value to a non-negative integer index, or null if it isn't one.
133// Slide / shape indices arrive from the DOM (data attributes, parsed as
134// strings) or hand-edited YAML (kept as strings by our parser), so accept
135// both numbers and numeric strings.
136function toIndex(v) {
137  if (v == null || v === '') return null;
138  var n = typeof v === 'number' ? v : parseInt(v, 10);
139  return (isFinite(n) && n >= 0) ? Math.floor(n) : null;
140}
141
142function normalizeComment(c) {
143  if (!isValidId(c.id)) return null;
144  var out = {
145    id: c.id,
146    kind: c.kind || (c.slide != null ? 'slide' : (c.quote ? 'inline' : 'block')),
147  };
148  // Anchor fields
149  if (out.kind === 'inline') {
150    out.quote = c.quote || '';
151    if (c.prefix) out.prefix = c.prefix;
152    if (c.suffix) out.suffix = c.suffix;
153  }
154  if (out.kind === 'element') {
155    if (!c.block || typeof c.element !== 'string' || !ELEMENT_PATH.test(c.element)) {
156      return null;
157    }
158    out.element = c.element;
159    if (c.element_text) out.element_text = c.element_text;
160    out.element_scope = c.element_scope === 'branch' ? 'branch' : 'self';
161  }
162  if (out.kind === 'table') {
163    if (!c.block || !TABLE_BLOCK.test(c.block)) return null;
164    var scope = /^(?:table|row|column|cell)$/.test(c.table_scope)
165      ? c.table_scope : 'table';
166    var row = toIndex(c.table_row);
167    var column = toIndex(c.table_column);
168    if ((scope === 'row' || scope === 'cell') && row == null) return null;
169    if ((scope === 'column' || scope === 'cell') && column == null) return null;
170    out.table_scope = scope;
171    if (row != null) out.table_row = row;
172    if (column != null) out.table_column = column;
173    if (c.row_text) out.row_text = c.row_text;
174    if (c.column_text) out.column_text = c.column_text;
175    if (c.cell_text) out.cell_text = c.cell_text;
176  }
177  if (out.kind === 'slide') {
178    // A slide comment without a resolvable slide index is meaningless;
179    // default to slide 0 rather than dropping the note entirely.
180    out.slide = toIndex(c.slide) == null ? 0 : toIndex(c.slide);
181    // `shapes` (array) is the multi-element form; a single index collapses to
182    // `shape` so single-element notes keep their existing on-disk shape. A
183    // whole-slide note carries neither.
184    if (Array.isArray(c.shapes)) {
185      var arr = c.shapes.map(toIndex).filter(function (n) { return n != null; });
186      if (arr.length === 1) out.shape = arr[0];
187      else if (arr.length > 1) out.shapes = arr;
188    } else {
189      var shp = toIndex(c.shape);
190      if (shp != null) out.shape = shp;
191    }
192    if (c.slide_text) out.slide_text = c.slide_text;
193  }
194  if (c.block) out.block = c.block;
195  // block_text: first ~60 chars of the block at write time. Used as a
196  // text-based fallback when the block index has drifted (e.g. a paragraph
197  // was inserted upstream). Block-comment analogue of inline's quote.
198  if (c.block_text) out.block_text = c.block_text;
199  // Author metadata
200  out.author = c.author || 'user';
201  out.color = sanitizeColor(c.color);
202  out.at = c.at || new Date().toISOString();
203  out.text = c.text || '';
204  // resolved: optional. Preserved only when truthy so the on-disk YAML
205  // stays terse for the common (unresolved) case. Coerced to boolean
206  // because the YAML parser intentionally keeps "true"/"false" as
207  // strings (its general contract); we apply boolean semantics here.
208  if (c.resolved === true || c.resolved === 'true') out.resolved = true;
209  return out;
210}
211
212// ── Mutations ───────────────────────────────────────────────────────────
213
214// anchor: { quote, prefix?, suffix?, block? }
215// noteMeta: { author?, color?, at?, text? }
216function addSelectionComment(meta, anchor, noteMeta) {
217  if (!anchor || typeof anchor.quote !== 'string' || !anchor.quote) {
218    throw new Error('addSelectionComment requires a non-empty quote');
219  }
220  var id = nextId(meta);
221  var c = normalizeComment({
222    id: id,
223    kind: 'inline',
224    quote: anchor.quote,
225    prefix: anchor.prefix || '',
226    suffix: anchor.suffix || '',
227    block: anchor.block || '',
228    author: (noteMeta || {}).author,
229    color: (noteMeta || {}).color,
230    at: (noteMeta || {}).at,
231    text: (noteMeta || {}).text,
232  });
233  var list = getComments(meta);
234  list.push(c);
235  return { meta: setComments(meta, list), id: id };
236}
237
238// anchor: { block, block_text? } where block is "tag:n" (e.g. "p:3").
239function addBlockComment(meta, anchor, noteMeta) {
240  if (!anchor || typeof anchor.block !== 'string' || !anchor.block) {
241    throw new Error('addBlockComment requires a block id');
242  }
243  var id = nextId(meta);
244  var c = normalizeComment({
245    id: id,
246    kind: 'block',
247    block: anchor.block,
248    block_text: anchor.block_text || '',
249    author: (noteMeta || {}).author,
250    color: (noteMeta || {}).color,
251    at: (noteMeta || {}).at,
252    text: (noteMeta || {}).text,
253  });
254  var list = getComments(meta);
255  list.push(c);
256  return { meta: setComments(meta, list), id: id };
257}
258
259// anchor: { block, element, element_text?, element_scope? }.
260// `element` is a direct-descendant path below the named top-level block.
261function addElementComment(meta, anchor, noteMeta) {
262  if (!anchor || typeof anchor.block !== 'string' || !anchor.block ||
263      typeof anchor.element !== 'string' || !ELEMENT_PATH.test(anchor.element)) {
264    throw new Error('addElementComment requires a block id and valid element path');
265  }
266  var id = nextId(meta);
267  var c = normalizeComment({
268    id: id,
269    kind: 'element',
270    block: anchor.block,
271    block_text: anchor.block_text || '',
272    element: anchor.element,
273    element_text: anchor.element_text || '',
274    element_scope: anchor.element_scope,
275    author: (noteMeta || {}).author,
276    color: (noteMeta || {}).color,
277    at: (noteMeta || {}).at,
278    text: (noteMeta || {}).text,
279  });
280  var list = getComments(meta);
281  list.push(c);
282  return { meta: setComments(meta, list), id: id };
283}
284
285// anchor: { block, table_scope, table_row?, table_column?, row_text?,
286//           column_text?, cell_text?, block_text? }.
287function addTableComment(meta, anchor, noteMeta) {
288  if (!anchor || typeof anchor.block !== 'string' ||
289      !TABLE_BLOCK.test(anchor.block)) {
290    throw new Error('addTableComment requires a table block id');
291  }
292  var id = nextId(meta);
293  var c = normalizeComment({
294    id: id,
295    kind: 'table',
296    block: anchor.block,
297    block_text: anchor.block_text || '',
298    table_scope: anchor.table_scope,
299    table_row: anchor.table_row,
300    table_column: anchor.table_column,
301    row_text: anchor.row_text || '',
302    column_text: anchor.column_text || '',
303    cell_text: anchor.cell_text || '',
304    author: (noteMeta || {}).author,
305    color: (noteMeta || {}).color,
306    at: (noteMeta || {}).at,
307    text: (noteMeta || {}).text,
308  });
309  if (!c) throw new Error('addTableComment requires a valid table target');
310  var list = getComments(meta);
311  list.push(c);
312  return { meta: setComments(meta, list), id: id };
313}
314
315// anchor: { slide, shape?, slide_text? } where slide is a 0-based slide
316// index and shape (optional) is a 0-based shape index within that slide.
317function addSlideComment(meta, anchor, noteMeta) {
318  if (!anchor || toIndex(anchor.slide) == null) {
319    throw new Error('addSlideComment requires a slide index');
320  }
321  var id = nextId(meta);
322  var c = normalizeComment({
323    id: id,
324    kind: 'slide',
325    slide: anchor.slide,
326    shape: anchor.shape,
327    shapes: anchor.shapes,
328    slide_text: anchor.slide_text || '',
329    author: (noteMeta || {}).author,
330    color: (noteMeta || {}).color,
331    at: (noteMeta || {}).at,
332    text: (noteMeta || {}).text,
333  });
334  var list = getComments(meta);
335  list.push(c);
336  return { meta: setComments(meta, list), id: id };
337}
338
339function removeComment(meta, id) {
340  var list = getComments(meta).filter(function (c) { return c.id !== id; });
341  return setComments(meta, list);
342}
343
344// Returns the input meta unchanged if no comment matches `id`, so
345// callers can compare reference-equality to detect a no-op.
346function updateComment(meta, id, patch) {
347  var changed = false;
348  var list = getComments(meta).map(function (c) {
349    if (c.id !== id) return c;
350    changed = true;
351    return normalizeComment(Object.assign({}, c, patch || {}));
352  });
353  return changed ? setComments(meta, list) : meta;
354}
355
356// ── Section slicing ─────────────────────────────────────────────────────
357
358// Walk lines and collect every top-level ATX heading (`#` ... `######`)
359// that sits OUTSIDE a fenced code block. Used by per-section copy to find
360// the section boundaries in the raw source.
361//
362// Why: a naive `^(#{1,N})\s+...` regex pulls in `##` lines that live inside
363// a ` ```markdown ` fence, which would silently truncate the copy at the
364// first inner heading. Tracking fence state line-by-line is enough.
365//
366// Fence rules (CommonMark-shaped, just enough for this use case):
367//   - opener:  ^[ ]{0,3}(`{3,}|~{3,}).*$
368//   - closer:  same fence char, count >= opener's count, only whitespace
369//              after the closing run
370function findTopHeadings(md) {
371  if (typeof md !== 'string') return [];
372  var out = [];
373  var lines = md.split('\n');
374  var fence = null; // null when outside; e.g. '```' or '~~~~' when inside
375  var pos = 0;
376  for (var i = 0; i < lines.length; i++) {
377    var line = lines[i];
378    var trimmed = line.replace(/^[ \t]{0,3}/, '');
379    var fenceMatch = /^(`{3,}|~{3,})/.exec(trimmed);
380    if (fence) {
381      if (fenceMatch && fenceMatch[1][0] === fence[0]
382          && fenceMatch[1].length >= fence.length
383          && /^\s*$/.test(trimmed.slice(fenceMatch[1].length))) {
384        fence = null;
385      }
386    } else if (fenceMatch) {
387      fence = fenceMatch[1];
388    } else {
389      var h = /^(#{1,6})\s+(.+?)\s*$/.exec(line);
390      if (h) {
391        out.push({ index: pos, level: h[1].length, text: h[2].trim() });
392      }
393    }
394    pos += line.length + 1; // +1 for the consumed '\n'
395  }
396  return out;
397}
398
399// Slice `md` to the substring covered by the section whose ATX heading
400// matches (level, headingText). Section ends at the next heading of equal
401// or higher rank, ignoring any headings inside fenced code blocks. Returns
402// null if the heading isn't found.
403function findSectionRange(md, level, headingText) {
404  var headings = findTopHeadings(md);
405  var startIdx = -1, startI = -1;
406  for (var i = 0; i < headings.length; i++) {
407    var h = headings[i];
408    if (h.level === level && h.text === headingText) {
409      startIdx = h.index;
410      startI = i;
411      break;
412    }
413  }
414  if (startIdx === -1) return null;
415  var endIdx = md.length;
416  for (var j = startI + 1; j < headings.length; j++) {
417    if (headings[j].level <= level) { endIdx = headings[j].index; break; }
418  }
419  return { startIdx: startIdx, endIdx: endIdx, body: md.slice(startIdx, endIdx) };
420}
421
422// ── Copy serializers ────────────────────────────────────────────────────
423
424/**
425 * serializeFootnotes(meta, body) -> string
426 *
427 * Emits the body with each inline comment's quote transformed into a
428 * `[quote][^cN]` footnote reference, and the comment texts appended as
429 * `[^cN]: author - text` lines. Block comments become footnote refs
430 * attached at the end of the document.
431 *
432 * Matching strategy, in priority order:
433 *   1. `c.occurrence` (number, 0-indexed) — the Nth occurrence of `quote`
434 *      in `body`. Caller computes this from the rendered DOM at copy time
435 *      so we always pick the same occurrence the user actually anchored
436 *      to, even when source markdown differs from rendered text (bold,
437 *      links, code formatting, etc.).
438 *   2. `prefix + quote + suffix` matches uniquely → land there.
439 *   3. `quote` alone → first occurrence (best-effort fallback).
440 *
441 * Replacement positions are computed against the original body in one
442 * pass, then applied in descending order. This means earlier comments'
443 * positions are never shifted by later replacements, AND occurrence
444 * counts always run against the unedited source text — so two comments
445 * targeting different occurrences of the same quote can both land
446 * correctly.
447 */
448function serializeFootnotes(meta, body) {
449  if (typeof body !== 'string') return body;
450  var comments = getComments(meta);
451  if (!comments.length) return body;
452  var inlineComments = comments.filter(function (c) { return c.kind === 'inline' && c.quote; });
453  var hits = [];
454  inlineComments.forEach(function (c) {
455    var pos = -1;
456    if (typeof c.occurrence === 'number' && c.occurrence >= 0) {
457      pos = nthIndexOf(body, c.quote, c.occurrence);
458    }
459    if (pos === -1) {
460      var needle = (c.prefix || '') + c.quote + (c.suffix || '');
461      var idx = body.indexOf(needle);
462      if (idx !== -1) pos = idx + (c.prefix || '').length;
463    }
464    if (pos === -1) pos = body.indexOf(c.quote);
465    if (pos !== -1) {
466      hits.push({ start: pos, end: pos + c.quote.length, id: c.id, quote: c.quote });
467    }
468  });
469  hits.sort(function (a, b) { return b.start - a.start; });
470  // Skip a hit whose range overlaps with one we've already replaced (we
471  // walk right-to-left, so "already replaced" means a hit further right).
472  // Without this guard, slice/replace math would corrupt the body when a
473  // later comment's range falls inside an earlier comment's range. The
474  // dropped hit still gets a footnote definition appended below — only
475  // its inline anchor is omitted.
476  var out = body;
477  var minRightEdge = body.length + 1;
478  for (var i = 0; i < hits.length; i++) {
479    var h = hits[i];
480    if (h.end > minRightEdge) continue;
481    var replacement = '[' + h.quote + '][^' + h.id + ']';
482    out = out.slice(0, h.start) + replacement + out.slice(h.end);
483    minRightEdge = h.start;
484  }
485  var footnotes = [];
486  comments.forEach(function (c) {
487    var label = sanitizeText(c.author || 'user');
488    if (c.resolved) label += ' [resolved]';
489    label += ' - ' + sanitizeText(c.text || '');
490    if (c.kind === 'block' && c.block) {
491      var blockTag = c.block;
492      if (c.block_text) blockTag += ' "' + sanitizeText(c.block_text) + '..."';
493      label += ' (block ' + blockTag + ')';
494    }
495    if (c.kind === 'element' && c.block && c.element) {
496      var elementTag = c.block + ' > ' + c.element;
497      if (c.element_scope === 'branch') elementTag += ' branch';
498      if (c.element_text) elementTag += ' "' + sanitizeText(c.element_text) + '..."';
499      label += ' (element ' + elementTag + ')';
500    }
501    if (c.kind === 'table' && c.block) {
502      var tableTag = c.block;
503      if (c.table_scope === 'row' || c.table_scope === 'cell') {
504        tableTag += ', row ' + (c.table_row + 1);
505        if (c.row_text) tableTag += ' "' + sanitizeText(c.row_text) + '"';
506      }
507      if (c.table_scope === 'column' || c.table_scope === 'cell') {
508        tableTag += ', column ' + (c.table_column + 1);
509        if (c.column_text) tableTag += ' "' + sanitizeText(c.column_text) + '"';
510      }
511      if (c.table_scope === 'cell' && c.cell_text) {
512        tableTag += ', cell "' + sanitizeText(c.cell_text) + '"';
513      }
514      label += ' (table ' + tableTag + ')';
515    }
516    if (c.kind === 'slide') {
517      // Slides don't anchor into body text (their source is the fenced DSL,
518      // which we never rewrite), so the location rides entirely in the
519      // footnote label. Slide number is 1-based to match what the user sees
520      // in the present-mode counter and the slide badge. When the note
521      // targets one shape, carry its index (matches `data-shape-idx` and the
522      // shape's line in the slide source) plus the shape's visible text.
523      var slideTag = 'slide ' + ((typeof c.slide === 'number' ? c.slide : 0) + 1);
524      if (Array.isArray(c.shapes) && c.shapes.length) {
525        // Multi-element note: list the indices joined with '+', then one
526        // quoted hint covering all of them (the UI joins the labels).
527        slideTag += ', elements ' + c.shapes.join('+');
528        if (c.slide_text) slideTag += ' "' + sanitizeText(c.slide_text) + '"';
529      } else if (typeof c.shape === 'number') {
530        slideTag += ', element ' + c.shape;
531        if (c.slide_text) slideTag += ' "' + sanitizeText(c.slide_text) + '"';
532      } else if (c.slide_text) {
533        slideTag += ' "' + sanitizeText(c.slide_text) + '"';
534      }
535      label += ' (' + slideTag + ')';
536    }
537    footnotes.push('[^' + c.id + ']: ' + label);
538  });
539  return out.replace(/\n*$/, '') + '\n\n' + footnotes.join('\n') + '\n';
540}
541
542// Index of the n-th (0-indexed) occurrence of `needle` in `hay`, or -1.
543function nthIndexOf(hay, needle, n) {
544  if (!needle) return -1;
545  var pos = -1, from = 0;
546  for (var i = 0; i <= n; i++) {
547    pos = hay.indexOf(needle, from);
548    if (pos === -1) return -1;
549    from = pos + needle.length;
550  }
551  return pos;
552}
553
554/**
555 * serializeClean(meta, body) -> string
556 *
557 * Returns the body verbatim. Supplied as the "strip all comments" flow;
558 * since the sidecar model never injected anything into the body, this
559 * is literally the identity function - but we wrap it for symmetry with
560 * the other serializers.
561 */
562function serializeClean(meta, body) { return body; }
563
564/**
565 * parseFootnotes(body) -> { comments, body }
566 *
567 * Inverse of serializeFootnotes. Recognises markdown footnotes whose ids
568 * follow our convention (cN where N is digits) and converts them into
569 * comment objects. Other footnote ids are left alone - academic citations
570 * and the like keep their footnote semantics.
571 *
572 * Recognised patterns:
573 *   Inline:  [quote][^cN]            anchor span = quote, kind = inline
574 *   Block:   ...end of paragraph.[^cN]   kind = block, anchor = containing
575 *                                         paragraph (block_text = first ~60
576 *                                         chars of the paragraph the marker
577 *                                         sits in)
578 *   Defn:    [^cN]: author - text [resolved]?   the comment text
579 *
580 * The returned `body` has the recognised refs/defs stripped, ready to feed
581 * to the markdown renderer. Unrecognised footnotes pass through unchanged.
582 *
583 * `block_text` is computed from the body string at parse time, before
584 * marked rendering. For block comments, this is the survival hint that
585 * lets the existing tier-2 resolver attach the comment to the right
586 * block in the rendered DOM (where tag:n indices are computed).
587 */
588function parseFootnotes(body) {
589  if (typeof body !== 'string' || body.indexOf('[^c') === -1) {
590    return { comments: [], body: body };
591  }
592  var ID_RE = /^c\d+$/;
593  // Pull out definitions first. Anchored at line start; tolerate trailing
594  // whitespace/newlines. Also tolerate `[^cN]:`-only lines (no text).
595  var defs = {};
596  var DEF_RE = /^\[\^(c\d+)\]:[ \t]*(.*)$/;
597  var lines = body.split('\n');
598  var keptLines = [];
599  for (var i = 0; i < lines.length; i++) {
600    var m = lines[i].match(DEF_RE);
601    if (m) {
602      defs[m[1]] = (m[2] || '').trim();
603    } else {
604      keptLines.push(lines[i]);
605    }
606  }
607  body = keptLines.join('\n');
608
609  // Pull author + text + resolved-marker out of a definition string.
610  // Format emitted by serializeFootnotes: "<author>[ [resolved]] - <text>".
611  function decodeDef(raw) {
612    var out = { text: raw || '', resolved: false, author: undefined };
613    if (!raw) return out;
614    if (/\[resolved\]/.test(raw)) out.resolved = true;
615    // Try to split on the first " - " that comes after a plausible author token.
616    var split = raw.match(/^(.+?)\s-\s(.+)$/);
617    if (split) {
618      var head = split[1].replace(/\s*\[resolved\]\s*$/, '').trim();
619      // Heuristic: treat head as an author handle if it's compact (no inner punctuation
620      // beyond `.`, `_`, `-`, brackets, spaces) and reasonably short.
621      if (head.length <= 40 && /^[\w][\w \[\].\-_]*$/.test(head)) {
622        out.author = head;
623        out.text = split[2];
624      }
625    }
626    // Trailing "(block tag:n)" hint, if present, gets stripped from text and
627    // surfaced separately.
628    var tag = (out.text || '').match(/\s*\(block\s+(\w+:\d+)\)\s*$/);
629    if (tag) {
630      out.block = tag[1];
631      out.text = out.text.replace(/\s*\(block\s+\w+:\d+\)\s*$/, '');
632    }
633    // Trailing nested-element hint emitted for list-item comments.
634    var eTag = (out.text || '').match(
635      /\s*\(element\s+(\w+:\d+)\s+>\s+((?:li|ul|ol):\d+(?:\/(?:li|ul|ol):\d+)*)(\s+branch)?(?:\s+"([^"]*)\.\.\.")?\)\s*$/);
636    if (eTag) {
637      out.block = eTag[1];
638      out.element = eTag[2];
639      out.element_scope = eTag[3] ? 'branch' : 'self';
640      if (eTag[4]) out.element_text = eTag[4];
641      out.text = out.text.replace(
642        /\s*\(element\s+\w+:\d+\s+>\s+(?:li|ul|ol):\d+(?:\/(?:li|ul|ol):\d+)*(?:\s+branch)?(?:\s+"[^"]*\.\.\.")?\)\s*$/,
643        '');
644    }
645    // Trailing table target hint. Row and column numbers are 1-based in the
646    // copied footnote so they match what the reader sees, then return to
647    // 0-based indices in the in-memory model.
648    var tTag = (out.text || '').match(
649      /\s*\(table\s+(table:\d+)(?:,\s*row\s+(\d+)(?:\s+"([^"]*)")?)?(?:,\s*column\s+(\d+)(?:\s+"([^"]*)")?)?(?:,\s*cell\s+"([^"]*)")?\)\s*$/);
650    if (tTag) {
651      out.block = tTag[1];
652      if (tTag[2] != null) out.table_row = parseInt(tTag[2], 10) - 1;
653      if (tTag[3]) out.row_text = tTag[3];
654      if (tTag[4] != null) out.table_column = parseInt(tTag[4], 10) - 1;
655      if (tTag[5]) out.column_text = tTag[5];
656      if (tTag[6]) out.cell_text = tTag[6];
657      out.table_scope = tTag[6] != null ? 'cell'
658        : (tTag[4] != null ? 'column' : (tTag[2] != null ? 'row' : 'table'));
659      out.text = out.text.replace(
660        /\s*\(table\s+table:\d+(?:,\s*row\s+\d+(?:\s+"[^"]*")?)?(?:,\s*column\s+\d+(?:\s+"[^"]*")?)?(?:,\s*cell\s+"[^"]*")?\)\s*$/,
661        '');
662    }
663    // Trailing "(slide N[, element M][ "text"])" hint - the inverse of the
664    // slide label serializeFootnotes emits. Slide number is 1-based on disk;
665    // store it 0-based to match the in-memory anchor.
666    var sTag = (out.text || '').match(/\s*\(slide\s+(\d+)(?:,\s*elements?\s+([\d+]+))?(?:\s+"([^"]*)")?\)\s*$/);
667    if (sTag) {
668      out.slide = parseInt(sTag[1], 10) - 1;
669      if (sTag[2] != null) {
670        var idxs = sTag[2].split('+').map(function (x) { return parseInt(x, 10); })
671          .filter(function (n) { return !isNaN(n); });
672        if (idxs.length > 1) out.shapes = idxs;
673        else if (idxs.length === 1) out.shape = idxs[0];
674      }
675      if (sTag[3]) out.slide_text = sTag[3];
676      out.text = out.text.replace(/\s*\(slide\s+\d+(?:,\s*elements?\s+[\d+]+)?(?:\s+"[^"]*")?\)\s*$/, '');
677    }
678    return out;
679  }
680
681  var comments = [];
682  var seen = {};
683
684  // Inline: [quote][^cN]
685  body = body.replace(/\[([^\]\n]+?)\]\[\^(c\d+)\]/g, function (_m, quote, id) {
686    if (!ID_RE.test(id)) return _m;
687    if (seen[id]) return _m;
688    seen[id] = true;
689    var d = decodeDef(defs[id]);
690    var c = { id: id, kind: 'inline', quote: quote, text: d.text };
691    if (d.author) c.author = d.author;
692    if (d.resolved) c.resolved = true;
693    comments.push(c);
694    return quote;
695  });
696
697  // Block markers: lone [^cN] not preceded by `]`. The surrounding paragraph
698  // becomes the block_text survival hint.
699  var REF_RE = /\[\^(c\d+)\]/g;
700  var match;
701  while ((match = REF_RE.exec(body)) !== null) {
702    var id = match[1];
703    if (!ID_RE.test(id) || seen[id]) continue;
704    var refStart = match.index;
705    if (body[refStart - 1] === ']') continue;
706    seen[id] = true;
707    var paraStart = body.lastIndexOf('\n\n', refStart);
708    paraStart = paraStart === -1 ? 0 : paraStart + 2;
709    var paraEnd = body.indexOf('\n\n', refStart);
710    if (paraEnd === -1) paraEnd = body.length;
711    var paragraph = body.slice(paraStart, paraEnd);
712    var clean = paragraph.replace(REF_RE, '').trim();
713    var d2 = decodeDef(defs[id]);
714    var c2 = { id: id, kind: 'block', block_text: clean.slice(0, 60), text: d2.text };
715    if (d2.block) c2.block = d2.block;
716    if (d2.author) c2.author = d2.author;
717    if (d2.resolved) c2.resolved = true;
718    comments.push(c2);
719  }
720  body = body.replace(REF_RE, function (_m, id) { return seen[id] ? '' : _m; });
721
722  // Orphan definitions (no body marker). Treat as block comments. This is the
723  // shape serializeFootnotes emits today for block-kind comments - the def
724  // carries a `(block tag:n)` hint and there is no body marker.
725  Object.keys(defs).forEach(function (id) {
726    if (!ID_RE.test(id) || seen[id]) return;
727    seen[id] = true;
728    var d3 = decodeDef(defs[id]);
729    var c3;
730    if (d3.table_scope) {
731      c3 = {
732        id: id, kind: 'table', block: d3.block,
733        table_scope: d3.table_scope, text: d3.text,
734      };
735      if (d3.table_row != null) c3.table_row = d3.table_row;
736      if (d3.table_column != null) c3.table_column = d3.table_column;
737      if (d3.row_text) c3.row_text = d3.row_text;
738      if (d3.column_text) c3.column_text = d3.column_text;
739      if (d3.cell_text) c3.cell_text = d3.cell_text;
740    } else if (d3.element) {
741      c3 = {
742        id: id, kind: 'element', block: d3.block, element: d3.element,
743        element_scope: d3.element_scope, text: d3.text,
744      };
745      if (d3.element_text) c3.element_text = d3.element_text;
746    } else if (typeof d3.slide === 'number') {
747      // Slide notes carry a "(slide N ...)" hint and no body marker, so they
748      // land here as orphan definitions. Rebuild the slide anchor.
749      c3 = { id: id, kind: 'slide', slide: d3.slide, text: d3.text };
750      if (Array.isArray(d3.shapes)) c3.shapes = d3.shapes;
751      else if (typeof d3.shape === 'number') c3.shape = d3.shape;
752      if (d3.slide_text) c3.slide_text = d3.slide_text;
753    } else {
754      c3 = { id: id, kind: 'block', text: d3.text };
755      if (d3.block) c3.block = d3.block;
756    }
757    if (d3.author) c3.author = d3.author;
758    if (d3.resolved) c3.resolved = true;
759    comments.push(c3);
760  });
761
762  body = body.replace(/\n{3,}$/, '\n').replace(/[ \t]+$/gm, '');
763  // Return in chronological (id) order rather than parse-pass order.
764  comments.sort(function (a, b) {
765    var na = parseInt((a.id || '').slice(1), 10);
766    var nb = parseInt((b.id || '').slice(1), 10);
767    return na - nb;
768  });
769  return { comments: comments, body: body };
770}
771
772// ── Public API ──────────────────────────────────────────────────────────
773
774exports.sanitizeColor       = sanitizeColor;
775exports.isValidId           = isValidId;
776exports.getComments         = getComments;
777exports.setComments         = setComments;
778exports.nextId              = nextId;
779exports.normalizeComment    = normalizeComment;
780exports.addSelectionComment = addSelectionComment;
781exports.addBlockComment     = addBlockComment;
782exports.addElementComment   = addElementComment;
783exports.addTableComment     = addTableComment;
784exports.addSlideComment     = addSlideComment;
785exports.removeComment       = removeComment;
786exports.updateComment       = updateComment;
787exports.serializeFootnotes  = serializeFootnotes;
788exports.parseFootnotes      = parseFootnotes;
789exports.serializeClean      = serializeClean;
790exports.findTopHeadings     = findTopHeadings;
791exports.findSectionRange    = findSectionRange;
792
793})(typeof module !== 'undefined' && module.exports ? module.exports : (window.SDocComments = {}));

Line numbers count LF bytes from the start of the resource, as the search results do. Vendor segments are library code the classifier recognised; they are stored but not indexed. Bytes are shown as Latin1 characters, one per byte.