1/** 2 * sdocs-comments.js - sidecar comment storage for comment mode. 3 * 4 * Comments live in the document's YAML front matter under `comments:` - a 5 * list of objects. The markdown body is NEVER touched; this removes every 6 * class of "marked parses our marker wrong" bug the old inline-HTML-comment 7 * format suffered from (line-start paragraph loss, inline-code backtick 8 * unbalancing, cross-block wrapping, etc.). 9 * 10 * Comment shape (all fields flat; easier to YAML-serialize): 11 * 12 * Selection-anchored: 13 * { id, kind: 'inline', quote, prefix, suffix, block, author, color, at, text } 14 * 15 * Block-anchored: 16 * { id, kind: 'block', block, author, color, at, text } 17 * 18 * Nested-element-anchored: 19 * { id, kind: 'element', block, element, element_text, element_scope, 20 * author, color, at, text } 21 * 22 * Table-anchored: 23 * { id, kind: 'table', block, table_scope, table_row?, table_column?, 24 * row_text?, column_text?, cell_text?, author, color, at, text } 25 * 26 * Slide-anchored (a ```slide block, optionally a single shape within it): 27 * { id, kind: 'slide', slide, shape?, slide_text?, author, color, at, text } 28 * 29 * - `quote` is the exact rendered-text phrase to highlight. 30 * - `prefix` / `suffix` (0-60 chars) disambiguate when `quote` appears 31 * multiple times in the target block. 32 * - `block` is a "tagname:index-among-siblings-of-that-tagname" hint, e.g. 33 * "p:3" = 4th <p> in render order. Per-type indexing is more resilient to 34 * block reordering than a single global ordinal. 35 * - `element` is a structural path below `block`, e.g. 36 * "li:1/ul:0/li:0". Each segment is the tag plus its index among direct 37 * siblings of the same tag. `element_text` is the element's leading text 38 * and lets the renderer recover when an item is inserted before it. 39 * `element_scope` is "self" for one item or "branch" for the item and its 40 * nested descendants. 41 * - `table_scope` is "table", "row", "column", or "cell". Row and column 42 * indices are 0-based. The text fields let comments follow table content 43 * when rows or columns move. 44 * - `slide` is the 0-based index of the ```slide block in document order. 45 * `shape` (optional) is the 0-based index of a shape within that slide's 46 * resolved DSL - it matches the `data-shape-idx` the renderer stamps on 47 * each shape, and lines up with a shape line the model can read in the 48 * slide source. Absent `shape` = a comment on the whole slide. 49 * `slide_text` is the visible text of the targeted shape (or the slide's 50 * leading text for a whole-slide note), kept as a human/AI-readable hint. 51 * 52 * Anchor resolution at render time uses a three-tier fallback: 53 * 1. Block-scoped: find block â locate (prefix+quote+suffix) in its text. 54 * 2. Global: search the whole rendered body for (prefix+quote+suffix). 55 * 3. Quote-only: last-resort global search for just the quote. 56 * 57 * UMD module so Node tests can call it directly. 58 */ 59(function (exports) { 60'use strict'; 61 62// ââ Helpers âââââââââââââââââââââââââââââââââââââââââââââââââââââââââââââ 63 64var DEFAULT_COLOR = '#ffbb00'; 65var HEX_COLOR = /^#(?:[0-9a-f]{3,4}|[0-9a-f]{6}|[0-9a-f]{8})$/i; 66var ID_FORMAT = /^c\d+$/; 67var ELEMENT_PATH = /^(?:li|ul|ol):\d+(?:\/(?:li|ul|ol):\d+)*$/; 68var TABLE_BLOCK = /^table:\d+$/; 69 70// Comment ids are written into a CSS selector 71// (`.sdoc-card[data-c="..."]`). Crafted ids with quotes or brackets 72// would break querySelector, or worse, match unrelated nodes. The 73// writer always produces `cN`, so reject anything else. 74function isValidId(id) { 75 return typeof id === 'string' && ID_FORMAT.test(id); 76} 77 78// Comment colours flow into `style.setProperty('--sdoc-...-color', c)`, 79// which is substituted directly into `background: var(--..., url())`-shaped 80// CSS. setProperty accepts arbitrary token sequences, so without a gate a 81// crafted shared URL could ship `url(https://attacker/p.gif)` as a colour 82// and every viewer would GET that on render. The colour <input> in the UI 83// emits #rrggbb only, so restrict to hex. 84function sanitizeColor(c) { 85 return (typeof c === 'string' && HEX_COLOR.test(c)) ? c : DEFAULT_COLOR; 86} 87 88// "Copy with comments" output gets pasted into agents, Slack, etc. 89// A crafted shared URL whose comment text contains an embedded newline 90// can forge additional [^cN]: footnote definitions in the copied bytes; 91// bidi format characters can make the rendered card look one way while 92// the copied bytes carry another. Strip both at serialization time so 93// the clipboard contents match what the user saw on screen. 94// - C0 controls (0x00-0x1F): stripped except \t 95// - C1 controls (0x80-0x9F): stripped 96// - Bidi format chars: U+202A-U+202E, U+2066-U+2069
97// - Embedded \n collapses to a single space (footnote labels are 98// single-line; preserving the bytes would forge new footnotes) 99var BIDI_OR_CTRL = /[\u0000-\u0008\u000B-\u001F\u007F-\u009F\u202A-\u202E\u2066-\u2069]/g; 100function sanitizeText(s) { 101 if (typeof s !== 'string') return ''; 102 return s.replace(/\n+/g, ' ').replace(BIDI_OR_CTRL, ''); 103} 104 105function getComments(meta) { 106 if (!meta || typeof meta !== 'object') return []; 107 var list = meta.comments; 108 if (!Array.isArray(list)) return []; 109 return list.slice(); 110} 111 112function setComments(meta, list) { 113 var out = Object.assign({}, meta || {}); 114 if (!list || list.length === 0) delete out.comments; 115 else out.comments = list.slice(); 116 return out; 117} 118 119function nextId(meta) { 120 var max = 0; 121 var list = getComments(meta); 122 for (var i = 0; i < list.length; i++) { 123 var m = /^c(\d+)$/.exec(list[i].id || ''); 124 if (m) { 125 var n = parseInt(m[1], 10); 126 if (!isNaN(n) && n > max) max = n; 127 } 128 } 129 return 'c' + (max + 1); 130} 131 132// Coerce a value to a non-negative integer index, or null if it isn't one. 133// Slide / shape indices arrive from the DOM (data attributes, parsed as 134// strings) or hand-edited YAML (kept as strings by our parser), so accept 135// both numbers and numeric strings. 136function toIndex(v) { 137 if (v == null || v === '') return null; 138 var n = typeof v === 'number' ? v : parseInt(v, 10); 139 return (isFinite(n) && n >= 0) ? Math.floor(n) : null; 140} 141 142function normalizeComment(c) { 143 if (!isValidId(c.id)) return null; 144 var out = { 145 id: c.id, 146 kind: c.kind || (c.slide != null ? 'slide' : (c.quote ? 'inline' : 'block')), 147 }; 148 // Anchor fields 149 if (out.kind === 'inline') { 150 out.quote = c.quote || ''; 151 if (c.prefix) out.prefix = c.prefix; 152 if (c.suffix) out.suffix = c.suffix; 153 } 154 if (out.kind === 'element') { 155 if (!c.block || typeof c.element !== 'string' || !ELEMENT_PATH.test(c.element)) { 156 return null; 157 } 158 out.element = c.element; 159 if (c.element_text) out.element_text = c.element_text; 160 out.element_scope = c.element_scope === 'branch' ? 'branch' : 'self'; 161 } 162 if (out.kind === 'table') { 163 if (!c.block || !TABLE_BLOCK.test(c.block)) return null; 164 var scope = /^(?:table|row|column|cell)$/.test(c.table_scope) 165 ? c.table_scope : 'table'; 166 var row = toIndex(c.table_row); 167 var column = toIndex(c.table_column); 168 if ((scope === 'row' || scope === 'cell') && row == null) return null; 169 if ((scope === 'column' || scope === 'cell') && column == null) return null; 170 out.table_scope = scope; 171 if (row != null) out.table_row = row; 172 if (column != null) out.table_column = column; 173 if (c.row_text) out.row_text = c.row_text; 174 if (c.column_text) out.column_text = c.column_text; 175 if (c.cell_text) out.cell_text = c.cell_text; 176 } 177 if (out.kind === 'slide') { 178 // A slide comment without a resolvable slide index is meaningless; 179 // default to slide 0 rather than dropping the note entirely. 180 out.slide = toIndex(c.slide) == null ? 0 : toIndex(c.slide); 181 // `shapes` (array) is the multi-element form; a single index collapses to 182 // `shape` so single-element notes keep their existing on-disk shape. A 183 // whole-slide note carries neither. 184 if (Array.isArray(c.shapes)) { 185 var arr = c.shapes.map(toIndex).filter(function (n) { return n != null; }); 186 if (arr.length === 1) out.shape = arr[0]; 187 else if (arr.length > 1) out.shapes = arr; 188 } else { 189 var shp = toIndex(c.shape); 190 if (shp != null) out.shape = shp; 191 } 192 if (c.slide_text) out.slide_text = c.slide_text; 193 } 194 if (c.block) out.block = c.block; 195 // block_text: first ~60 chars of the block at write time. Used as a 196 // text-based fallback when the block index has drifted (e.g. a paragraph 197 // was inserted upstream). Block-comment analogue of inline's quote. 198 if (c.block_text) out.block_text = c.block_text; 199 // Author metadata 200 out.author = c.author || 'user'; 201 out.color = sanitizeColor(c.color); 202 out.at = c.at || new Date().toISOString(); 203 out.text = c.text || ''; 204 // resolved: optional. Preserved only when truthy so the on-disk YAML 205 // stays terse for the common (unresolved) case. Coerced to boolean 206 // because the YAML parser intentionally keeps "true"/"false" as 207 // strings (its general contract); we apply boolean semantics here. 208 if (c.resolved === true || c.resolved === 'true') out.resolved = true; 209 return out; 210} 211 212// ââ Mutations âââââââââââââââââââââââââââââââââââââââââââââââââââââââââââ 213 214// anchor: { quote, prefix?, suffix?, block? } 215// noteMeta: { author?, color?, at?, text? } 216function addSelectionComment(meta, anchor, noteMeta) { 217 if (!anchor || typeof anchor.quote !== 'string' || !anchor.quote) { 218 throw new Error('addSelectionComment requires a non-empty quote'); 219 } 220 var id = nextId(meta); 221 var c = normalizeComment({ 222 id: id, 223 kind: 'inline', 224 quote: anchor.quote, 225 prefix: anchor.prefix || '', 226 suffix: anchor.suffix || '', 227 block: anchor.block || '', 228 author: (noteMeta || {}).author, 229 color: (noteMeta || {}).color, 230 at: (noteMeta || {}).at, 231 text: (noteMeta || {}).text, 232 }); 233 var list = getComments(meta); 234 list.push(c); 235 return { meta: setComments(meta, list), id: id }; 236} 237 238// anchor: { block, block_text? } where block is "tag:n" (e.g. "p:3"). 239function addBlockComment(meta, anchor, noteMeta) { 240 if (!anchor || typeof anchor.block !== 'string' || !anchor.block) { 241 throw new Error('addBlockComment requires a block id'); 242 } 243 var id = nextId(meta); 244 var c = normalizeComment({ 245 id: id, 246 kind: 'block', 247 block: anchor.block, 248 block_text: anchor.block_text || '', 249 author: (noteMeta || {}).author, 250 color: (noteMeta || {}).color, 251 at: (noteMeta || {}).at, 252 text: (noteMeta || {}).text, 253 }); 254 var list = getComments(meta); 255 list.push(c); 256 return { meta: setComments(meta, list), id: id }; 257} 258 259// anchor: { block, element, element_text?, element_scope? }.
260// `element` is a direct-descendant path below the named top-level block. 261function addElementComment(meta, anchor, noteMeta) { 262 if (!anchor || typeof anchor.block !== 'string' || !anchor.block || 263 typeof anchor.element !== 'string' || !ELEMENT_PATH.test(anchor.element)) { 264 throw new Error('addElementComment requires a block id and valid element path'); 265 } 266 var id = nextId(meta); 267 var c = normalizeComment({ 268 id: id, 269 kind: 'element', 270 block: anchor.block, 271 block_text: anchor.block_text || '', 272 element: anchor.element, 273 element_text: anchor.element_text || '', 274 element_scope: anchor.element_scope, 275 author: (noteMeta || {}).author, 276 color: (noteMeta || {}).color, 277 at: (noteMeta || {}).at, 278 text: (noteMeta || {}).text, 279 }); 280 var list = getComments(meta); 281 list.push(c); 282 return { meta: setComments(meta, list), id: id }; 283} 284 285// anchor: { block, table_scope, table_row?, table_column?, row_text?, 286// column_text?, cell_text?, block_text? }. 287function addTableComment(meta, anchor, noteMeta) { 288 if (!anchor || typeof anchor.block !== 'string' || 289 !TABLE_BLOCK.test(anchor.block)) { 290 throw new Error('addTableComment requires a table block id'); 291 } 292 var id = nextId(meta); 293 var c = normalizeComment({ 294 id: id, 295 kind: 'table', 296 block: anchor.block, 297 block_text: anchor.block_text || '', 298 table_scope: anchor.table_scope, 299 table_row: anchor.table_row, 300 table_column: anchor.table_column, 301 row_text: anchor.row_text || '', 302 column_text: anchor.column_text || '', 303 cell_text: anchor.cell_text || '', 304 author: (noteMeta || {}).author, 305 color: (noteMeta || {}).color, 306 at: (noteMeta || {}).at, 307 text: (noteMeta || {}).text, 308 }); 309 if (!c) throw new Error('addTableComment requires a valid table target'); 310 var list = getComments(meta); 311 list.push(c); 312 return { meta: setComments(meta, list), id: id }; 313} 314 315// anchor: { slide, shape?, slide_text? } where slide is a 0-based slide 316// index and shape (optional) is a 0-based shape index within that slide. 317function addSlideComment(meta, anchor, noteMeta) { 318 if (!anchor || toIndex(anchor.slide) == null) { 319 throw new Error('addSlideComment requires a slide index'); 320 } 321 var id = nextId(meta); 322 var c = normalizeComment({ 323 id: id, 324 kind: 'slide', 325 slide: anchor.slide, 326 shape: anchor.shape, 327 shapes: anchor.shapes, 328 slide_text: anchor.slide_text || '', 329 author: (noteMeta || {}).author, 330 color: (noteMeta || {}).color, 331 at: (noteMeta || {}).at, 332 text: (noteMeta || {}).text, 333 }); 334 var list = getComments(meta); 335 list.push(c); 336 return { meta: setComments(meta, list), id: id }; 337} 338 339function removeComment(meta, id) { 340 var list = getComments(meta).filter(function (c) { return c.id !== id; }); 341 return setComments(meta, list); 342} 343 344// Returns the input meta unchanged if no comment matches `id`, so 345// callers can compare reference-equality to detect a no-op. 346function updateComment(meta, id, patch) { 347 var changed = false; 348 var list = getComments(meta).map(function (c) { 349 if (c.id !== id) return c; 350 changed = true; 351 return normalizeComment(Object.assign({}, c, patch || {})); 352 }); 353 return changed ? setComments(meta, list) : meta; 354} 355 356// ââ Section slicing âââââââââââââââââââââââââââââââââââââââââââââââââââââ 357 358// Walk lines and collect every top-level ATX heading (`#` ... `######`) 359// that sits OUTSIDE a fenced code block. Used by per-section copy to find 360// the section boundaries in the raw source. 361// 362// Why: a naive `^(#{1,N})\s+...` regex pulls in `##` lines that live inside 363// a ` ```markdown ` fence, which would silently truncate the copy at the 364// first inner heading. Tracking fence state line-by-line is enough. 365// 366// Fence rules (CommonMark-shaped, just enough for this use case): 367// - opener: ^[ ]{0,3}(`{3,}|~{3,}).*$ 368// - closer: same fence char, count >= opener's count, only whitespace 369// after the closing run 370function findTopHeadings(md) { 371 if (typeof md !== 'string') return []; 372 var out = []; 373 var lines = md.split('\n'); 374 var fence = null; // null when outside; e.g. '```' or '~~~~' when inside 375 var pos = 0; 376 for (var i = 0; i < lines.length; i++) { 377 var line = lines[i]; 378 var trimmed = line.replace(/^[ \t]{0,3}/, ''); 379 var fenceMatch = /^(`{3,}|~{3,})/.exec(trimmed); 380 if (fence) { 381 if (fenceMatch && fenceMatch[1][0] === fence[0] 382 && fenceMatch[1].length >= fence.length 383 && /^\s*$/.test(trimmed.slice(fenceMatch[1].length))) { 384 fence = null; 385 } 386 } else if (fenceMatch) { 387 fence = fenceMatch[1]; 388 } else { 389 var h = /^(#{1,6})\s+(.+?)\s*$/.exec(line); 390 if (h) { 391 out.push({ index: pos, level: h[1].length, text: h[2].trim() }); 392 } 393 } 394 pos += line.length + 1; // +1 for the consumed '\n' 395 } 396 return out; 397} 398 399// Slice `md` to the substring covered by the section whose ATX heading 400// matches (level, headingText). Section ends at the next heading of equal 401// or higher rank, ignoring any headings inside fenced code blocks. Returns 402// null if the heading isn't found. 403function findSectionRange(md, level, headingText) { 404 var headings = findTopHeadings(md); 405 var startIdx = -1, startI = -1; 406 for (var i = 0; i < headings.length; i++) { 407 var h = headings[i]; 408 if (h.level === level && h.text === headingText) { 409 startIdx = h.index; 410 startI = i; 411 break; 412 } 413 } 414 if (startIdx === -1) return null; 415 var endIdx = md.length; 416 for (var j = startI + 1; j < headings.length; j++) { 417 if (headings[j].level <= level) { endIdx = headings[j].index; break; } 418 } 419 return { startIdx: startIdx, endIdx: endIdx, body: md.slice(startIdx, endIdx) }; 420} 421 422// ââ Copy serializers ââââââââââââââââââââââââââââââââââââââââââââââââââââ 423 424/** 425 * serializeFootnotes(meta, body) -> string 426 * 427 * Emits the body with each inline comment's quote transformed into a 428 * `[quote][^cN]` footnote reference, and the comment texts appended as 429 * `[^cN]: author - text` lines. Block comments become footnote refs 430 * attached at the end of the document. 431 * 432 * Matching strategy, in priority order: 433 * 1. `c.occurrence` (number, 0-indexed) â the Nth occurrence of `quote` 434 * in `body`. Caller computes this from the rendered DOM at copy time 435 * so we always pick the same occurrence the user actually anchored 436 * to, even when source markdown differs from rendered text (bold, 437 * links, code formatting, etc.). 438 * 2. `prefix + quote + suffix` matches uniquely â land there. 439 * 3. `quote` alone â first occurrence (best-effort fallback). 440 * 441 * Replacement positions are computed against the original body in one 442 * pass, then applied in descending order. This means earlier comments' 443 * positions are never shifted by later replacements, AND occurrence 444 * counts always run against the unedited source text â so two comments 445 * targeting different occurrences of the same quote can both land 446 * correctly. 447 */ 448function serializeFootnotes(meta, body) { 449 if (typeof body !== 'string') return body; 450 var comments = getComments(meta); 451 if (!comments.length) return body; 452 var inlineComments = comments.filter(function (c) { return c.kind === 'inline' && c.quote; }); 453 var hits = [];
454 inlineComments.forEach(function (c) { 455 var pos = -1; 456 if (typeof c.occurrence === 'number' && c.occurrence >= 0) { 457 pos = nthIndexOf(body, c.quote, c.occurrence); 458 } 459 if (pos === -1) { 460 var needle = (c.prefix || '') + c.quote + (c.suffix || ''); 461 var idx = body.indexOf(needle); 462 if (idx !== -1) pos = idx + (c.prefix || '').length; 463 } 464 if (pos === -1) pos = body.indexOf(c.quote); 465 if (pos !== -1) { 466 hits.push({ start: pos, end: pos + c.quote.length, id: c.id, quote: c.quote }); 467 } 468 }); 469 hits.sort(function (a, b) { return b.start - a.start; }); 470 // Skip a hit whose range overlaps with one we've already replaced (we 471 // walk right-to-left, so "already replaced" means a hit further right). 472 // Without this guard, slice/replace math would corrupt the body when a 473 // later comment's range falls inside an earlier comment's range. The 474 // dropped hit still gets a footnote definition appended below â only 475 // its inline anchor is omitted. 476 var out = body; 477 var minRightEdge = body.length + 1; 478 for (var i = 0; i < hits.length; i++) { 479 var h = hits[i]; 480 if (h.end > minRightEdge) continue; 481 var replacement = '[' + h.quote + '][^' + h.id + ']'; 482 out = out.slice(0, h.start) + replacement + out.slice(h.end); 483 minRightEdge = h.start; 484 } 485 var footnotes = []; 486 comments.forEach(function (c) { 487 var label = sanitizeText(c.author || 'user'); 488 if (c.resolved) label += ' [resolved]'; 489 label += ' - ' + sanitizeText(c.text || ''); 490 if (c.kind === 'block' && c.block) { 491 var blockTag = c.block; 492 if (c.block_text) blockTag += ' "' + sanitizeText(c.block_text) + '..."'; 493 label += ' (block ' + blockTag + ')'; 494 } 495 if (c.kind === 'element' && c.block && c.element) { 496 var elementTag = c.block + ' > ' + c.element; 497 if (c.element_scope === 'branch') elementTag += ' branch'; 498 if (c.element_text) elementTag += ' "' + sanitizeText(c.element_text) + '..."'; 499 label += ' (element ' + elementTag + ')'; 500 } 501 if (c.kind === 'table' && c.block) { 502 var tableTag = c.block; 503 if (c.table_scope === 'row' || c.table_scope === 'cell') { 504 tableTag += ', row ' + (c.table_row + 1); 505 if (c.row_text) tableTag += ' "' + sanitizeText(c.row_text) + '"'; 506 } 507 if (c.table_scope === 'column' || c.table_scope === 'cell') { 508 tableTag += ', column ' + (c.table_column + 1); 509 if (c.column_text) tableTag += ' "' + sanitizeText(c.column_text) + '"'; 510 } 511 if (c.table_scope === 'cell' && c.cell_text) { 512 tableTag += ', cell "' + sanitizeText(c.cell_text) + '"'; 513 } 514 label += ' (table ' + tableTag + ')'; 515 } 516 if (c.kind === 'slide') { 517 // Slides don't anchor into body text (their source is the fenced DSL, 518 // which we never rewrite), so the location rides entirely in the 519 // footnote label. Slide number is 1-based to match what the user sees 520 // in the present-mode counter and the slide badge. When the note 521 // targets one shape, carry its index (matches `data-shape-idx` and the 522 // shape's line in the slide source) plus the shape's visible text. 523 var slideTag = 'slide ' + ((typeof c.slide === 'number' ? c.slide : 0) + 1); 524 if (Array.isArray(c.shapes) && c.shapes.length) { 525 // Multi-element note: list the indices joined with '+', then one 526 // quoted hint covering all of them (the UI joins the labels). 527 slideTag += ', elements ' + c.shapes.join('+'); 528 if (c.slide_text) slideTag += ' "' + sanitizeText(c.slide_text) + '"'; 529 } else if (typeof c.shape === 'number') { 530 slideTag += ', element ' + c.shape; 531 if (c.slide_text) slideTag += ' "' + sanitizeText(c.slide_text) + '"'; 532 } else if (c.slide_text) { 533 slideTag += ' "' + sanitizeText(c.slide_text) + '"'; 534 } 535 label += ' (' + slideTag + ')'; 536 } 537 footnotes.push('[^' + c.id + ']: ' + label); 538 }); 539 return out.replace(/\n*$/, '') + '\n\n' + footnotes.join('\n') + '\n'; 540} 541 542// Index of the n-th (0-indexed) occurrence of `needle` in `hay`, or -1. 543function nthIndexOf(hay, needle, n) { 544 if (!needle) return -1; 545 var pos = -1, from = 0; 546 for (var i = 0; i <= n; i++) { 547 pos = hay.indexOf(needle, from); 548 if (pos === -1) return -1; 549 from = pos + needle.length; 550 } 551 return pos; 552} 553 554/** 555 * serializeClean(meta, body) -> string 556 * 557 * Returns the body verbatim. Supplied as the "strip all comments" flow; 558 * since the sidecar model never injected anything into the body, this 559 * is literally the identity function - but we wrap it for symmetry with 560 * the other serializers. 561 */ 562function serializeClean(meta, body) { return body; } 563 564/** 565 * parseFootnotes(body) -> { comments, body } 566 * 567 * Inverse of serializeFootnotes. Recognises markdown footnotes whose ids 568 * follow our convention (cN where N is digits) and converts them into 569 * comment objects. Other footnote ids are left alone - academic citations 570 * and the like keep their footnote semantics. 571 * 572 * Recognised patterns: 573 * Inline: [quote][^cN] anchor span = quote, kind = inline 574 * Block: ...end of paragraph.[^cN] kind = block, anchor = containing 575 * paragraph (block_text = first ~60 576 * chars of the paragraph the marker 577 * sits in) 578 * Defn: [^cN]: author - text [resolved]? the comment text 579 * 580 * The returned `body` has the recognised refs/defs stripped, ready to feed 581 * to the markdown renderer. Unrecognised footnotes pass through unchanged. 582 * 583 * `block_text` is computed from the body string at parse time, before 584 * marked rendering. For block comments, this is the survival hint that 585 * lets the existing tier-2 resolver attach the comment to the right 586 * block in the rendered DOM (where tag:n indices are computed). 587 */ 588function parseFootnotes(body) { 589 if (typeof body !== 'string' || body.indexOf('[^c') === -1) { 590 return { comments: [], body: body }; 591 } 592 var ID_RE = /^c\d+$/; 593 // Pull out definitions first. Anchored at line start; tolerate trailing 594 // whitespace/newlines. Also tolerate `[^cN]:`-only lines (no text). 595 var defs = {}; 596 var DEF_RE = /^\[\^(c\d+)\]:[ \t]*(.*)$/; 597 var lines = body.split('\n'); 598 var keptLines = []; 599 for (var i = 0; i < lines.length; i++) { 600 var m = lines[i].match(DEF_RE); 601 if (m) { 602 defs[m[1]] = (m[2] || '').trim(); 603 } else { 604 keptLines.push(lines[i]); 605 } 606 } 607 body = keptLines.join('\n'); 608
609 // Pull author + text + resolved-marker out of a definition string. 610 // Format emitted by serializeFootnotes: "<author>[ [resolved]] - <text>". 611 function decodeDef(raw) { 612 var out = { text: raw || '', resolved: false, author: undefined }; 613 if (!raw) return out; 614 if (/\[resolved\]/.test(raw)) out.resolved = true; 615 // Try to split on the first " - " that comes after a plausible author token. 616 var split = raw.match(/^(.+?)\s-\s(.+)$/); 617 if (split) { 618 var head = split[1].replace(/\s*\[resolved\]\s*$/, '').trim(); 619 // Heuristic: treat head as an author handle if it's compact (no inner punctuation 620 // beyond `.`, `_`, `-`, brackets, spaces) and reasonably short. 621 if (head.length <= 40 && /^[\w][\w \[\].\-_]*$/.test(head)) { 622 out.author = head; 623 out.text = split[2]; 624 } 625 } 626 // Trailing "(block tag:n)" hint, if present, gets stripped from text and 627 // surfaced separately. 628 var tag = (out.text || '').match(/\s*\(block\s+(\w+:\d+)\)\s*$/); 629 if (tag) { 630 out.block = tag[1]; 631 out.text = out.text.replace(/\s*\(block\s+\w+:\d+\)\s*$/, ''); 632 } 633 // Trailing nested-element hint emitted for list-item comments. 634 var eTag = (out.text || '').match( 635 /\s*\(element\s+(\w+:\d+)\s+>\s+((?:li|ul|ol):\d+(?:\/(?:li|ul|ol):\d+)*)(\s+branch)?(?:\s+"([^"]*)\.\.\.")?\)\s*$/); 636 if (eTag) { 637 out.block = eTag[1]; 638 out.element = eTag[2]; 639 out.element_scope = eTag[3] ? 'branch' : 'self'; 640 if (eTag[4]) out.element_text = eTag[4]; 641 out.text = out.text.replace( 642 /\s*\(element\s+\w+:\d+\s+>\s+(?:li|ul|ol):\d+(?:\/(?:li|ul|ol):\d+)*(?:\s+branch)?(?:\s+"[^"]*\.\.\.")?\)\s*$/, 643 ''); 644 } 645 // Trailing table target hint. Row and column numbers are 1-based in the 646 // copied footnote so they match what the reader sees, then return to 647 // 0-based indices in the in-memory model. 648 var tTag = (out.text || '').match( 649 /\s*\(table\s+(table:\d+)(?:,\s*row\s+(\d+)(?:\s+"([^"]*)")?)?(?:,\s*column\s+(\d+)(?:\s+"([^"]*)")?)?(?:,\s*cell\s+"([^"]*)")?\)\s*$/); 650 if (tTag) { 651 out.block = tTag[1]; 652 if (tTag[2] != null) out.table_row = parseInt(tTag[2], 10) - 1; 653 if (tTag[3]) out.row_text = tTag[3]; 654 if (tTag[4] != null) out.table_column = parseInt(tTag[4], 10) - 1; 655 if (tTag[5]) out.column_text = tTag[5]; 656 if (tTag[6]) out.cell_text = tTag[6]; 657 out.table_scope = tTag[6] != null ? 'cell' 658 : (tTag[4] != null ? 'column' : (tTag[2] != null ? 'row' : 'table')); 659 out.text = out.text.replace( 660 /\s*\(table\s+table:\d+(?:,\s*row\s+\d+(?:\s+"[^"]*")?)?(?:,\s*column\s+\d+(?:\s+"[^"]*")?)?(?:,\s*cell\s+"[^"]*")?\)\s*$/, 661 ''); 662 } 663 // Trailing "(slide N[, element M][ "text"])" hint - the inverse of the 664 // slide label serializeFootnotes emits. Slide number is 1-based on disk; 665 // store it 0-based to match the in-memory anchor. 666 var sTag = (out.text || '').match(/\s*\(slide\s+(\d+)(?:,\s*elements?\s+([\d+]+))?(?:\s+"([^"]*)")?\)\s*$/); 667 if (sTag) { 668 out.slide = parseInt(sTag[1], 10) - 1; 669 if (sTag[2] != null) { 670 var idxs = sTag[2].split('+').map(function (x) { return parseInt(x, 10); }) 671 .filter(function (n) { return !isNaN(n); }); 672 if (idxs.length > 1) out.shapes = idxs; 673 else if (idxs.length === 1) out.shape = idxs[0]; 674 } 675 if (sTag[3]) out.slide_text = sTag[3]; 676 out.text = out.text.replace(/\s*\(slide\s+\d+(?:,\s*elements?\s+[\d+]+)?(?:\s+"[^"]*")?\)\s*$/, ''); 677 } 678 return out; 679 } 680 681 var comments = []; 682 var seen = {}; 683 684 // Inline: [quote][^cN] 685 body = body.replace(/\[([^\]\n]+?)\]\[\^(c\d+)\]/g, function (_m, quote, id) { 686 if (!ID_RE.test(id)) return _m; 687 if (seen[id]) return _m; 688 seen[id] = true; 689 var d = decodeDef(defs[id]); 690 var c = { id: id, kind: 'inline', quote: quote, text: d.text }; 691 if (d.author) c.author = d.author; 692 if (d.resolved) c.resolved = true; 693 comments.push(c); 694 return quote; 695 }); 696 697 // Block markers: lone [^cN] not preceded by `]`. The surrounding paragraph 698 // becomes the block_text survival hint. 699 var REF_RE = /\[\^(c\d+)\]/g; 700 var match; 701 while ((match = REF_RE.exec(body)) !== null) { 702 var id = match[1]; 703 if (!ID_RE.test(id) || seen[id]) continue; 704 var refStart = match.index; 705 if (body[refStart - 1] === ']') continue; 706 seen[id] = true; 707 var paraStart = body.lastIndexOf('\n\n', refStart); 708 paraStart = paraStart === -1 ? 0 : paraStart + 2; 709 var paraEnd = body.indexOf('\n\n', refStart); 710 if (paraEnd === -1) paraEnd = body.length; 711 var paragraph = body.slice(paraStart, paraEnd); 712 var clean = paragraph.replace(REF_RE, '').trim(); 713 var d2 = decodeDef(defs[id]); 714 var c2 = { id: id, kind: 'block', block_text: clean.slice(0, 60), text: d2.text }; 715 if (d2.block) c2.block = d2.block; 716 if (d2.author) c2.author = d2.author; 717 if (d2.resolved) c2.resolved = true; 718 comments.push(c2); 719 } 720 body = body.replace(REF_RE, function (_m, id) { return seen[id] ? '' : _m; }); 721 722 // Orphan definitions (no body marker). Treat as block comments. This is the 723 // shape serializeFootnotes emits today for block-kind comments - the def 724 // carries a `(block tag:n)` hint and there is no body marker.
725 Object.keys(defs).forEach(function (id) { 726 if (!ID_RE.test(id) || seen[id]) return; 727 seen[id] = true; 728 var d3 = decodeDef(defs[id]); 729 var c3; 730 if (d3.table_scope) { 731 c3 = { 732 id: id, kind: 'table', block: d3.block, 733 table_scope: d3.table_scope, text: d3.text, 734 }; 735 if (d3.table_row != null) c3.table_row = d3.table_row; 736 if (d3.table_column != null) c3.table_column = d3.table_column; 737 if (d3.row_text) c3.row_text = d3.row_text; 738 if (d3.column_text) c3.column_text = d3.column_text; 739 if (d3.cell_text) c3.cell_text = d3.cell_text; 740 } else if (d3.element) { 741 c3 = { 742 id: id, kind: 'element', block: d3.block, element: d3.element, 743 element_scope: d3.element_scope, text: d3.text, 744 }; 745 if (d3.element_text) c3.element_text = d3.element_text; 746 } else if (typeof d3.slide === 'number') { 747 // Slide notes carry a "(slide N ...)" hint and no body marker, so they 748 // land here as orphan definitions. Rebuild the slide anchor. 749 c3 = { id: id, kind: 'slide', slide: d3.slide, text: d3.text }; 750 if (Array.isArray(d3.shapes)) c3.shapes = d3.shapes; 751 else if (typeof d3.shape === 'number') c3.shape = d3.shape; 752 if (d3.slide_text) c3.slide_text = d3.slide_text; 753 } else { 754 c3 = { id: id, kind: 'block', text: d3.text }; 755 if (d3.block) c3.block = d3.block; 756 } 757 if (d3.author) c3.author = d3.author; 758 if (d3.resolved) c3.resolved = true; 759 comments.push(c3); 760 }); 761 762 body = body.replace(/\n{3,}$/, '\n').replace(/[ \t]+$/gm, ''); 763 // Return in chronological (id) order rather than parse-pass order. 764 comments.sort(function (a, b) { 765 var na = parseInt((a.id || '').slice(1), 10); 766 var nb = parseInt((b.id || '').slice(1), 10); 767 return na - nb; 768 }); 769 return { comments: comments, body: body }; 770} 771 772// ââ Public API ââââââââââââââââââââââââââââââââââââââââââââââââââââââââââ 773 774exports.sanitizeColor = sanitizeColor; 775exports.isValidId = isValidId; 776exports.getComments = getComments; 777exports.setComments = setComments; 778exports.nextId = nextId; 779exports.normalizeComment = normalizeComment; 780exports.addSelectionComment = addSelectionComment; 781exports.addBlockComment = addBlockComment; 782exports.addElementComment = addElementComment; 783exports.addTableComment = addTableComment; 784exports.addSlideComment = addSlideComment; 785exports.removeComment = removeComment; 786exports.updateComment = updateComment; 787exports.serializeFootnotes = serializeFootnotes; 788exports.parseFootnotes = parseFootnotes; 789exports.serializeClean = serializeClean; 790exports.findTopHeadings = findTopHeadings; 791exports.findSectionRange = findSectionRange; 792 793})(typeof module !== 'undefined' && module.exports ? module.exports : (window.SDocComments = {}));
Line numbers count LF bytes from the start of the resource, as the search results do. Vendor segments are library code the classifier recognised; they are stored but not indexed. Bytes are shown as Latin1 characters, one per byte.