1// Check doc for content 2async function googleDocHasContent(publishedUrl) { 3 const url = publishedUrl + (publishedUrl.includes('?') ? '&' : '?') + 'v=' + Date.now(); 4 const res = await fetch(url, { cache: "no-store" }); 5 if (!res.ok) return false; 6 7 const html = await res.text(); 8 const doc = new DOMParser().parseFromString(html, 'text/html'); 9 10 // Prefer #contents, fallback to body 11 const contents = doc.querySelector('#contents') || doc.body; 12 if (!contents) return false; 13 14 // Work on a copy so we don't care what doc is doing 15 const copy = contents.cloneNode(true); 16 17 // â Remove non-content sources that pollute textContent 18 copy.querySelectorAll('style, script, noscript').forEach(el => el.remove()); 19 copy.querySelector('#publish-banner')?.remove(); 20 copy.querySelector('#footer')?.remove(); 21 22 // â Remove elements that are hidden (sometimes Docs includes hidden spans) 23 copy.querySelectorAll('[style*="display:none"], [hidden], .hidden').forEach(el => el.remove()); 24 25 // Consider meaningful content to be inside paragraphs/list items/headings/table cells 26 const meaningful = Array.from(copy.querySelectorAll('p, li, h1, h2, h3, h4, td, th')) 27 .map(el => (el.textContent || '').replace(/\u00a0/g, ' ').replace(/\s+/g, ' ').trim()) 28 .some(t => t.length > 0); 29 30 return meaningful; 31}
Line numbers count LF bytes from the start of the resource, as the search results do. Vendor segments are library code the classifier recognised; they are stored but not indexed. Bytes are shown as Latin1 characters, one per byte.