1// HTML-to-PDF Generator using Puppeteer backend 2// This replaces the jsPDF approach with a proper HTML-to-PDF solution 3 4/** 5 * Generates a PDF from the current conversation and infoboxes using HTML-to-PDF conversion 6 * @param {HTMLElement} chatContainer - The chat messages container 7 * @param {HTMLElement} infoboxesContainer - The Wikipedia infoboxes container 8 * @param {string} title - Optional title for the PDF 9 */ 10async function generateConversationPDF(chatContainer, infoboxesContainer, title = 'WikiLive Conversation') { 11 try { 12 console.log('[HTML-to-PDF] Starting PDF generation...'); 13 14 // Collect all conversation content 15 const conversationHtml = collectConversationHTML(chatContainer, infoboxesContainer); 16 17 if (!conversationHtml || conversationHtml.trim().length === 0) { 18 throw new Error('No conversation content found to export'); 19 } 20 21 console.log('[HTML-to-PDF] Sending content to backend for PDF generation...'); 22 23 // Send HTML content to Puppeteer-based PDF generator 24 const response = await fetch('/.netlify/functions/export-conversation-pdf-working', { 25 method: 'POST', 26 headers: { 27 'Content-Type': 'application/json', 28 }, 29 body: JSON.stringify({ 30 htmlContent: conversationHtml, 31 title: title, 32 options: { 33 format: 'A4', 34 printBackground: true, 35 margin: { 36 top: '20px', 37 right: '20px', 38 bottom: '20px', 39 left: '20px' 40 } 41 } 42 }) 43 }); 44 45 if (!response.ok) { 46 const errorData = await response.json().catch(() => ({})); 47 48 // Check if this is a structured fallback response 49 if (errorData.fallback === true || response.status === 503) { 50 console.log('[HTML-to-PDF] Received fallback signal from backend, switching to legacy PDF generator...'); 51 if (typeof generateWikiLivePDF === 'function') { 52 return await generateWikiLivePDF(chatContainer, infoboxesContainer); 53 } else { 54 throw new Error('Legacy PDF generator not available'); 55 } 56 } 57 58 throw new Error(errorData.error || `HTTP ${response.status}: ${response.statusText}`); 59 } 60 61 console.log('[HTML-to-PDF] PDF generated successfully, downloading...'); 62 63 // Check if response is actually a PDF or a fallback JSON response 64 const contentType = response.headers.get('content-type'); 65 if (contentType && contentType.includes('application/json')) { 66 // This is a JSON response, likely a fallback signal 67 const jsonData = await response.json(); 68 if (jsonData.fallback === true) { 69 console.log('[HTML-to-PDF] Backend returned fallback signal, switching to legacy PDF generator...'); 70 if (typeof generateWikiLivePDF === 'function') { 71 return await generateWikiLivePDF(chatContainer, infoboxesContainer); 72 } else { 73 throw new Error('Legacy PDF generator not available'); 74 } 75 } 76 throw new Error(jsonData.error || 'Unexpected JSON response from PDF service'); 77 } 78 79 // Get the PDF blob and trigger download 80 const pdfBlob = await response.blob(); 81 const url = window.URL.createObjectURL(pdfBlob); 82 83 const a = document.createElement('a'); 84 a.href = url; 85 a.download = `${title.replace(/[^a-zA-Z0-9]/g, '_')}_${new Date().toISOString().split('T')[0]}.pdf`; 86 document.body.appendChild(a); 87 a.click(); 88 document.body.removeChild(a); 89 window.URL.revokeObjectURL(url); 90 91 console.log('[HTML-to-PDF] PDF download initiated successfully'); 92 return true; 93 94 } catch (error) { 95 console.error('HTML-to-PDF generation failed:', error); 96 97 // Check if this is a service unavailable error (fallback recommended) 98 if (error.message && (error.message.includes('503') || error.message.includes('Puppeteer') || error.message.includes('temporarily unavailable'))) { 99 console.log('Puppeteer service unavailable, falling back to legacy PDF generator...'); 100 // Fall back to legacy jsPDF method 101 if (typeof generateWikiLivePDF === 'function') { 102 return await generateWikiLivePDF(chatContainer, infoboxesContainer); 103 } else { 104 throw new Error('Both HTML-to-PDF and legacy PDF generators are unavailable'); 105 } 106 } 107
108 // For other errors, try fallback anyway 109 console.log('Attempting fallback to legacy PDF generator...'); 110 if (typeof generateWikiLivePDF === 'function') { 111 return await generateWikiLivePDF(chatContainer, infoboxesContainer); 112 } 113 114 throw error; 115 } 116} 117 118/** 119 * Collects and formats HTML content from conversation and infoboxes 120 * @param {HTMLElement} chatContainer 121 * @param {HTMLElement} infoboxesContainer 122 * @returns {string} Formatted HTML content 123 */ 124function collectConversationHTML(chatContainer, infoboxesContainer) { 125 let html = ''; 126 127 // Process chat messages 128 if (chatContainer) { 129 const messages = chatContainer.querySelectorAll('.message'); 130 131 messages.forEach((message, index) => { 132 const isUser = message.classList.contains('user'); 133 const messageContent = cleanHTMLForPDF(message.cloneNode(true)); 134 135 if (messageContent && messageContent.textContent.trim()) { 136 html += ` 137 <div class="message ${isUser ? 'user' : 'assistant'}"> 138 <div class="message-label">${isUser ? 'User' : 'Assistant'}</div> 139 <div class="message-content"> 140 ${messageContent.innerHTML} 141 </div> 142 </div> 143 `; 144 } 145 }); 146 } 147 148 // Process infoboxes 149 if (infoboxesContainer) { 150 const infoboxes = infoboxesContainer.querySelectorAll('.wikipedia-infobox'); 151 152 if (infoboxes.length > 0) { 153 html += '<div class="page-break"></div>'; 154 html += '<h2 style="color: #667eea; margin: 30px 0 20px 0; font-size: 24px;">Wikipedia Information</h2>'; 155 156 infoboxes.forEach((infobox) => { 157 const cleanedInfobox = cleanHTMLForPDF(infobox.cloneNode(true)); 158 if (cleanedInfobox) { 159 html += cleanedInfobox.outerHTML; 160 } 161 }); 162 } 163 } 164 165 // Process AI summary boxes 166 const summaryBoxes = document.querySelectorAll('.ai-summary-box, .realtime-ai-facts-box'); 167 if (summaryBoxes.length > 0) { 168 summaryBoxes.forEach((box) => { 169 const cleanedBox = cleanHTMLForPDF(box.cloneNode(true)); 170 if (cleanedBox && cleanedBox.textContent.trim()) { 171 html += cleanedBox.outerHTML; 172 } 173 }); 174 } 175 176 return html; 177} 178 179/** 180 * Cleans HTML content for PDF generation 181 * @param {HTMLElement} element 182 * @returns {HTMLElement} Cleaned element 183 */ 184function cleanHTMLForPDF(element) { 185 if (!element) return null; 186 187 // Remove elements that shouldn't appear in PDF 188 const unwantedSelectors = [ 189 'script', 'style', 'button', '.pdf-button', '.no-print', 190 '[data-pdf-ignore="true"]', '.copy-button', '.speak-button', 191 '.loading-indicator', '.skeleton-loading' 192 ]; 193 194 unwantedSelectors.forEach(selector => { 195 element.querySelectorAll(selector).forEach(el => el.remove()); 196 }); 197 198 // Convert relative image URLs to absolute URLs 199 element.querySelectorAll('img').forEach(img => { 200 if (img.src && !img.src.startsWith('http')) { 201 // Convert relative URLs to absolute 202 img.src = new URL(img.src, window.location.origin).href; 203 } 204 205 // Add alt text if missing 206 if (!img.alt && img.title) { 207 img.alt = img.title; 208 } 209 }); 210 211 // Convert links to absolute URLs 212 element.querySelectorAll('a').forEach(link => { 213 if (link.href && !link.href.startsWith('http')) { 214 link.href = new URL(link.href, window.location.origin).href; 215 } 216 }); 217 218 // Clean up empty elements 219 element.querySelectorAll('*').forEach(el => { 220 if (!el.textContent.trim() && !el.querySelector('img') && !el.querySelector('video')) { 221 // Keep elements that might have important styling or structure 222 if (!['br', 'hr', 'img', 'video', 'audio'].includes(el.tagName.toLowerCase())) { 223 if (el.children.length === 0) { 224 el.remove(); 225 } 226 } 227 } 228 }); 229 230 return element; 231} 232
233/** 234 * Legacy fallback to jsPDF if Puppeteer fails 235 */ 236async function fallbackToJsPDF(chatContainer, infoboxesContainer) { 237 console.log('[HTML-to-PDF] Falling back to jsPDF...'); 238 239 // Check if the legacy PDF generator is available 240 if (typeof window.generatePDF === 'function') { 241 return await window.generatePDF(chatContainer, infoboxesContainer); 242 } else if (window.PdfGenerator && typeof window.PdfGenerator.generate === 'function') { 243 return await window.PdfGenerator.generate(chatContainer, infoboxesContainer); 244 } else { 245 throw new Error('No PDF generation method available'); 246 } 247} 248 249// Export functions 250if (typeof module !== 'undefined' && module.exports) { 251 module.exports = { 252 generateConversationPDF, 253 collectConversationHTML, 254 cleanHTMLForPDF, 255 fallbackToJsPDF 256 }; 257} else { 258 // Browser environment 259 window.HtmlToPdfGenerator = { 260 generate: generateConversationPDF, 261 collectHTML: collectConversationHTML, 262 cleanHTML: cleanHTMLForPDF, 263 fallback: fallbackToJsPDF 264 }; 265 266 // Override the existing generatePDF function to use the new approach 267 window.generatePDF = async function(chatContainer, infoboxesContainer, title) { 268 try { 269 return await generateConversationPDF(chatContainer, infoboxesContainer, title); 270 } catch (error) { 271 console.warn('[HTML-to-PDF] Primary method failed, trying fallback:', error.message); 272 return await fallbackToJsPDF(chatContainer, infoboxesContainer); 273 } 274 }; 275} 276 277console.log('[HTML-to-PDF] HTML-to-PDF Generator loaded and ready');
Line numbers count LF bytes from the start of the resource, as the search results do. Vendor segments are library code the classifier recognised; they are stored but not indexed. Bytes are shown as Latin1 characters, one per byte.