PageSourceSearch

https://dynamic-salmiakki-095693.netlify.app/js/utils/html-to-pdf-generator.js?v=1781614395

js dynamic-salmiakki-095693.netlify.app collected 2026-10-03 10:40:42 UTC 10,735 bytes, 277 lines download raw bytes

1// HTML-to-PDF Generator using Puppeteer backend
2// This replaces the jsPDF approach with a proper HTML-to-PDF solution
3
4/**
5 * Generates a PDF from the current conversation and infoboxes using HTML-to-PDF conversion
6 * @param {HTMLElement} chatContainer - The chat messages container
7 * @param {HTMLElement} infoboxesContainer - The Wikipedia infoboxes container
8 * @param {string} title - Optional title for the PDF
9 */
10async function generateConversationPDF(chatContainer, infoboxesContainer, title = 'WikiLive Conversation') {
11    try {
12        console.log('[HTML-to-PDF] Starting PDF generation...');
13        
14        // Collect all conversation content
15        const conversationHtml = collectConversationHTML(chatContainer, infoboxesContainer);
16        
17        if (!conversationHtml || conversationHtml.trim().length === 0) {
18            throw new Error('No conversation content found to export');
19        }
20
21        console.log('[HTML-to-PDF] Sending content to backend for PDF generation...');
22        
23        // Send HTML content to Puppeteer-based PDF generator
24        const response = await fetch('/.netlify/functions/export-conversation-pdf-working', {
25            method: 'POST',
26            headers: {
27                'Content-Type': 'application/json',
28            },
29            body: JSON.stringify({
30                htmlContent: conversationHtml,
31                title: title,
32                options: {
33                    format: 'A4',
34                    printBackground: true,
35                    margin: {
36                        top: '20px',
37                        right: '20px',
38                        bottom: '20px',
39                        left: '20px'
40                    }
41                }
42            })
43        });
44
45        if (!response.ok) {
46            const errorData = await response.json().catch(() => ({}));
47            
48            // Check if this is a structured fallback response
49            if (errorData.fallback === true || response.status === 503) {
50                console.log('[HTML-to-PDF] Received fallback signal from backend, switching to legacy PDF generator...');
51                if (typeof generateWikiLivePDF === 'function') {
52                    return await generateWikiLivePDF(chatContainer, infoboxesContainer);
53                } else {
54                    throw new Error('Legacy PDF generator not available');
55                }
56            }
57            
58            throw new Error(errorData.error || `HTTP ${response.status}: ${response.statusText}`);
59        }
60
61        console.log('[HTML-to-PDF] PDF generated successfully, downloading...');
62
63        // Check if response is actually a PDF or a fallback JSON response
64        const contentType = response.headers.get('content-type');
65        if (contentType && contentType.includes('application/json')) {
66            // This is a JSON response, likely a fallback signal
67            const jsonData = await response.json();
68            if (jsonData.fallback === true) {
69                console.log('[HTML-to-PDF] Backend returned fallback signal, switching to legacy PDF generator...');
70                if (typeof generateWikiLivePDF === 'function') {
71                    return await generateWikiLivePDF(chatContainer, infoboxesContainer);
72                } else {
73                    throw new Error('Legacy PDF generator not available');
74                }
75            }
76            throw new Error(jsonData.error || 'Unexpected JSON response from PDF service');
77        }
78
79        // Get the PDF blob and trigger download
80        const pdfBlob = await response.blob();
81        const url = window.URL.createObjectURL(pdfBlob);
82        
83        const a = document.createElement('a');
84        a.href = url;
85        a.download = `${title.replace(/[^a-zA-Z0-9]/g, '_')}_${new Date().toISOString().split('T')[0]}.pdf`;
86        document.body.appendChild(a);
87        a.click();
88        document.body.removeChild(a);
89        window.URL.revokeObjectURL(url);
90
91        console.log('[HTML-to-PDF] PDF download initiated successfully');
92        return true;
93
94    } catch (error) {
95        console.error('HTML-to-PDF generation failed:', error);
96        
97        // Check if this is a service unavailable error (fallback recommended)
98        if (error.message && (error.message.includes('503') || error.message.includes('Puppeteer') || error.message.includes('temporarily unavailable'))) {
99            console.log('Puppeteer service unavailable, falling back to legacy PDF generator...');
100            // Fall back to legacy jsPDF method
101            if (typeof generateWikiLivePDF === 'function') {
102                return await generateWikiLivePDF(chatContainer, infoboxesContainer);
103            } else {
104                throw new Error('Both HTML-to-PDF and legacy PDF generators are unavailable');
105            }
106        }
107        
108        // For other errors, try fallback anyway
109        console.log('Attempting fallback to legacy PDF generator...');
110        if (typeof generateWikiLivePDF === 'function') {
111            return await generateWikiLivePDF(chatContainer, infoboxesContainer);
112        }
113        
114        throw error;
115    }
116}
117
118/**
119 * Collects and formats HTML content from conversation and infoboxes
120 * @param {HTMLElement} chatContainer 
121 * @param {HTMLElement} infoboxesContainer 
122 * @returns {string} Formatted HTML content
123 */
124function collectConversationHTML(chatContainer, infoboxesContainer) {
125    let html = '';
126
127    // Process chat messages
128    if (chatContainer) {
129        const messages = chatContainer.querySelectorAll('.message');
130        
131        messages.forEach((message, index) => {
132            const isUser = message.classList.contains('user');
133            const messageContent = cleanHTMLForPDF(message.cloneNode(true));
134            
135            if (messageContent && messageContent.textContent.trim()) {
136                html += `
137                    <div class="message ${isUser ? 'user' : 'assistant'}">
138                        <div class="message-label">${isUser ? 'User' : 'Assistant'}</div>
139                        <div class="message-content">
140                            ${messageContent.innerHTML}
141                        </div>
142                    </div>
143                `;
144            }
145        });
146    }
147
148    // Process infoboxes
149    if (infoboxesContainer) {
150        const infoboxes = infoboxesContainer.querySelectorAll('.wikipedia-infobox');
151        
152        if (infoboxes.length > 0) {
153            html += '<div class="page-break"></div>';
154            html += '<h2 style="color: #667eea; margin: 30px 0 20px 0; font-size: 24px;">Wikipedia Information</h2>';
155            
156            infoboxes.forEach((infobox) => {
157                const cleanedInfobox = cleanHTMLForPDF(infobox.cloneNode(true));
158                if (cleanedInfobox) {
159                    html += cleanedInfobox.outerHTML;
160                }
161            });
162        }
163    }
164
165    // Process AI summary boxes
166    const summaryBoxes = document.querySelectorAll('.ai-summary-box, .realtime-ai-facts-box');
167    if (summaryBoxes.length > 0) {
168        summaryBoxes.forEach((box) => {
169            const cleanedBox = cleanHTMLForPDF(box.cloneNode(true));
170            if (cleanedBox && cleanedBox.textContent.trim()) {
171                html += cleanedBox.outerHTML;
172            }
173        });
174    }
175
176    return html;
177}
178
179/**
180 * Cleans HTML content for PDF generation
181 * @param {HTMLElement} element 
182 * @returns {HTMLElement} Cleaned element
183 */
184function cleanHTMLForPDF(element) {
185    if (!element) return null;
186
187    // Remove elements that shouldn't appear in PDF
188    const unwantedSelectors = [
189        'script', 'style', 'button', '.pdf-button', '.no-print', 
190        '[data-pdf-ignore="true"]', '.copy-button', '.speak-button',
191        '.loading-indicator', '.skeleton-loading'
192    ];
193    
194    unwantedSelectors.forEach(selector => {
195        element.querySelectorAll(selector).forEach(el => el.remove());
196    });
197
198    // Convert relative image URLs to absolute URLs
199    element.querySelectorAll('img').forEach(img => {
200        if (img.src && !img.src.startsWith('http')) {
201            // Convert relative URLs to absolute
202            img.src = new URL(img.src, window.location.origin).href;
203        }
204        
205        // Add alt text if missing
206        if (!img.alt && img.title) {
207            img.alt = img.title;
208        }
209    });
210
211    // Convert links to absolute URLs
212    element.querySelectorAll('a').forEach(link => {
213        if (link.href && !link.href.startsWith('http')) {
214            link.href = new URL(link.href, window.location.origin).href;
215        }
216    });
217
218    // Clean up empty elements
219    element.querySelectorAll('*').forEach(el => {
220        if (!el.textContent.trim() && !el.querySelector('img') && !el.querySelector('video')) {
221            // Keep elements that might have important styling or structure
222            if (!['br', 'hr', 'img', 'video', 'audio'].includes(el.tagName.toLowerCase())) {
223                if (el.children.length === 0) {
224                    el.remove();
225                }
226            }
227        }
228    });
229
230    return element;
231}
232
233/**
234 * Legacy fallback to jsPDF if Puppeteer fails
235 */
236async function fallbackToJsPDF(chatContainer, infoboxesContainer) {
237    console.log('[HTML-to-PDF] Falling back to jsPDF...');
238    
239    // Check if the legacy PDF generator is available
240    if (typeof window.generatePDF === 'function') {
241        return await window.generatePDF(chatContainer, infoboxesContainer);
242    } else if (window.PdfGenerator && typeof window.PdfGenerator.generate === 'function') {
243        return await window.PdfGenerator.generate(chatContainer, infoboxesContainer);
244    } else {
245        throw new Error('No PDF generation method available');
246    }
247}
248
249// Export functions
250if (typeof module !== 'undefined' && module.exports) {
251    module.exports = {
252        generateConversationPDF,
253        collectConversationHTML,
254        cleanHTMLForPDF,
255        fallbackToJsPDF
256    };
257} else {
258    // Browser environment
259    window.HtmlToPdfGenerator = {
260        generate: generateConversationPDF,
261        collectHTML: collectConversationHTML,
262        cleanHTML: cleanHTMLForPDF,
263        fallback: fallbackToJsPDF
264    };
265    
266    // Override the existing generatePDF function to use the new approach
267    window.generatePDF = async function(chatContainer, infoboxesContainer, title) {
268        try {
269            return await generateConversationPDF(chatContainer, infoboxesContainer, title);
270        } catch (error) {
271            console.warn('[HTML-to-PDF] Primary method failed, trying fallback:', error.message);
272            return await fallbackToJsPDF(chatContainer, infoboxesContainer);
273        }
274    };
275}
276
277console.log('[HTML-to-PDF] HTML-to-PDF Generator loaded and ready');

Line numbers count LF bytes from the start of the resource, as the search results do. Vendor segments are library code the classifier recognised; they are stored but not indexed. Bytes are shown as Latin1 characters, one per byte.