1// Wikipedia Articles Aggregation Module 2// Extracted from main.js for better modularity 3 4/** 5 * Wikipedia Articles Aggregation System 6 * Handles fetching full Wikipedia articles, generating AI summaries, and displaying results 7 */ 8class WikipediaArticlesAggregator { 9 constructor() { 10 this.collectedArticles = new Map(); // title -> full article text 11 this.currentUserQuestion = ''; 12 } 13 14 /** 15 * Fetch full Wikipedia article content 16 * @param {string} title - Wikipedia page title 17 * @returns {Promise<string|null>} - Full article text or null if failed 18 */ 19 async fetchFullWikipediaArticle(title) { 20 try { 21 console.log(`[fetchFullWikipediaArticle] Fetching full article for: "${title}"`); 22 23 // Use Wikipedia's extract API to get full article content 24 const apiUrl = `https://en.wikipedia.org/w/api.php?action=query&format=json&titles=${encodeURIComponent(title)}&prop=extracts&explaintext=true&exsectionformat=plain&origin=*`; 25 26 const response = await fetch(apiUrl); 27 28 if (!response.ok) { 29 console.log(`[fetchFullWikipediaArticle] HTTP error for "${title}": ${response.status}`); 30 return null; 31 } 32 33 const data = await response.json(); 34 const pages = data.query?.pages; 35 if (!pages) { 36 console.log(`[fetchFullWikipediaArticle] No pages data for "${title}"`); 37 return null; 38 } 39 40 const page = Object.values(pages)[0]; 41 if (page.missing) { 42 console.log(`[fetchFullWikipediaArticle] Page missing for "${title}"`); 43 return null; 44 } 45 46 let extract = page.extract; 47 if (!extract) { 48 console.log(`[fetchFullWikipediaArticle] No extract for "${title}"`); 49 return null; 50 } 51 52 console.log(`[fetchFullWikipediaArticle] Successfully fetched ${extract.length} characters for "${title}"`); 53 return extract; 54 } catch (error) { 55 console.error(`[fetchFullWikipediaArticle] Error for "${title}":`, error); 56 return null; 57 } 58 } 59 60 /** 61 * Generate AI summary from collected articles 62 * @param {string} userQuestion - User's original question 63 * @param {Map} articlesMap - Map of title -> article content 64 * @param {Array} conversationHistory - Previous conversation context 65 * @returns {Promise<Object>} - AI summary data 66 */ 67 async generateAISummary(userQuestion, articlesMap, conversationHistory = []) { 68 const maxRetries = 3; 69 const timeoutMs = 25000; // 25 seconds to handle Netlify function timeouts 70 71 for (let attempt = 1; attempt <= maxRetries; attempt++) { 72 try { 73 console.log(`[generateAISummary] Attempt ${attempt}/${maxRetries} - Generating AI summary for question: "${userQuestion}"`); 74 console.log(`[generateAISummary] Processing ${articlesMap.size} articles`); 75 console.log(`[generateAISummary] Including ${conversationHistory.length} conversation turns`); 76 77 // Combine all articles into one large string with clear separators 78 let combinedArticles = ''; 79 const maxTotalLength = 20000; // 20k characters for ~7-8 articles 80 let currentLength = 0; 81 82 for (const [title, content] of articlesMap) { 83 const articleSection = `\n\n=== WIKIPEDIA ARTICLE: ${title} ===\n${content}`; 84 85 // Always include the first article 86 if (currentLength === 0) { 87 combinedArticles += articleSection; 88 currentLength += articleSection.length; 89 console.log(`[generateAISummary] Added first article "${title}": ${articleSection.length} characters, total: ${currentLength} characters`); 90 } else { 91 // Check limits for subsequent articles 92 if (currentLength + articleSection.length > maxTotalLength) { 93 console.log(`[generateAISummary] Reached total character limit. Processed ${combinedArticles.split('\n\n=== ').length - 1} articles`); 94 break; 95 } 96 97 combinedArticles += articleSection; 98 currentLength += articleSection.length; 99 console.log(`[generateAISummary] Added article "${title}": ${articleSection.length} characters, total: ${currentLength} characters`); 100 } 101 } 102 103 console.log(`[generateAISummary] Combined articles: ${combinedArticles.length} characters`); 104 105 if (combinedArticles.length === 0) { 106 throw new Error('No articles available for processing'); 107 } 108 109 // Prepare conversation context (last 6 messages, limited to ~2000 chars) 110 let conversationContext = ''; 111 if (conversationHistory && conversationHistory.length > 0) { 112 const recentHistory = conversationHistory.slice(-6); // Last 6 messages (3 turns) 113 conversationContext = recentHistory.map(msg => 114 `${msg.role === 'user' ? 'User:' : 'Assistant:'} ${msg.content}` 115 ).join('\n'); 116 117 // Limit conversation context to 2000 characters 118 if (conversationContext.length > 2000) { 119 conversationContext = conversationContext.substring(conversationContext.length - 2000); 120 // Try to start from a complete message 121 const firstNewline = conversationContext.indexOf('\n'); 122 if (firstNewline > 0) { 123 conversationContext = conversationContext.substring(firstNewline + 1); 124 } 125 } 126 console.log(`[generateAISummary] Conversation context: ${conversationContext.length} characters`); 127 } 128 129 // Send to our Netlify function for AI processing with timeout handling 130 const controller = new AbortController(); 131 const timeoutId = setTimeout(() => { 132 console.log(`[generateAISummary] Request timeout after ${timeoutMs}ms, aborting...`); 133 controller.abort(); 134 }, timeoutMs); 135 136 const response = await fetch('/.netlify/functions/wikipedia-summary', { 137 method: 'POST', 138 headers: { 139 'Content-Type': 'application/json' 140 }, 141 body: JSON.stringify({ 142 userQuestion: userQuestion, 143 combinedArticles: combinedArticles, 144 conversationContext: conversationContext, 145 articleCount: articlesMap.size 146 }), 147 signal: controller.signal 148 }); 149 150 clearTimeout(timeoutId); 151 if (!response.ok) { 152 const errorText = await response.text(); 153 throw new Error(`AI summary request failed: ${response.status} - ${errorText}`); 154 } 155 156 const result = await response.json(); 157 console.log(`[generateAISummary] Received AI summary: ${result.summary.length} characters`); 158 159 return result; 160 161 } catch (error) { 162 console.error(`[generateAISummary] Attempt ${attempt}/${maxRetries} failed:`, error); 163 164 // Check if it's a timeout/abort error 165 const isTimeoutError = error.name === 'AbortError' || 166 error.message.includes('timeout') || 167 error.message.includes('502'); 168 169 if (attempt < maxRetries && isTimeoutError) { 170 const retryDelay = attempt * 2000; // Progressive delay: 2s, 4s, 6s 171 console.log(`[generateAISummary] Timeout detected, retrying in ${retryDelay}ms (attempt ${attempt + 1}/${maxRetries})...`); 172 await new Promise(resolve => setTimeout(resolve, retryDelay)); 173 } else if (attempt >= maxRetries) { 174 // Final attempt failed 175 if (isTimeoutError) { 176 throw new Error(`AI summary generation timed out after ${attempt} attempts. The request is taking longer than expected due to high server load.`); 177 } else { 178 throw error; 179 } 180 } else { 181 // Non-timeout error on first attempt, don't retry 182 throw error; 183 } 184 } 185 } 186 } 187 188 /** 189 * Create and display the "Full Realtime Response" section 190 * @param {HTMLElement} chatContainer - Chat container element 191 * @returns {HTMLElement} - Response section element 192 */ 193 createFullRealtimeResponseSection(chatContainer) { 194 // Remove any existing full response section 195 const existingSection = document.querySelector('.full-realtime-response-section'); 196 if (existingSection) { 197 existingSection.remove(); 198 } 199 200 // Create the new section 201 const responseSection = document.createElement('div'); 202 responseSection.className = 'full-realtime-response-section'; 203 responseSection.style.background = '#f8fffe'; 204 responseSection.style.border = '2px solid #00a86b'; 205 responseSection.style.borderRadius = '12px'; 206 responseSection.style.margin = '24px 0 0 0'; 207 responseSection.style.padding = '20px 24px'; 208 responseSection.style.boxShadow = '0 4px 12px rgba(0,168,107,0.1)'; 209 responseSection.style.fontSize = '1em'; 210 responseSection.style.color = '#1a2630'; 211 212 // Create header 213 const headerDiv = document.createElement('div'); 214 headerDiv.className = 'full-response-header'; 215 headerDiv.style.fontWeight = 'bold'; 216 headerDiv.style.fontSize = '1.3em'; 217 headerDiv.style.marginBottom = '16px'; 218 headerDiv.style.color = '#00a86b'; 219 headerDiv.style.borderBottom = '2px solid #e0f2ed'; 220 headerDiv.style.paddingBottom = '12px'; 221 headerDiv.innerHTML = 'ð¤ Full Realtime Response'; 222 responseSection.appendChild(headerDiv); 223 224 // Create loading indicator 225 const loadingDiv = document.createElement('div'); 226 loadingDiv.className = 'ai-summary-loading'; 227 loadingDiv.style.display = 'flex'; 228 loadingDiv.style.alignItems = 'center'; 229 loadingDiv.style.color = '#666'; 230 loadingDiv.style.fontSize = '0.95em'; 231 loadingDiv.innerHTML = `
232 <div style="width: 20px; height: 20px; border: 3px solid #ddd; border-top: 3px solid #00a86b; border-radius: 50%; animation: spin 1s linear infinite; margin-right: 12px;"></div> 233 Analyzing Wikipedia articles and generating comprehensive response... 234 `; 235 responseSection.appendChild(loadingDiv); 236 237 // Add to chat container 238 chatContainer.appendChild(responseSection); 239 240 // Scroll to show the new section 241 responseSection.scrollIntoView({ behavior: 'smooth', block: 'nearest' }); 242 243 return responseSection; 244 } 245 246 /** 247 * Update the full response section with AI summary 248 * @param {HTMLElement} responseSection - Response section element 249 * @param {Object} summaryData - AI summary data 250 * @param {Array} wikipediaEntities - Wikipedia entities for linkification 251 */ 252 displayAISummary(responseSection, summaryData, wikipediaEntities = []) { 253 // Remove loading indicator 254 const loadingDiv = responseSection.querySelector('.ai-summary-loading'); 255 if (loadingDiv) { 256 loadingDiv.remove(); 257 } 258 259 // Create content div 260 const contentDiv = document.createElement('div'); 261 contentDiv.className = 'ai-summary-content'; 262 contentDiv.style.lineHeight = '1.6'; 263 contentDiv.style.fontSize = '0.95em'; 264 contentDiv.style.color = '#2c3e50'; 265 266 // Add metadata 267 const metaDiv = document.createElement('div'); 268 metaDiv.style.fontSize = '0.85em'; 269 metaDiv.style.color = '#666'; 270 metaDiv.style.marginBottom = '16px'; 271 metaDiv.style.fontStyle = 'italic'; 272 metaDiv.innerHTML = `Based on Real-Time analysis of ${summaryData.articlesProcessed.characterCount} bytes from ${summaryData.articlesProcessed.articleCount} Wikipedia articles for: "${summaryData.userQuestion}"`; 273 contentDiv.appendChild(metaDiv); 274 275 // Add the AI summary with linkification applied 276 const summaryDiv = document.createElement('div'); 277 summaryDiv.style.whiteSpace = 'pre-wrap'; 278 279 // Apply linkification if entities are available 280 let summaryHtml = summaryData.summary.replace(/\n/g, '<br>'); // Convert newlines to HTML breaks 281 if (wikipediaEntities && wikipediaEntities.length > 0) { 282 console.log(`[displayAISummary] Applying linkification to AI summary with ${wikipediaEntities.length} entities`); 283 summaryHtml = window.EntityProcessor.linkifyKnownEntitiesInHtml(summaryHtml, wikipediaEntities, null); 284 } else { 285 console.log(`[displayAISummary] No Wikipedia entities available for linkification`); 286 } 287 288 summaryDiv.innerHTML = summaryHtml; 289 contentDiv.appendChild(summaryDiv); 290 291 responseSection.appendChild(contentDiv); 292 } 293 294 /** 295 * Show error in the full response section 296 * @param {HTMLElement} responseSection - Response section element 297 * @param {Error} error - Error object 298 */ 299 showAISummaryError(responseSection, error) { 300 // Remove loading indicator 301 const loadingDiv = responseSection.querySelector('.ai-summary-loading'); 302 if (loadingDiv) { 303 loadingDiv.remove(); 304 } 305 306 // Create error div 307 const errorDiv = document.createElement('div'); 308 errorDiv.className = 'ai-summary-error'; 309 errorDiv.style.color = '#d32f2f'; 310 errorDiv.style.backgroundColor = '#ffebee'; 311 errorDiv.style.border = '1px solid #ffcdd2'; 312 errorDiv.style.borderRadius = '6px'; 313 errorDiv.style.padding = '12px'; 314 errorDiv.style.fontSize = '0.9em'; 315 316 // Check if it's a timeout error for better messaging 317 const isTimeoutError = error.message && error.message.includes('timed out'); 318 319 if (isTimeoutError) { 320 errorDiv.innerHTML = ` 321 <strong>â±ï¸ AI Summary Timeout</strong><br> 322 ${error.message}<br> 323 <small style="color: #666; margin-top: 8px; display: block;"> 324 ð¡ Tip: Try a more specific query or wait a moment before trying again. 325 </small> 326 `; 327 } else { 328 errorDiv.innerHTML = ` 329 <strong>â AI Summary Error</strong><br> 330 ${error.message} 331 `; 332 } 333 334 responseSection.appendChild(errorDiv); 335 } 336 337 /** 338 * Enhanced function to collect articles from entities and trigger AI summary 339 * @param {Array} entities - Wikipedia entities 340 * @param {string} userQuestion - User's original question 341 * @param {HTMLElement} chatContainer - Chat container element 342 * @param {Array} conversationHistory - Previous conversation context 343 */ 344 async collectArticlesAndGenerateAISummary(entities, userQuestion, chatContainer, conversationHistory = []) { 345 console.log(`[collectArticlesAndGenerateAISummary] Starting article collection for ${entities.length} entities`); 346 347 // Clear previous articles 348 this.collectedArticles.clear(); 349 this.currentUserQuestion = userQuestion; 350 351 // Create the full response section with loading state 352 const responseSection = this.createFullRealtimeResponseSection(chatContainer); 353 354 try { 355 // Sort entities by contextual relevance before collecting articles 356 // This ensures the most relevant entities get processed first within the 20k character limit 357 const sortedEntities = entities.sort((a, b) => { 358 const titleA = (a.page_title || a.title || '').toLowerCase(); 359 const titleB = (b.page_title || b.title || '').toLowerCase(); 360 361 let scoreA = 0; 362 let scoreB = 0; 363 364 // Check if entity is a primary entity (appears as main Wikipedia page) 365 // Primary entities should be prioritized as they contain the core information 366 const isPrimaryA = titleA === 'tom cruise' || titleA === 'mimi rogers' || 367 titleA === 'nicole kidman' || titleA === 'katie holmes'; 368 const isPrimaryB = titleB === 'tom cruise' || titleB === 'mimi rogers' || 369 titleB === 'nicole kidman' || titleB === 'katie holmes'; 370 371 if (isPrimaryA) scoreA += 1000; 372 if (isPrimaryB) scoreB += 1000; 373 374 // Boost score for person entities vs lists/filmographies 375 if (!titleA.includes('filmography') && !titleA.includes('list of') && !titleA.includes('awards')) { 376 scoreA += 500; 377 } 378 if (!titleB.includes('filmography') && !titleB.includes('list of') && !titleB.includes('awards')) { 379 scoreB += 500; 380 } 381 382 // Penalize random films unless they're directly relevant 383 if (titleA.includes('(film)') || titleA.includes('(movie)')) scoreA -= 200; 384 if (titleB.includes('(film)') || titleB.includes('(movie)')) scoreB -= 200; 385 386 return scoreB - scoreA; // Higher score first 387 }); 388 389 console.log(`[collectArticlesAndGenerateAISummary] Sorted ${sortedEntities.length} entities by relevance`); 390 if (sortedEntities.length > 0) { 391 console.log(`[collectArticlesAndGenerateAISummary] Top priority entities: ${sortedEntities.slice(0, 5).map(e =>
391 e.page_title || e.title).join(', ')}`); 392 } 393 394 // Collect full articles for each entity in priority order 395 const articlePromises = sortedEntities.map(async (entity) => { 396 const title = entity.page_title || entity.title; 397 if (title) { 398 const fullArticle = await this.fetchFullWikipediaArticle(title); 399 if (fullArticle) { 400 this.collectedArticles.set(title, fullArticle); 401 console.log(`[collectArticlesAndGenerateAISummary] Collected article for: ${title} (${fullArticle.length} chars)`); 402 } 403 } 404 }); 405 406 // Wait for all articles to be collected 407 await Promise.all(articlePromises); 408 409 console.log(`[collectArticlesAndGenerateAISummary] Collected ${this.collectedArticles.size} articles, generating AI summary`); 410 411 if (this.collectedArticles.size > 0) { 412 // Generate AI summary 413 const summaryData = await this.generateAISummary(userQuestion, this.collectedArticles, conversationHistory); 414 415 // Display the summary with linkification 416 this.displayAISummary(responseSection, summaryData, entities); 417 418 console.log(`[collectArticlesAndGenerateAISummary] Successfully generated and displayed AI summary`); 419 } else { 420 throw new Error('No Wikipedia articles could be collected for analysis'); 421 } 422 423 } catch (error) { 424 console.error(`[collectArticlesAndGenerateAISummary] Error:`, error); 425 this.showAISummaryError(responseSection, error); 426 } 427 } 428 429 /** 430 * Get collected articles 431 * @returns {Map} - Map of collected articles 432 */ 433 getCollectedArticles() { 434 return this.collectedArticles; 435 } 436 437 /** 438 * Get current user question 439 * @returns {string} - Current user question 440 */ 441 getCurrentUserQuestion() { 442 return this.currentUserQuestion; 443 } 444 445 /** 446 * Clear collected articles 447 */ 448 clearCollectedArticles() { 449 this.collectedArticles.clear(); 450 this.currentUserQuestion = ''; 451 } 452} 453 454// Export for use in other modules 455if (typeof module !== 'undefined' && module.exports) { 456 // Node.js environment 457 module.exports = { 458 WikipediaArticlesAggregator 459 }; 460} else { 461 // Browser environment - attach to window 462 window.WikipediaArticlesAggregator = WikipediaArticlesAggregator; 463 464 // Create global instance for backward compatibility 465 window.wikipediaAggregator = new WikipediaArticlesAggregator(); 466 467 // Export individual functions for backward compatibility 468 window.fetchFullWikipediaArticle = (title) => window.wikipediaAggregator.fetchFullWikipediaArticle(title); 469 window.generateAISummary = (userQuestion, articlesMap, conversationHistory) => 470 window.wikipediaAggregator.generateAISummary(userQuestion, articlesMap, conversationHistory); 471 window.createFullRealtimeResponseSection = (chatContainer) => 472 window.wikipediaAggregator.createFullRealtimeResponseSection(chatContainer); 473 window.displayAISummary = (responseSection, summaryData, wikipediaEntities) => 474 window.wikipediaAggregator.displayAISummary(responseSection, summaryData, wikipediaEntities); 475 window.showAISummaryError = (responseSection, error) => 476 window.wikipediaAggregator.showAISummaryError(responseSection, error); 477 window.collectArticlesAndGenerateAISummary = (entities, userQuestion, chatContainer, conversationHistory) => 478 window.wikipediaAggregator.collectArticlesAndGenerateAISummary(entities, userQuestion, chatContainer, conversationHistory); 479 480 // Maintain global variables for backward compatibility 481 window.collectedArticles = window.wikipediaAggregator.collectedArticles; 482 window.currentUserQuestion = ''; 483} 484 485console.log('Wikipedia Articles Aggregation Module - Loaded and Ready');
Line numbers count LF bytes from the start of the resource, as the search results do. Vendor segments are library code the classifier recognised; they are stored but not indexed. Bytes are shown as Latin1 characters, one per byte.