1/** 2 * PDF to PPTX Worker (via Pyodide + PyMuPDF + python-pptx) 3 * 4 * Converts PDF pages to images and creates a PPTX with each page as a slide. 5 */ 6 7import { loadPyodide } from '/pymupdf-wasm/pyodide.js'; 8 9let pyodide = null; 10let initPromise = null; 11 12async function init() { 13 if (pyodide) return pyodide; 14 15 self.postMessage({ type: 'status', message: 'Loading Python environment...' }); 16 17 // Initialize Pyodide 18 pyodide = await loadPyodide({ 19 indexURL: '/pymupdf-wasm/', 20 fullStdLib: false 21 }); 22 23 self.postMessage({ type: 'status', message: 'Installing dependencies...' }); 24 25 const install = async (url) => { 26 await pyodide.loadPackage(url); 27 }; 28 29 const basePath = '/pymupdf-wasm/'; 30 31 // Mock missing non-critical dependencies 32 pyodide.runPython(` 33 import sys 34 from types import ModuleType 35 36 # Mock tqdm (used for progress bars) 37 tqdm_mod = ModuleType("tqdm") 38 def tqdm(iterable=None, *args, **kwargs): 39 return iterable if iterable else [] 40 tqdm_mod.tqdm = tqdm 41 sys.modules["tqdm"] = tqdm_mod 42 `); 43 44 // Install required packages 45 // Install Pillow (local wheel) 46 self.postMessage({ type: 'status', message: 'Installing Pillow...' }); 47 await install(basePath + 'pillow-11.2.1-cp313-cp313-pyodide_2025_0_wasm32.whl'); 48 49 await install(basePath + 'typing_extensions-4.12.2-py3-none-any.whl'); 50 await install(basePath + 'lxml-5.4.0-cp313-cp313-pyodide_2025_0_wasm32.whl'); 51 await install(basePath + 'pymupdf-1.26.3-cp313-none-pyodide_2025_0_wasm32.whl'); 52 53 // Install python-pptx and its dependency 54 self.postMessage({ type: 'status', message: 'Installing python-pptx...' }); 55 await install(basePath + 'python_pptx-1.0.2-py3-none-any.whl'); 56 57 // Define the python processing script 58 self.postMessage({ type: 'status', message: 'Initializing converter script...' }); 59 60 pyodide.runPython(` 61import os 62import io 63import fitz # PyMuPDF 64from pptx import Presentation 65from pptx.util import Inches, Emu 66 67def convert_pdf_to_pptx(input_obj, dpi=150): 68 """ 69 Convert PDF to PPTX by rendering each page as an image and adding to slides. 70 """ 71 # Convert JsProxy to bytes 72 if hasattr(input_obj, "to_py"): 73 input_bytes = input_obj.to_py() 74 else: 75 input_bytes = input_obj 76 77 # Open PDF 78 pdf_doc = fitz.open(stream=input_bytes, filetype="pdf") 79 80 # Create presentation 81 prs = Presentation() 82 83 # Process each page 84 for page_num in range(len(pdf_doc)): 85 page = pdf_doc[page_num] 86 87 # Get page dimensions 88 rect = page.rect 89 width_pt = rect.width 90 height_pt = rect.height 91 92 # Set slide dimensions to match PDF page aspect ratio 93 # Standard slide is 10x7.5 inches, but we'll match PDF aspect ratio 94 aspect_ratio = width_pt / height_pt 95 96 # Use standard width of 10 inches 97 slide_width = Inches(10) 98 slide_height = Emu(slide_width / aspect_ratio) 99 100 prs.slide_width = slide_width 101 prs.slide_height = slide_height 102 103 # Render page to image 104 mat = fitz.Matrix(dpi / 72, dpi / 72) # Scale for DPI 105 pix = page.get_pixmap(matrix=mat, alpha=False) 106 img_bytes = pix.tobytes("png") 107 108 # Add blank slide 109 blank_layout = prs.slide_layouts[6] # Blank layout 110 slide = prs.slides.add_slide(blank_layout) 111 112 # Save image temporarily 113 img_path = f"temp_page_{page_num}.png" 114 with open(img_path, "wb") as f: 115 f.write(img_bytes) 116 117 # Add image to slide (full slide size) 118 slide.shapes.add_picture( 119 img_path, 120 left=0, 121 top=0, 122 width=slide_width, 123 height=slide_height 124 ) 125 126 # Cleanup temp image 127 os.remove(img_path) 128 129 pdf_doc.close() 130 131 # Save presentation to bytes 132 output = io.BytesIO() 133 prs.save(output) 134 output.seek(0) 135 pptx_bytes = output.read() 136 137 return pptx_bytes 138 `); 139 140 return pyodide; 141} 142 143self.onmessage = async (event) => { 144 const { type, id, data } = event.data; 145 146 try { 147 if (type === 'init') { 148 if (!initPromise) initPromise = init(); 149 await initPromise; 150 self.postMessage({ id, type: 'init-complete' }); 151 return; 152 } 153 154 if (type === 'convert') { 155 if (!pyodide) { 156 if (!initPromise) initPromise = init(); 157 await initPromise; 158 } 159 160 const { file, dpi = 150 } = data; 161 const arrayBuffer = await file.arrayBuffer(); 162 const inputBytes = new Uint8Array(arrayBuffer); 163 164 self.postMessage({ type: 'status', message: 'Convert
164ing PDF pages to slides...' }); 165 166 // Call Python function 167 const convertFunc = pyodide.globals.get('convert_pdf_to_pptx'); 168 const resultProxy = convertFunc(inputBytes, dpi); 169 const resultBytes = resultProxy.toJs(); 170 resultProxy.destroy(); 171 172 const resultBlob = new Blob([resultBytes], { 173 type: 'application/vnd.openxmlformats-officedocument.presentationml.presentation' 174 }); 175 176 self.postMessage({ 177 id, 178 type: 'convert-complete', 179 result: resultBlob 180 }); 181 } 182 183 } catch (error) { 184 console.error('Worker error:', error); 185 self.postMessage({ 186 id, 187 type: 'error', 188 error: error.message || String(error) 189 }); 190 } 191};
Line numbers count LF bytes from the start of the resource, as the search results do. Vendor segments are library code the classifier recognised; they are stored but not indexed. Bytes are shown as Latin1 characters, one per byte.