| 1301 | """ |
| 1302 | |
| 1303 | def parse_into_text(self, content: bytes) -> Document: |
| 1304 | import pypdfium2 as pdfium |
| 1305 | |
| 1306 | images = {} |
| 1307 | markdown_lines = [] |
| 1308 | base_name = os.path.splitext(self.file_name or "document")[0] |
| 1309 | |
| 1310 | logger.info( |
| 1311 | "PDFScannedParser: Rendering PDF pages to JPEG images for %s", |
| 1312 | self.file_name, |
| 1313 | ) |
| 1314 | |
| 1315 | try: |
| 1316 | with parser_worker_limit("pdf_render", CONFIG.pdf_render_max_workers): |
| 1317 | pdf = pdfium.PdfDocument(content) |
| 1318 | try: |
| 1319 | page_count = len(pdf) |
| 1320 | scale = max(1, CONFIG.pdf_render_dpi) / 72 |
| 1321 | quality = _normalize_image_quality(CONFIG.pdf_jpeg_quality) |
| 1322 | |
| 1323 | rendered = _render_scanned_pages( |
| 1324 | pdf, |
| 1325 | content, |
| 1326 | list(range(page_count)), |
| 1327 | scale, |
| 1328 | quality, |
| 1329 | CONFIG.pdf_render_max_edge, |
| 1330 | ) |
| 1331 | finally: |
| 1332 | _close_pdfium_resource(pdf) |
| 1333 | |
| 1334 | for i in range(page_count): |
| 1335 | page_filename = f"{base_name}_page_{i+1}.jpg" |
| 1336 | ref_path = f"images/{page_filename}" |
| 1337 | markdown_lines.append(f"") |
| 1338 | images[ref_path] = base64.b64encode(rendered[i]).decode("utf-8") |
| 1339 | |
| 1340 | text = "\n\n".join(markdown_lines) |
| 1341 | return Document( |
| 1342 | content=text, |
| 1343 | images=images, |
| 1344 | metadata={ |
| 1345 | "image_source_type": "scanned_pdf", |
| 1346 | "page_count": page_count, |
| 1347 | }, |
| 1348 | ) |
| 1349 | except Exception as e: |
| 1350 | logger.exception("PDFScannedParser failed to parse PDF: %s", e) |
| 1351 | raise e |
| 1352 | |
| 1353 | |
| 1354 | class PDFParser(BaseParser): |