MCPcopy Create free account
hub / github.com/Tencent/WeKnora / parse_into_text

Method parse_into_text

docreader/parser/pdf_parser.py:1303–1351  ·  view source on GitHub ↗
(self, content: bytes)

Source from the content-addressed store, hash-verified

1301 """
1302
1303 def parse_into_text(self, content: bytes) -> Document:
1304 import pypdfium2 as pdfium
1305
1306 images = {}
1307 markdown_lines = []
1308 base_name = os.path.splitext(self.file_name or "document")[0]
1309
1310 logger.info(
1311 "PDFScannedParser: Rendering PDF pages to JPEG images for %s",
1312 self.file_name,
1313 )
1314
1315 try:
1316 with parser_worker_limit("pdf_render", CONFIG.pdf_render_max_workers):
1317 pdf = pdfium.PdfDocument(content)
1318 try:
1319 page_count = len(pdf)
1320 scale = max(1, CONFIG.pdf_render_dpi) / 72
1321 quality = _normalize_image_quality(CONFIG.pdf_jpeg_quality)
1322
1323 rendered = _render_scanned_pages(
1324 pdf,
1325 content,
1326 list(range(page_count)),
1327 scale,
1328 quality,
1329 CONFIG.pdf_render_max_edge,
1330 )
1331 finally:
1332 _close_pdfium_resource(pdf)
1333
1334 for i in range(page_count):
1335 page_filename = f"{base_name}_page_{i+1}.jpg"
1336 ref_path = f"images/{page_filename}"
1337 markdown_lines.append(f"![{page_filename}]({ref_path})")
1338 images[ref_path] = base64.b64encode(rendered[i]).decode("utf-8")
1339
1340 text = "\n\n".join(markdown_lines)
1341 return Document(
1342 content=text,
1343 images=images,
1344 metadata={
1345 "image_source_type": "scanned_pdf",
1346 "page_count": page_count,
1347 },
1348 )
1349 except Exception as e:
1350 logger.exception("PDFScannedParser failed to parse PDF: %s", e)
1351 raise e
1352
1353
1354class PDFParser(BaseParser):

Calls 6

parser_worker_limitFunction · 0.90
DocumentClass · 0.90
maxFunction · 0.85
_normalize_image_qualityFunction · 0.85
_render_scanned_pagesFunction · 0.85
_close_pdfium_resourceFunction · 0.85