Parser for MHTML web archives.
| 30 | |
| 31 | |
| 32 | class MHTMLParser(BaseParser): |
| 33 | """Parser for MHTML web archives.""" |
| 34 | |
| 35 | def __init__(self, *args, extract_images: bool = True, **kwargs): |
| 36 | super().__init__(*args, **kwargs) |
| 37 | self.extract_images = extract_images |
| 38 | |
| 39 | def parse_into_text(self, content: bytes) -> Document: |
| 40 | logger.info( |
| 41 | "Parsing MHTML file: %s, size: %d bytes", self.file_name, len(content) |
| 42 | ) |
| 43 | msg = email.message_from_bytes(content) |
| 44 | |
| 45 | html_parts = [] |
| 46 | images: Dict[str, str] = {} |
| 47 | image_aliases: Dict[str, str] = {} |
| 48 | metadata: Dict[str, object] = {} |
| 49 | |
| 50 | for part in msg.walk(): |
| 51 | content_type = part.get_content_type() |
| 52 | location = part.get("Content-Location", "") |
| 53 | |
| 54 | if content_type == "text/html": |
| 55 | payload = part.get_payload(decode=True) |
| 56 | if not payload: |
| 57 | continue |
| 58 | charset = part.get_content_charset() or "utf-8" |
| 59 | try: |
| 60 | html_text = payload.decode(charset, errors="ignore") |
| 61 | except LookupError: |
| 62 | html_text = payload.decode("utf-8", errors="ignore") |
| 63 | html_parts.append( |
| 64 | { |
| 65 | "content": html_text, |
| 66 | "location": location, |
| 67 | "size": len(html_text), |
| 68 | } |
| 69 | ) |
| 70 | elif content_type.startswith("image/") and self.extract_images: |
| 71 | image_data = part.get_payload(decode=True) |
| 72 | if image_data: |
| 73 | image_path = self._image_path_for_part(part, content_type, images) |
| 74 | images[image_path] = base64.b64encode(image_data).decode("utf-8") |
| 75 | self._add_image_aliases(image_aliases, part, image_path) |
| 76 | |
| 77 | main_html = self._select_main_html(html_parts) |
| 78 | if not main_html: |
| 79 | logger.warning("No HTML content found in MHTML file") |
| 80 | return Document( |
| 81 | content="", images=images, metadata={"source_format": "mhtml"} |
| 82 | ) |
| 83 | html_content = main_html["content"] |
| 84 | |
| 85 | try: |
| 86 | markdown_text = self._html_to_markdown( |
| 87 | html_content, |
| 88 | image_aliases=image_aliases, |
| 89 | base_location=main_html.get("location", ""), |
no outgoing calls