MCPcopy Create free account
hub / github.com/Tencent/WeKnora / MHTMLParser

Class MHTMLParser

docreader/parser/mhtml_parser.py:32–323  ·  view source on GitHub ↗

Parser for MHTML web archives.

Source from the content-addressed store, hash-verified

30
31
32class MHTMLParser(BaseParser):
33 """Parser for MHTML web archives."""
34
35 def __init__(self, *args, extract_images: bool = True, **kwargs):
36 super().__init__(*args, **kwargs)
37 self.extract_images = extract_images
38
39 def parse_into_text(self, content: bytes) -> Document:
40 logger.info(
41 "Parsing MHTML file: %s, size: %d bytes", self.file_name, len(content)
42 )
43 msg = email.message_from_bytes(content)
44
45 html_parts = []
46 images: Dict[str, str] = {}
47 image_aliases: Dict[str, str] = {}
48 metadata: Dict[str, object] = {}
49
50 for part in msg.walk():
51 content_type = part.get_content_type()
52 location = part.get("Content-Location", "")
53
54 if content_type == "text/html":
55 payload = part.get_payload(decode=True)
56 if not payload:
57 continue
58 charset = part.get_content_charset() or "utf-8"
59 try:
60 html_text = payload.decode(charset, errors="ignore")
61 except LookupError:
62 html_text = payload.decode("utf-8", errors="ignore")
63 html_parts.append(
64 {
65 "content": html_text,
66 "location": location,
67 "size": len(html_text),
68 }
69 )
70 elif content_type.startswith("image/") and self.extract_images:
71 image_data = part.get_payload(decode=True)
72 if image_data:
73 image_path = self._image_path_for_part(part, content_type, images)
74 images[image_path] = base64.b64encode(image_data).decode("utf-8")
75 self._add_image_aliases(image_aliases, part, image_path)
76
77 main_html = self._select_main_html(html_parts)
78 if not main_html:
79 logger.warning("No HTML content found in MHTML file")
80 return Document(
81 content="", images=images, metadata={"source_format": "mhtml"}
82 )
83 html_content = main_html["content"]
84
85 try:
86 markdown_text = self._html_to_markdown(
87 html_content,
88 image_aliases=image_aliases,
89 base_location=main_html.get("location", ""),

Calls

no outgoing calls