MCPcopy Create free account
hub / github.com/Tencent/WeKnora / DocParser

Class DocParser

docreader/parser/doc_parser.py:98–343  ·  view source on GitHub ↗

DOC document parser

Source from the content-addressed store, hash-verified

96
97
98class DocParser(Docx2Parser):
99 """DOC document parser"""
100
101 def __init__(self, *args, **kwargs):
102 """Initialize DOC parser with sandbox executor"""
103 super().__init__(*args, **kwargs)
104 self.sandbox_executor = SandboxExecutor()
105
106 def parse_into_text(self, content: bytes) -> Document:
107 logger.info(f"Parsing DOC document, content size: {len(content)} bytes")
108
109 handle_chain = [
110 # 1. Try to convert to docx format to extract images
111 self._parse_with_docx,
112 # 2. If image extraction is not needed or conversion failed,
113 # try using antiword to extract text
114 self._parse_with_antiword,
115 # 3. If antiword extraction fails, use textract
116 # NOTE: _parse_with_textract is disabled due to SSRF vulnerability
117 # self._parse_with_textract,
118 ]
119
120 # Save byte content as a temporary file
121 with TempFileContext(content, ".doc") as temp_file_path:
122 for handle in handle_chain:
123 try:
124 document = handle(temp_file_path)
125 if document:
126 return document
127 except Exception as e:
128 logger.warning(f"Failed to parse DOC with {handle.__name__} {e}")
129
130 return Document(content="")
131
132 def _parse_with_docx(self, temp_file_path: str) -> Document:
133 logger.info("Multimodal enabled, attempting to extract images from DOC")
134
135 docx_content = self._try_convert_doc_to_docx(temp_file_path)
136 if not docx_content:
137 raise RuntimeError("Failed to convert DOC to DOCX")
138
139 logger.info("Successfully converted DOC to DOCX, using DocxParser")
140 # Use existing DocxParser to parse the converted docx
141 document = super(Docx2Parser, self).parse_into_text(docx_content)
142 logger.info(f"Extracted {len(document.content)} characters using DocxParser")
143 return document
144
145 def _parse_with_antiword(self, temp_file_path: str) -> Document:
146 logger.info("Attempting to parse DOC file with antiword")
147
148 # Check if antiword is installed
149 antiword_path = self._try_find_antiword()
150 if not antiword_path:
151 raise RuntimeError("antiword not found in PATH")
152
153 # Use antiword to extract text directly in sandbox
154 cmd = [antiword_path, temp_file_path]
155 logger.info("Executing antiword in sandbox with proxy configuration")

Callers 1

doc_parser.pyFile · 0.85

Calls

no outgoing calls

Tested by

no test coverage detected