Recursively extract all text content from an element. Skips text nodes that contain only whitespace (spaces, tabs, newlines), which typically represent XML formatting rather than document content. Args: elem: defusedxml.minidom.Element to extract text f
(self, elem)
| 181 | return matches[0] |
| 182 | |
| 183 | def _get_element_text(self, elem): |
| 184 | """ |
| 185 | Recursively extract all text content from an element. |
| 186 | |
| 187 | Skips text nodes that contain only whitespace (spaces, tabs, newlines), |
| 188 | which typically represent XML formatting rather than document content. |
| 189 | |
| 190 | Args: |
| 191 | elem: defusedxml.minidom.Element to extract text from |
| 192 | |
| 193 | Returns: |
| 194 | str: Concatenated text from all non-whitespace text nodes within the element |
| 195 | """ |
| 196 | text_parts = [] |
| 197 | for node in elem.childNodes: |
| 198 | if node.nodeType == node.TEXT_NODE: |
| 199 | # Skip whitespace-only text nodes (XML formatting) |
| 200 | if node.data.strip(): |
| 201 | text_parts.append(node.data) |
| 202 | elif node.nodeType == node.ELEMENT_NODE: |
| 203 | text_parts.append(self._get_element_text(node)) |
| 204 | return "".join(text_parts) |
| 205 | |
| 206 | def replace_node(self, elem, new_content): |
| 207 | """ |