解析单个文件
(self, file_path: str)
| 205 | return f"[解析失败: {e}]" |
| 206 | |
| 207 | def _parse_file(self, file_path: str) -> str: |
| 208 | """解析单个文件""" |
| 209 | path = Path(file_path) |
| 210 | |
| 211 | if not path.exists(): |
| 212 | logger.warning(f"DocumentNode: 文件不存在: {file_path}") |
| 213 | return "" |
| 214 | |
| 215 | try: |
| 216 | md = _get_markitdown() |
| 217 | result = md.convert(str(path)) |
| 218 | |
| 219 | content = result.text_content if hasattr(result, 'text_content') else str(result) |
| 220 | |
| 221 | # 添加元数据 |
| 222 | if self.include_metadata: |
| 223 | metadata = [ |
| 224 | f"**文件名**: {path.name}", |
| 225 | f"**大小**: {path.stat().st_size} bytes", |
| 226 | ] |
| 227 | if hasattr(result, 'title') and result.title: |
| 228 | metadata.append(f"**标题**: {result.title}") |
| 229 | content = "\n".join(metadata) + "\n\n" + content |
| 230 | |
| 231 | # 限制长度 |
| 232 | if self.max_length > 0 and len(content) > self.max_length: |
| 233 | content = content[:self.max_length] + "\n\n... [内容已截断]" |
| 234 | |
| 235 | logger.info(f"DocumentNode: 解析完成 {path.name} ({len(content)} 字符)") |
| 236 | return content |
| 237 | |
| 238 | except Exception as e: |
| 239 | logger.error(f"DocumentNode: 解析失败 {file_path}: {e}") |
| 240 | return f"[解析失败: {e}]" |
| 241 | |
| 242 | |
| 243 | class DocumentExtractNode(DocumentNode): |
no test coverage detected