Parse web page content into a Document object. Args: content: URL encoded as bytes Returns: Document object containing the parsed markdown content
(self, content: bytes)
| 204 | return empty |
| 205 | |
| 206 | def parse_into_text(self, content: bytes) -> Document: |
| 207 | """Parse web page content into a Document object. |
| 208 | |
| 209 | Args: |
| 210 | content: URL encoded as bytes |
| 211 | |
| 212 | Returns: |
| 213 | Document object containing the parsed markdown content |
| 214 | """ |
| 215 | url = endecode.decode_bytes(content) |
| 216 | |
| 217 | logger.info(f"Scraping web page: {url}") |
| 218 | scrape_result = asyncio.run(self.scrape(url)) |
| 219 | if not scrape_result.html and not scrape_result.visible_text: |
| 220 | logger.error("Failed to scrape web page (no HTML or visible text)") |
| 221 | return Document(content=f"Error parsing web page: {url}") |
| 222 | |
| 223 | md_text = extract_markdown_from_html(scrape_result.html) |
| 224 | if not md_text: |
| 225 | md_text = build_visible_text_fallback( |
| 226 | scrape_result.visible_text, |
| 227 | scrape_result.page_title, |
| 228 | ) |
| 229 | if md_text: |
| 230 | logger.info( |
| 231 | "Trafilatura empty; using Playwright visible-text fallback (%d chars)", |
| 232 | len(md_text), |
| 233 | ) |
| 234 | |
| 235 | if not md_text: |
| 236 | logger.error("Failed to parse web page") |
| 237 | return Document(content=f"Error parsing web page: {url}") |
| 238 | |
| 239 | metadata = {} |
| 240 | title_match = re.search(r"^title:\s*(.+)", md_text, re.MULTILINE) |
| 241 | if title_match: |
| 242 | extracted_title = title_match.group(1).strip() |
| 243 | if extracted_title: |
| 244 | metadata["title"] = extracted_title |
| 245 | logger.info( |
| 246 | f"Extracted article title from trafilatura: {extracted_title}" |
| 247 | ) |
| 248 | elif scrape_result.page_title: |
| 249 | metadata["title"] = scrape_result.page_title.strip() |
| 250 | logger.info( |
| 251 | "Using page title from Playwright: %s", metadata["title"] |
| 252 | ) |
| 253 | else: |
| 254 | logger.info( |
| 255 | "No title found in trafilatura output, first 200 chars: %r", |
| 256 | md_text[:200], |
| 257 | ) |
| 258 | return Document(content=md_text, metadata=metadata) |
| 259 | |
| 260 | |
| 261 | class WebParser(PipelineParser): |
no test coverage detected