MCPcopy Create free account
hub / github.com/nlweb-ai/NLWeb / LinkExtractor

Class LinkExtractor

AskAgent/python/misc/podcast_scraper.py:11–34  ·  view source on GitHub ↗

Source from the content-addressed store, hash-verified

9
10
11class LinkExtractor(HTMLParser):
12 def __init__(self):
13 super().__init__()
14 self.links = []
15 self.current_tag = None
16 self.current_attrs = {}
17 self.current_text = ""
18
19 def handle_starttag(self, tag, attrs):
20 self.current_tag = tag
21 self.current_attrs = dict(attrs)
22 if tag == 'a' and 'href' in self.current_attrs:
23 self.links.append({
24 'href': self.current_attrs['href'],
25 'text': '',
26 'attrs': self.current_attrs
27 })
28
29 def handle_data(self, data):
30 if self.current_tag == 'a' and self.links:
31 self.links[-1]['text'] += data.strip()
32
33 def handle_endtag(self, tag):
34 self.current_tag = None
35
36class NPRPodcastScraperComplete:
37 def __init__(self):

Callers 1

extract_linksMethod · 0.85

Calls

no outgoing calls

Tested by

no test coverage detected