| 9 | |
| 10 | |
| 11 | class LinkExtractor(HTMLParser): |
| 12 | def __init__(self): |
| 13 | super().__init__() |
| 14 | self.links = [] |
| 15 | self.current_tag = None |
| 16 | self.current_attrs = {} |
| 17 | self.current_text = "" |
| 18 | |
| 19 | def handle_starttag(self, tag, attrs): |
| 20 | self.current_tag = tag |
| 21 | self.current_attrs = dict(attrs) |
| 22 | if tag == 'a' and 'href' in self.current_attrs: |
| 23 | self.links.append({ |
| 24 | 'href': self.current_attrs['href'], |
| 25 | 'text': '', |
| 26 | 'attrs': self.current_attrs |
| 27 | }) |
| 28 | |
| 29 | def handle_data(self, data): |
| 30 | if self.current_tag == 'a' and self.links: |
| 31 | self.links[-1]['text'] += data.strip() |
| 32 | |
| 33 | def handle_endtag(self, tag): |
| 34 | self.current_tag = None |
| 35 | |
| 36 | class NPRPodcastScraperComplete: |
| 37 | def __init__(self): |