(file_path: str, extract_image: bool = False)
| 182 | |
| 183 | |
| 184 | def parse_tsv(file_path: str, extract_image: bool = False) -> List[dict]: |
| 185 | if extract_image: |
| 186 | raise ValueError('Currently, extracting images is not supported!') |
| 187 | |
| 188 | import pandas as pd |
| 189 | md_tables = [] |
| 190 | try: |
| 191 | df = pd.read_csv(file_path, sep='\t', encoding_errors='replace', on_bad_lines='skip') |
| 192 | except Exception as ex: |
| 193 | # Directly converted from Excel |
| 194 | logger.warning(ex) |
| 195 | return parse_excel(file_path, extract_image) |
| 196 | md_table = df_to_md(df) |
| 197 | md_tables.append(md_table) # There is only one table available |
| 198 | |
| 199 | return [{'page_num': i + 1, 'content': [{'table': md_tables[i]}]} for i in range(len(md_tables))] |
| 200 | |
| 201 | |
| 202 | def parse_html_bs(path: str, extract_image: bool = False): |
no test coverage detected