(attribute, column_threshold=.5, entity_threshold=.5)
| 20 | #Check each cell to see if it is text. Then if enough number of cells are |
| 21 | #text, the column is considered as a text column. |
| 22 | def getColumnType(attribute, column_threshold=.5, entity_threshold=.5): |
| 23 | attribute = [item for item in attribute if str(item) != "nan"] |
| 24 | if len(attribute) == 0: |
| 25 | return 0 |
| 26 | strAttribute = [item for item in attribute if type(item) == str] |
| 27 | strAtt = [item for item in strAttribute if not item.isdigit()] |
| 28 | for i in range(len(strAtt)-1, -1, -1): |
| 29 | entity = strAtt[i] |
| 30 | num_count = 0 |
| 31 | for char in entity: |
| 32 | if char.isdigit(): |
| 33 | num_count += 1 |
| 34 | if num_count/len(entity) > entity_threshold: |
| 35 | del strAtt[i] |
| 36 | if len(strAtt)/len(attribute) > column_threshold: |
| 37 | return 1 |
| 38 | else: |
| 39 | return 0 |
| 40 | |
| 41 | #removes punctuations and whitespaces from string. The same preprocessing |
| 42 | #is done in yago label file |
nothing calls this directly
no outgoing calls
no test coverage detected