(id, sentence, punctuations, stemmer, token_max_len=40)
| 127 | |
| 128 | |
| 129 | def transform_sentence_to_vector(id, sentence, punctuations, stemmer, token_max_len=40): |
| 130 | stopwords = { |
| 131 | "a", |
| 132 | "and", |
| 133 | "are", |
| 134 | "as", |
| 135 | "at", |
| 136 | "be", |
| 137 | "but", |
| 138 | "by", |
| 139 | "for", |
| 140 | "if", |
| 141 | "in", |
| 142 | "into", |
| 143 | "is", |
| 144 | "it", |
| 145 | "no", |
| 146 | "not", |
| 147 | "of", |
| 148 | "on", |
| 149 | "or", |
| 150 | "s", |
| 151 | "such", |
| 152 | "t", |
| 153 | "that", |
| 154 | "the", |
| 155 | "their", |
| 156 | "then", |
| 157 | "there", |
| 158 | "these", |
| 159 | "they", |
| 160 | "this", |
| 161 | "to", |
| 162 | "was", |
| 163 | "will", |
| 164 | "with", |
| 165 | "www", |
| 166 | } |
| 167 | cleaned = remove_non_alphanumeric(sentence) |
| 168 | tokens = SimpleTokenizer.tokenize(cleaned) |
| 169 | processed_tokens = [ |
| 170 | stemmer.stem_word(token.lower()) |
| 171 | for token in tokens |
| 172 | if token not in punctuations |
| 173 | and token.lower() not in stopwords |
| 174 | and len(token) <= token_max_len |
| 175 | ] |
| 176 | terms, length = construct_sparse_vector(processed_tokens) |
| 177 | |
| 178 | return { |
| 179 | "id": id, |
| 180 | "text": sentence, |
| 181 | "indices": [term[0] for term in terms], |
| 182 | "raw_term_frequencies": [term[1] for term in terms], |
| 183 | "length": length, |
| 184 | } |
| 185 | |
| 186 |
nothing calls this directly
no test coverage detected