Tokenize text for the FTS inverted index. Extracts lowercase alphanumeric words of length >= 3, filtering out stop words (common English + programming terms) that match too many nodes to be useful for ranking.
(text: &str)
| 794 | /// stop words (common English + programming terms) that match too many |
| 795 | /// nodes to be useful for ranking. |
| 796 | pub fn tokenize_for_fts(text: &str) -> Vec<String> { |
| 797 | text.split(|c: char| !c.is_alphanumeric() && c != '_') |
| 798 | .filter(|w| w.len() >= 3) |
| 799 | .map(|w| w.to_lowercase()) |
| 800 | .filter(|w| !FTS_STOP_WORDS.contains(&w.as_str())) |
| 801 | .collect() |
| 802 | } |
| 803 | |
| 804 | /// Encode an embedding key as "path\0chunk_idx". |
| 805 | #[inline] |