* Graph-derived prompt matching for the front-load hook's MEDIUM tier: * which indexed symbols do these prose words name? "state machine des * commandes" → `OrderStateMachine`, in any human language whose technical * nouns are Latin script — no keyword list involved. * * Precision com
(words: string[], limit: number = 6)
| 1800 | * returned symbol is guaranteed to exist right now. |
| 1801 | */ |
| 1802 | getSegmentMatches(words: string[], limit: number = 6): SegmentMatch[] { |
| 1803 | if (words.length === 0) return []; |
| 1804 | // Variant → original word (plural folding), for coverage accounting. |
| 1805 | const variantToWord = new Map<string, string>(); |
| 1806 | for (const word of words) { |
| 1807 | for (const variant of segmentLookupVariants(word)) { |
| 1808 | if (!variantToWord.has(variant)) variantToWord.set(variant, word); |
| 1809 | } |
| 1810 | } |
| 1811 | const variants = [...variantToWord.keys()]; |
| 1812 | |
| 1813 | // Tier A: co-occurrence. The SQL folds variants back to their original |
| 1814 | // word (#1146), so minWords=2 means two distinct PROMPT WORDS — a name |
| 1815 | // matching both `service` and `services` can't tie with (or crowd past |
| 1816 | // the LIMIT) a genuine two-word match. The JS re-check below recomputes |
| 1817 | // the fold from live segments as the honesty layer. |
| 1818 | const variantPairs = [...variantToWord.entries()].map(([segment, word]) => ({ segment, word })); |
| 1819 | const candidates: Array<{ name: string; matchedWords: Set<string> }> = []; |
| 1820 | for (const hit of this.queries.getSegmentCoOccurrence(variantPairs, 2, 24)) { |
| 1821 | const matched = this.wordsMatchingName(hit.name, variantToWord); |
| 1822 | if (matched.size >= 2) candidates.push({ name: hit.name, matchedWords: matched }); |
| 1823 | } |
| 1824 | |
| 1825 | // Tier B: single rare word. Only when co-occurrence found nothing — a |
| 1826 | // co-occurring name is categorically stronger evidence — and under |
| 1827 | // stricter rules, because one word is thin: the word must be ≥5 chars |
| 1828 | // (measured FPs: "this", "typo"); the segment must appear in AT LEAST TWO |
| 1829 | // names (a concept the codebase is about clusters across names — |
| 1830 | // CheckoutService/CheckoutController — while a prose coincidence is a |
| 1831 | // singleton: measured FP "deploy to PRODUCTION" → the one name |
| 1832 | // matchesNonProductionDir); and the candidate name must have ≥2 segments |
| 1833 | // (a bare common verb matching a bare function name — "write" → `write` — |
| 1834 | // is prose coincidence, not the user naming a symbol). |
| 1835 | if (candidates.length === 0) { |
| 1836 | const singleWordVariants = variants.filter((v) => variantToWord.get(v)!.length >= 5); |
| 1837 | const counts = this.queries.getSegmentNameCounts(singleWordVariants); |
| 1838 | const rare = [...counts.entries()] |
| 1839 | .filter(([, n]) => n >= 2 && n <= CodeGraph.SEGMENT_RARITY_CEILING) |
| 1840 | .sort((a, b) => a[1] - b[1]) |
| 1841 | .slice(0, 2); |
| 1842 | for (const [variant] of rare) { |
| 1843 | const word = variantToWord.get(variant)!; |
| 1844 | for (const name of this.queries.getNamesForSegment(variant, 12)) { |
| 1845 | if (splitIdentifierSegments(name).length < 2) continue; |
| 1846 | candidates.push({ name, matchedWords: new Set([word]) }); |
| 1847 | } |
| 1848 | } |
| 1849 | } |
| 1850 | |
| 1851 | // Verify against nodes (the honesty gate) and pick a representative |
| 1852 | // definition per name. A name whose only nodes are file/import kind has |
| 1853 | // no real definition to point at — surfacing the import statement instead |
| 1854 | // reads as a matched symbol but isn't one (#1144) — so it's skipped, the |
| 1855 | // same way an orphaned vocab row is. (Import names no longer enter the |
| 1856 | // vocab at write time, but rows written before that exclusion persist |
| 1857 | // until the next full index.) |
| 1858 | const out: SegmentMatch[] = []; |
| 1859 | const seen = new Set<string>(); |
no test coverage detected