MCPcopy Create free account
hub / github.com/colbymchenry/codegraph / scorePathRelevance

Function scorePathRelevance

src/search/query-utils.ts:221–285  ·  view source on GitHub ↗
(
  filePath: string,
  query: string,
  projectNameTokens?: Set<string>,
  isDeprioritized?: boolean,
)

Source from the content-addressed store, hash-verified

219 * Higher score = more relevant path
220 */
221export function scorePathRelevance(
222 filePath: string,
223 query: string,
224 projectNameTokens?: Set<string>,
225 isDeprioritized?: boolean,
226): number {
227 const pathLower = filePath.toLowerCase();
228 const fileName = path.basename(filePath).toLowerCase();
229 const dirName = path.dirname(filePath).toLowerCase();
230 let score = 0;
231
232 // Score per original query WORD, not per sub-token. A single PascalCase word
233 // splits into many sub-tokens (a project name "SuperBizAgent" →
234 // superbizagent / super / biz / agent) that all match the SAME path segment,
235 // so summing per sub-token boosted that path 4× for one concept — enough to
236 // bury the rest of the query's stack (#720). A word matches a path level if
237 // ANY of its sub-tokens do, and counts ONCE; distinct words still each add.
238 // Split the ORIGINAL-case query into words; extractSearchTerms does the
239 // camelCase/snake split per word (so `getUserName` still matches a
240 // `get_user_name` path) — we just attribute each word's matches once.
241 const allWords = query.split(/\s+/).filter((w) => w.length > 0);
242 if (allWords.length === 0) return 0;
243
244 // A query word that just names the PROJECT (its go.mod / package.json / repo
245 // name) carries no discriminative path signal — drop it so the rest of the
246 // query decides the ranking, instead of every file under a `<ProjectName>…/`
247 // tree winning on the project name alone (#720). Only when OTHER words remain,
248 // so a bare project-name query still scores on its path.
249 const words =
250 projectNameTokens && projectNameTokens.size > 0
251 ? allWords.filter((w) => !projectNameTokens.has(normalizeNameToken(w)))
252 : allWords;
253 const scored = words.length > 0 ? words : allWords;
254
255 for (const word of scored) {
256 // Use base terms only — stem variants inflate path scores by generating
257 // many near-duplicate terms that all match the same path segments.
258 const subtokens = extractSearchTerms(word, { stems: false });
259 if (subtokens.length === 0) continue;
260 // Exact filename match (strongest)
261 if (subtokens.some((t) => fileName.includes(t))) score += 10;
262 // Directory match
263 if (subtokens.some((t) => dirName.includes(t))) score += 5;
264 // General path match
265 else if (subtokens.some((t) => pathLower.includes(t))) score += 3;
266 }
267
268 // Deprioritize test files unless the query is explicitly about tests, and
269 // apply the same -15 to a path the project declared peripheral (#982).
270 //
271 // Two deliberate asymmetries, both pinned by tests:
272 // - the built-in test/fixture penalty is waived for a test-y query, because
273 // the tool inferred that classification; a `deprioritize` pattern is a
274 // standing statement by the project, so it is NOT waived. The name-bonus
275 // damping at the call site is what keeps such a tree findable.
276 // - a path that is both is docked ONCE, not twice.
277 const queryLower = query.toLowerCase();
278 const isTestQuery = queryLower.includes('test') || queryLower.includes('spec');

Callers 4

findRelevantContextMethod · 0.90
searchNodesMethod · 0.90

Calls 4

normalizeNameTokenFunction · 0.85
extractSearchTermsFunction · 0.85
hasMethod · 0.80
isTestFileFunction · 0.70

Tested by

no test coverage detected