From 081056e6e54053d523469f5a70b22df856df1b94 Mon Sep 17 00:00:00 2001 From: Bohdan Triapitsyn Date: Mon, 24 Aug 2026 14:54:14 +0300 Subject: [PATCH] feat(search): canonical rankByQuery matcher for searchable dropdowns One matcher for every dropdown filter: multi-token queries match in any order, matching is punctuation-insensitive, single-token queries tolerate typos, and results come back relevance-ordered (prefix, then word-boundary, then substring, original order on ties). matchesRankQuery is the boolean companion for lists that keep their own grouping. --- .../ui/src/lib/search/fuzzySearch.test.ts | 68 +++++++++ packages/ui/src/lib/search/fuzzySearch.ts | 130 ++++++++++++++++++ 2 files changed, 198 insertions(+) create mode 100644 packages/ui/src/lib/search/fuzzySearch.test.ts diff --git a/packages/ui/src/lib/search/fuzzySearch.test.ts b/packages/ui/src/lib/search/fuzzySearch.test.ts new file mode 100644 index 00000000..82a402bf --- /dev/null +++ b/packages/ui/src/lib/search/fuzzySearch.test.ts @@ -0,0 +1,68 @@ +import { describe, expect, test } from 'bun:test'; + +import { matchesRankQuery, rankByQuery } from './fuzzySearch'; + +const rank = (items: string[], query: string) => rankByQuery(items, query, (item) => [item]); + +describe('rankByQuery', () => { + test('orders word-boundary matches above mid-word matches, earlier positions first', () => { + const items = ['prefixed-thing', 'workspace-fix', 'feat/fix-scroll']; + expect(rank(items, 'fix')).toEqual(['feat/fix-scroll', 'workspace-fix', 'prefixed-thing']); + }); + + test('exact prefix comes first, ties keep original order', () => { + const items = ['main', 'feat/main-menu', 'maintenance', 'release/main']; + const ranked = rank(items, 'main'); + expect(ranked[0]).toBe('main'); + expect(ranked[1]).toBe('maintenance'); + expect(ranked.slice(2)).toEqual(['feat/main-menu', 'release/main']); + }); + + test('multi-token queries match in any order and all tokens are required', () => { + const items = ['feat/scroll-anchored-chat', 'fix/chat-header', 'feat/scroll-perf']; + expect(rank(items, 'chat scroll')).toEqual(['feat/scroll-anchored-chat']); + }); + + test('punctuation-insensitive compact matching finds joined words', () => { + const items = ['gpt-4o-mini', 'claude-sonnet-5']; + expect(rank(items, 'gpt4o')).toEqual(['gpt-4o-mini']); + expect(rank(items, 'sonnet5')).toEqual(['claude-sonnet-5']); + }); + + test('single-token queries tolerate typos via fuzzy fallback', () => { + const items = ['workspace-rail-layout', 'unrelated']; + expect(rank(items, 'worskpace')).toEqual(['workspace-rail-layout']); + }); + + test('fuzzy fallback can be disabled', () => { + const items = ['workspace-rail-layout']; + expect(rankByQuery(items, 'worskpace', (item) => [item], { fuzzy: false })).toEqual([]); + }); + + test('earlier fields outrank later fields', () => { + const items = [ + { name: 'docs', path: '/repo/build-agent' }, + { name: 'build-agent', path: '/repo/build-agent' }, + ]; + const ranked = rankByQuery(items, 'build', (item) => [item.name, item.path]); + expect(ranked[0].name).toBe('build-agent'); + expect(ranked).toHaveLength(2); + }); + + test('empty query returns items unchanged within the limit', () => { + expect(rank(['b', 'a'], ' ')).toEqual(['b', 'a']); + expect(rankByQuery(['a', 'b', 'c'], '', (item) => [item], { limit: 2 })).toEqual(['a', 'b']); + }); +}); + +describe('matchesRankQuery', () => { + test('requires every token across the fields', () => { + expect(matchesRankQuery(['GLM-5.3', 'Zhipu'], 'zhipu glm')).toBe(true); + expect(matchesRankQuery(['GLM-5.3', 'Zhipu'], 'zhipu gpt')).toBe(false); + }); + + test('is punctuation-insensitive and skips empty fields', () => { + expect(matchesRankQuery([null, 'claude-sonnet-5', undefined], 'sonnet5')).toBe(true); + expect(matchesRankQuery([''], 'a')).toBe(false); + }); +}); diff --git a/packages/ui/src/lib/search/fuzzySearch.ts b/packages/ui/src/lib/search/fuzzySearch.ts index 99cd7446..6d809201 100644 --- a/packages/ui/src/lib/search/fuzzySearch.ts +++ b/packages/ui/src/lib/search/fuzzySearch.ts @@ -141,6 +141,136 @@ export function scoreByFuzzyQuery( return scored.slice(0, limit); } +const RANK_TOKEN_MISS = Number.POSITIVE_INFINITY; + +const tokenizeRankQuery = (query: string): string[] => + query.trim().toLowerCase().split(/\s+/).filter(Boolean); + +const compactText = (value: string): string => value.replace(/[^a-z0-9]+/g, ''); + +type RankFields = { fields: string[]; compact: string[] }; + +const buildRankFields = (texts: ReadonlyArray): RankFields => { + const fields: string[] = []; + const compact: string[] = []; + for (const text of texts) { + if (!text) continue; + const lower = text.toLowerCase(); + fields.push(lower); + compact.push(compactText(lower)); + } + return { fields, compact }; +}; + +/** + * Score one query token against an item's fields. Lower is better: + * field prefix < word-boundary substring < mid-word substring < + * punctuation-insensitive ("compact") substring. Earlier fields win ties, so + * callers should order `getTexts` by importance (name before path/description). + */ +const scoreRankToken = (token: string, { fields, compact }: RankFields): number => { + let best = RANK_TOKEN_MISS; + for (let fieldIndex = 0; fieldIndex < fields.length; fieldIndex++) { + const field = fields[fieldIndex]; + const fieldPenalty = fieldIndex * 0.01; + const idx = field.indexOf(token); + let score = RANK_TOKEN_MISS; + if (idx === 0) { + score = fieldPenalty; + } else if (idx > 0) { + const boundary = !/[a-z0-9]/.test(field[idx - 1]); + score = (boundary ? 0.1 : 0.2) + idx / 1000 + fieldPenalty; + } else { + const compactIdx = compact[fieldIndex].indexOf(compactText(token)); + if (compactIdx >= 0 && token.length > 1) { + score = 0.4 + compactIdx / 1000 + fieldPenalty; + } + } + if (score < best) best = score; + } + return best; +}; + +export interface RankByQueryOptions { + limit?: number; + /** Typo-tolerant Fuse fallback for single-token queries (default true). */ + fuzzy?: boolean; +} + +/** + * The canonical dropdown matcher: every whitespace-separated query token must + * match somewhere in the item's fields (any order, punctuation-insensitive), + * and results come back ordered by relevance — exact/prefix matches first, + * then word-boundary and substring matches, original order breaking ties. + * Single-token queries additionally fall back to typo-tolerant fuzzy matching. + * + * Use this for every searchable dropdown (projects, agents, branches, models) + * instead of ad hoc `toLowerCase().includes` filters, so matching quality and + * ordering stay consistent across pickers. + */ +export function rankByQuery( + items: readonly T[], + query: string, + getTexts: (item: T) => ReadonlyArray, + options?: RankByQueryOptions, +): T[] { + const tokens = tokenizeRankQuery(query); + const limit = options?.limit ?? items.length; + if (tokens.length === 0) return items.slice(0, limit); + + const scored: { item: T; score: number; order: number }[] = []; + const missed: { item: T; joined: string; order: number }[] = []; + + for (let order = 0; order < items.length; order++) { + const item = items[order]; + const rankFields = buildRankFields(getTexts(item)); + let total = 0; + for (const token of tokens) { + const tokenScore = scoreRankToken(token, rankFields); + if (tokenScore === RANK_TOKEN_MISS) { + total = RANK_TOKEN_MISS; + break; + } + total += tokenScore; + } + if (total === RANK_TOKEN_MISS) { + missed.push({ item, joined: rankFields.fields.join(' '), order }); + } else { + scored.push({ item, score: total, order }); + } + } + + const fuzzyEnabled = options?.fuzzy ?? true; + if (fuzzyEnabled && tokens.length === 1 && tokens[0].length >= 3 && missed.length > 0) { + const fuse = new Fuse( + missed.map((entry) => entry.joined), + { threshold: 0.35, ignoreLocation: true, distance: 100, includeScore: true, minMatchCharLength: 2 }, + ); + for (const result of fuse.search(tokens[0])) { + const entry = missed[result.refIndex]; + scored.push({ item: entry.item, score: 1 + (result.score ?? 1), order: entry.order }); + } + } + + scored.sort((a, b) => (a.score - b.score) || (a.order - b.order)); + return scored.slice(0, limit).map((entry) => entry.item); +} + +/** + * Boolean companion to `rankByQuery` for lists that keep their own grouping or + * order: every token must match one of the fields, punctuation-insensitive, + * without the fuzzy fallback. + */ +export function matchesRankQuery( + texts: ReadonlyArray, + query: string, +): boolean { + const tokens = tokenizeRankQuery(query); + if (tokens.length === 0) return true; + const rankFields = buildRankFields(texts); + return tokens.every((token) => scoreRankToken(token, rankFields) !== RANK_TOKEN_MISS); +} + export function partitionByFuzzyQuery( items: T[], query: string,