refactor(ts): one fuzzy + model-search-text helper in @hermes/shared; desktop picker ranks with fuzzyRank

Three byte-identical (modulo prettier and a "keep in sync" header comment)
copies of model-search-text.ts and two of fuzzy.ts collapse into one copy
each under apps/shared/src, exported from the package root and as the
subpaths `@hermes/shared/fuzzy` / `@hermes/shared/model-search-text` (the
TUI compiles with lib ES2023 and imports subpaths, never the DOM-typed
root). The vitest suites move with the code; no re-export shims remain.

Sites (path::symbol -> canonical):

  ui-tui/src/lib/fuzzy.ts::fuzzyScore/fuzzyScoreMulti/fuzzyRank   -> apps/shared/src/fuzzy.ts (moved)
  web/src/lib/fuzzy.ts::fuzzyScore/fuzzyScoreMulti/fuzzyRank      -> deleted
  ui-tui/src/lib/model-search-text.ts::modelSearchText            -> apps/shared/src/model-search-text.ts (moved)
  web/src/lib/model-search-text.ts::modelSearchText               -> deleted
  apps/desktop/src/lib/model-search-text.ts::modelSearchText      -> deleted
  ui-tui/src/lib/fuzzy.test.ts                                    -> apps/shared/src/fuzzy.test.ts (moved)
  ui-tui/src/lib/model-search-text.test.ts                        -> apps/shared/src/model-search-text.test.ts (moved)
  ui-tui/src/components/modelPicker.tsx::fuzzyRank, modelSearchText     -> @hermes/shared/fuzzy, @hermes/shared/model-search-text
  web/src/components/ModelPickerDialog.tsx::fuzzyRank, modelSearchText  -> @hermes/shared
  web/src/lib/model-picker-filter.ts::fuzzyScoreMulti                   -> @hermes/shared
  apps/desktop/src/components/model-picker.tsx::modelSearchText         -> @hermes/shared (+ fuzzyRank, see below)

The header comment now names only the cross-language twin
(hermes_cli/model_search.py) as the thing to keep in sync.

Behavior change (desktop only): the desktop model picker used to filter
model rows with `foldIncludes` substring matching and keep the curated
order; it now ranks them with the same `fuzzyRank(models, query,
modelSearchText)` the web and TUI pickers use. What a user sees
differently while typing a query:

  - subsequence queries match: "g4o" now finds "gpt-4o" (previously only
    a literal substring such as "gpt-4" or "4o" matched);
  - the best match floats to the top instead of rows staying in curated
    order (exact > prefix > word-boundary > contiguous > scattered);
  - a query that matches the provider name/slug still shows that
    provider's full curated list in order, exactly as before;
  - an empty query still shows the curated list verbatim.

The in-row highlight is unchanged (substring emphasis via HighlightMatches),
so a fuzzy-only hit renders without emphasis rather than mis-highlighting.

Tests: apps/desktop/src/components/model-picker.test.tsx::"orders model
rows exactly as the shared fuzzyRank does" asserts the rendered row order
equals the shared fuzzyRank order for the same inputs (fails on both the
old substring filter and a reversed ranking).
This commit is contained in:
teknium1
2026-09-12 19:39:02 -07:00
committed by Teknium
parent 158ab33518
commit a2ae8f229d
14 changed files with 61 additions and 280 deletions

View File

@@ -1,6 +1,7 @@
import type { ModelOptionsResponse } from '@hermes/shared'
import { fuzzyRank, modelSearchText } from '@hermes/shared'
import { QueryClient, QueryClientProvider } from '@tanstack/react-query'
import { cleanup, render, screen, waitFor } from '@testing-library/react'
import { cleanup, fireEvent, render, screen, waitFor } from '@testing-library/react'
import type { ReactElement } from 'react'
import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'
@@ -150,3 +151,31 @@ describe('ModelPickerDialog download rows', () => {
})
})
})
describe('ModelPickerDialog search ranking', () => {
// Rows must come out in the order the shared fuzzyRank produces — the same
// helper the web and TUI pickers use — so a query ranks identically on
// every surface. Curated order puts the scattered match first; the ranked
// order does not, which is what proves the picker is not substring-filtering.
const MODELS = ['glm-4.6-omni', 'claude-sonnet-4', 'gpt-4o']
it('orders model rows exactly as the shared fuzzyRank does', async () => {
vi.mocked(requestModelOptions).mockResolvedValue({
providers: [{ slug: 'nous', name: 'Nous', models: MODELS, authenticated: true }]
})
renderPicker({ currentModel: 'gpt-4o', currentProvider: 'nous' })
await screen.findByText('gpt-4o')
const query = 'g4o'
fireEvent.change(screen.getByRole('combobox'), { target: { value: query } })
const expected = fuzzyRank(MODELS, query, modelSearchText).map(r => r.item)
expect(expected).not.toEqual(MODELS.filter(m => expected.includes(m)))
await waitFor(() => {
const rows = screen.getAllByRole('option').map(el => el.textContent?.trim())
expect(rows).toEqual(expected)
})
})
})

View File

@@ -1,11 +1,11 @@
import type { ModelOptionProvider, ModelPricing } from '@hermes/shared'
import { fuzzyRank, modelSearchText } from '@hermes/shared'
import { useQuery } from '@tanstack/react-query'
import { useEffect, useMemo, useState } from 'react'
import { getLocalModelsStatus } from '@/hermes'
import { useI18n } from '@/i18n'
import { catalogProviderMatches, modelOptionsQueryKey, requestModelOptions } from '@/lib/model-options'
import { modelSearchText } from '@/lib/model-search-text'
import { currentPickerSelection } from '@/lib/model-status-label'
import { foldIncludes, normalize } from '@/lib/text'
import { useStoreSelector } from '@/lib/use-session-slice'
@@ -62,8 +62,8 @@ export function ModelPickerDialog({
// Own the search term so we can filter manually. cmdk's built-in
// shouldFilter reorders items by its fuzzy-match score (≈alphabetical with
// an empty query), which destroys the backend's curated order. We disable
// it and do a plain substring filter that preserves array order — matching
// the `hermes model` CLI picker, which shows the curated list verbatim.
// it: an empty query shows the curated list verbatim (like the `hermes
// model` CLI picker) and a query ranks with the shared fuzzyRank.
const [search, setSearch] = useState('')
const modelOptions = useQuery({
@@ -265,8 +265,16 @@ function ModelResults({
const q = normalize(search)
const matches = (provider: ModelOptionProvider, model: string) =>
!q || foldIncludes(modelSearchText(model), q) || foldIncludes(provider.name, q) || foldIncludes(provider.slug, q)
// Model rows rank with the same fuzzyRank + modelSearchText the web and TUI
// pickers use, so one query orders identically on every surface. A query
// that names the provider itself keeps its whole curated list, in order.
const rankModels = (provider: ModelOptionProvider, models: readonly string[]) => {
if (!q || foldIncludes(provider.name, q) || foldIncludes(provider.slug, q)) {
return [...models]
}
return fuzzyRank(models, q, modelSearchText).map(r => r.item)
}
// Only configured providers (those with curated models) are selectable
// here. Switching to a NOT-yet-configured provider goes through the
@@ -289,8 +297,8 @@ function ModelResults({
return (
<>
{configured.map(provider => {
// Preserve the backend's curated order — filter in place, no re-sort.
const models = (provider.models ?? []).filter(m => matches(provider, m))
// Empty query: the backend's curated order, verbatim.
const models = rankModels(provider, provider.models ?? [])
const groupDownloads = provider.slug === LOCAL_PROVIDER_SLUG ? visibleDownloads : []
if (models.length === 0 && groupDownloads.length === 0) {

View File

@@ -1,33 +0,0 @@
/**
* Extra tokens used only for model-picker search ranking.
*
* Wire IDs stay unchanged — some providers report short or brand-less ids
* (Kimi Coding's flagship is literally `k3`) that users still search for by
* the familiar `kimi-…` naming of sibling models.
*
* Keep in sync with ui-tui/src/lib/model-search-text.ts,
* web/src/lib/model-search-text.ts, and hermes_cli/model_search.py.
*/
const MODEL_SEARCH_ALIASES: Record<string, readonly string[]> = {
k3: ['kimi-k3', 'kimi'],
// OpenCode Zen serves the "Ox Alpha" stealth model under an opaque
// preview slug; let users find it by its public codename.
'x-preview-f-free': ['ox-alpha', 'ox']
}
/** Haystack for fuzzy/substring model search; never changes the wire id. */
export function modelSearchText(model: string): string {
const id = model.trim()
if (!id) {
return model
}
const aliases = MODEL_SEARCH_ALIASES[id.toLowerCase()]
if (!aliases?.length) {
return id
}
return `${id} ${aliases.join(' ')}`
}

View File

@@ -8,10 +8,12 @@
"./billing": "./src/billing-types.ts",
"./billing-policy": "./src/billing-policy.ts",
"./charge-settlement": "./src/charge-settlement.ts",
"./fuzzy": "./src/fuzzy.ts",
"./gateway-events": "./src/gateway-events.ts",
"./skin": "./src/skin.ts",
"./json-rpc-channel": "./src/json-rpc-channel.ts",
"./reconnect-backoff": "./src/reconnect-backoff.ts"
"./model-search-text": "./src/model-search-text.ts",
"./reconnect-backoff": "./src/reconnect-backoff.ts",
"./skin": "./src/skin.ts"
},
"types": "./src/index.ts",
"scripts": {

View File

@@ -1,6 +1,6 @@
import { describe, expect, it } from 'vitest'
import { fuzzyRank, fuzzyScore, fuzzyScoreMulti } from './fuzzy.js'
import { fuzzyRank, fuzzyScore, fuzzyScoreMulti } from './fuzzy'
describe('fuzzyScore', () => {
it('matches a query as a subsequence (g4o → gpt-4o)', () => {

View File

@@ -11,10 +11,8 @@
// intentionally simple — no external dependency — but good enough to make
// `son4` rank `claude-sonnet-4` above an incidental scattered hit.
//
// The WebUI ships a logically identical copy of this module at
// web/src/lib/fuzzy.ts (only prettier formatting differs); keep the two in
// sync. The TUI copy carries the vitest suite (the web package has no test
// runner), so changes should be validated here.
// Shared by the desktop, web and TUI pickers via `@hermes/shared/fuzzy` so a
// query ranks identically on every surface.
export interface FuzzyMatch {
/** Total score; higher is better. */

View File

@@ -46,6 +46,7 @@ export {
DATA_URL_READ_MAX_MAX_MB,
DATA_URL_READ_MIN_MAX_MB
} from './data-url-read-max'
export { type FuzzyMatch, fuzzyRank, fuzzyScore, fuzzyScoreMulti, type RankedItem } from './fuzzy'
export {
type ApprovalRequestPayload,
BACKEND_EVENT_NAMES,
@@ -111,6 +112,7 @@ export {
type WebSocketLike
} from './json-rpc-gateway'
export { reconnectBackoffDelayMs, type ReconnectBackoffOptions } from './reconnect-backoff'
export { modelSearchText } from './model-search-text'
export { skillInvocationText } from './skill-scaffold'
export {
type HermesSkin,

View File

@@ -1,7 +1,7 @@
import { describe, expect, it } from 'vitest'
import { fuzzyRank } from './fuzzy.js'
import { modelSearchText } from './model-search-text.js'
import { fuzzyRank } from './fuzzy'
import { modelSearchText } from './model-search-text'
describe('modelSearchText', () => {
it('keeps ordinary model ids unchanged', () => {

View File

@@ -5,8 +5,7 @@
* (Kimi Coding's flagship is literally `k3`) that users still search for by
* the familiar `kimi-…` naming of sibling models.
*
* Keep in sync with web/src/lib/model-search-text.ts and
* hermes_cli/model_search.py.
* Keep in sync with the Python twin, hermes_cli/model_search.py.
*/
const MODEL_SEARCH_ALIASES: Record<string, readonly string[]> = {
k3: ['kimi-k3', 'kimi'],

View File

@@ -1,12 +1,12 @@
import { Box, Text, useInput, useStdout } from '@hermes/ink'
import type { ModelOptionProvider, ModelOptionsResponse } from '@hermes/shared/gateway-events'
import { fuzzyRank } from '@hermes/shared/fuzzy'
import { modelSearchText } from '@hermes/shared/model-search-text'
import { useEffect, useMemo, useState } from 'react'
import { providerDisplayNames } from '../domain/providers.js'
import { TUI_SESSION_MODEL_FLAG } from '../domain/slash.js'
import type { GatewayClient } from '../gatewayClient.js'
import { fuzzyRank } from '../lib/fuzzy.js'
import { modelSearchText } from '../lib/model-search-text.js'
import { asRpcResult, rpcErrorMessage } from '../lib/rpc.js'
import type { Theme } from '../theme.js'

View File

@@ -11,9 +11,8 @@ import { Check, RefreshCw, Search, X } from "lucide-react";
import { useEffect, useMemo, useRef, useState } from "react";
import { createPortal } from "react-dom";
import { cn, themedBody } from "@/lib/utils";
import { fuzzyRank } from "@/lib/fuzzy";
import { queryMatchesProviderOnly } from "@/lib/model-picker-filter";
import { modelSearchText } from "@/lib/model-search-text";
import { fuzzyRank, modelSearchText } from "@hermes/shared";
/**
* Two-stage model picker modal.

View File

@@ -1,192 +0,0 @@
// Lightweight fuzzy subsequence scorer for picker filtering.
//
// Matches a query as an ordered subsequence of the target (so `g4o` matches
// `gpt-4o`) and scores by match quality so callers can rank results. Higher
// score is a better match. Returns the matched character indices so callers
// can highlight them.
//
// The scoring favours, in rough order: exact full match, prefix match, matches
// that start on a word boundary (after `-`, `_`, `/`, `.`, space, or a
// lower→upper case transition), contiguous runs, and earlier matches. This is
// intentionally simple — no external dependency — but good enough to make
// `son4` rank `claude-sonnet-4` above an incidental scattered hit.
//
// This is a logically identical copy of ui-tui/src/lib/fuzzy.ts (only prettier
// formatting differs); keep the two in sync. The TUI copy carries the vitest
// suite (this `web` package has no test runner), so behavioural changes should
// be validated there.
export interface FuzzyMatch {
/** Total score; higher is better. */
score: number;
/** Indices into the original (non-lowercased) target that were matched. */
positions: number[];
}
const WORD_BOUNDARY = /[-_/.\s]/;
function isBoundary(target: string, index: number): boolean {
if (index === 0) {
return true;
}
const prev = target[index - 1];
if (WORD_BOUNDARY.test(prev)) {
return true;
}
// camelCase / lower→upper transition (e.g. the `O` in `gptO`).
const cur = target[index];
return (
prev === prev.toLowerCase() &&
cur !== cur.toLowerCase() &&
cur === cur.toUpperCase()
);
}
/**
* Score a single query token against a target. Returns null when the token is
* not a subsequence of the target. An empty query scores 0 with no positions.
*/
export function fuzzyScore(target: string, query: string): FuzzyMatch | null {
if (!query) {
return { score: 0, positions: [] };
}
const lowerTarget = target.toLowerCase();
const lowerQuery = query.toLowerCase();
const positions: number[] = [];
let score = 0;
let prevIndex = -1;
let searchFrom = 0;
for (const ch of lowerQuery) {
const idx = lowerTarget.indexOf(ch, searchFrom);
if (idx < 0) {
return null;
}
positions.push(idx);
// Base point for the matched character.
score += 1;
// Contiguous with the previous match → strong bonus.
if (prevIndex >= 0 && idx === prevIndex + 1) {
score += 5;
} else if (prevIndex >= 0) {
// Penalise the gap we had to skip (capped), so contiguous beats scattered.
score -= Math.min(idx - prevIndex - 1, 3);
}
// Word-boundary / start-of-string matches are meaningful.
if (isBoundary(target, idx)) {
score += 3;
}
// Matching the very first character of the target is the strongest signal.
if (idx === 0) {
score += 5;
}
prevIndex = idx;
searchFrom = idx + 1;
}
// Prefix bonus: the query matched a contiguous prefix of the target.
if (
positions.length &&
positions[0] === 0 &&
positions[positions.length - 1] === positions.length - 1
) {
score += 8;
}
// Exact full match dominates everything else.
if (lowerTarget === lowerQuery) {
score += 20;
}
// Slightly prefer shorter targets when scores are otherwise close, so a
// query that fully prefixes a short id beats the same prefix on a long one.
score -= lowerTarget.length * 0.01;
return { score, positions };
}
/**
* Score a target against a whitespace-separated, multi-token query. Every token
* must match (AND semantics); the result aggregates per-token scores and the
* union of matched positions. Returns null if any token fails to match.
*/
export function fuzzyScoreMulti(
target: string,
query: string,
): FuzzyMatch | null {
const tokens = query.trim().toLowerCase().split(/\s+/).filter(Boolean);
if (!tokens.length) {
return { score: 0, positions: [] };
}
let score = 0;
const positionSet = new Set<number>();
for (const token of tokens) {
const match = fuzzyScore(target, token);
if (!match) {
return null;
}
score += match.score;
for (const pos of match.positions) {
positionSet.add(pos);
}
}
return { score, positions: [...positionSet].sort((a, b) => a - b) };
}
export interface RankedItem<T> {
item: T;
score: number;
positions: number[];
}
/**
* Filter + rank a list by a fuzzy query against a derived text key. Non-matching
* items are dropped; matches are sorted by score (descending), ties broken by
* the original index so ordering is stable for equal scores. An empty query
* returns every item in original order with no positions.
*/
export function fuzzyRank<T>(
items: readonly T[],
query: string,
toText: (item: T) => string,
): RankedItem<T>[] {
const trimmed = query.trim();
if (!trimmed) {
return items.map((item) => ({ item, score: 0, positions: [] }));
}
const ranked: Array<RankedItem<T> & { index: number }> = [];
items.forEach((item, index) => {
const match = fuzzyScoreMulti(toText(item), trimmed);
if (match) {
ranked.push({ item, score: match.score, positions: match.positions, index });
}
});
ranked.sort((a, b) => b.score - a.score || a.index - b.index);
return ranked.map(({ item, score, positions }) => ({ item, score, positions }));
}

View File

@@ -1,4 +1,4 @@
import { fuzzyScoreMulti } from "@/lib/fuzzy";
import { fuzzyScoreMulti } from "@hermes/shared";
/**
* True when `trimmedQuery` located the selected provider by name/slug but

View File

@@ -1,31 +0,0 @@
/**
* Extra tokens used only for model-picker search ranking.
*
* Wire IDs stay unchanged — some providers report short or brand-less ids
* (Kimi Coding's flagship is literally `k3`) that users still search for by
* the familiar `kimi-…` naming of sibling models.
*
* Keep in sync with ui-tui/src/lib/model-search-text.ts and
* hermes_cli/model_search.py. Behavioural tests live in the TUI package.
*/
const MODEL_SEARCH_ALIASES: Record<string, readonly string[]> = {
k3: ["kimi-k3", "kimi"],
// OpenCode Zen serves the "Ox Alpha" stealth model under an opaque
// preview slug; let users find it by its public codename.
"x-preview-f-free": ["ox-alpha", "ox"],
};
/** Haystack for fuzzy/substring model search; never changes the wire id. */
export function modelSearchText(model: string): string {
const id = model.trim();
if (!id) {
return model;
}
const aliases = MODEL_SEARCH_ALIASES[id.toLowerCase()];
if (!aliases?.length) {
return id;
}
return `${id} ${aliases.join(" ")}`;
}