Attempt at an improved fuzzy search for pickers (page, document, command)
This commit is contained in:
@@ -6,7 +6,7 @@ import type { FunctionalComponent } from "preact";
|
||||
import { useEffect, useRef, useState } from "preact/hooks";
|
||||
import type { FilterOption } from "@silverbulletmd/silverbullet/type/client";
|
||||
import { MiniEditor } from "./mini_editor.tsx";
|
||||
import { fuzzySearchAndSort } from "../lib/fuse_search.ts";
|
||||
import { fuzzySearchAndSort } from "../lib/fuzzy_search.ts";
|
||||
import { deepEqual } from "../../plug-api/lib/json.ts";
|
||||
import { AlwaysShownModal } from "./basic_modals.tsx";
|
||||
import type { EditorView } from "@codemirror/view";
|
||||
|
||||
@@ -1,40 +0,0 @@
|
||||
import { expect, test } from "vitest";
|
||||
import type { FilterOption } from "@silverbulletmd/silverbullet/type/client";
|
||||
import { fuzzySearchAndSort } from "./fuse_search.ts";
|
||||
|
||||
test("testFuzzyFilter", () => {
|
||||
const array: FilterOption[] = [
|
||||
{ name: "My Company/Hank", orderId: 2 },
|
||||
{ name: "My Company/Hane", orderId: 1 },
|
||||
{ name: "My Company/Steve Co" },
|
||||
{ name: "Other/Steve" },
|
||||
{ name: "Steve" },
|
||||
];
|
||||
|
||||
// Prioritize match in last path part
|
||||
let results = fuzzySearchAndSort(array, "");
|
||||
expect(results.length).toEqual(array.length);
|
||||
results = fuzzySearchAndSort(array, "Steve");
|
||||
expect(results.length).toEqual(3);
|
||||
results = fuzzySearchAndSort(array, "Co");
|
||||
// Match in last path part
|
||||
expect(results[0].name).toEqual("My Company/Steve Co");
|
||||
// Due to orderId
|
||||
expect(results[1].name).toEqual("My Company/Hane");
|
||||
expect(results[2].name).toEqual("My Company/Hank");
|
||||
});
|
||||
|
||||
test("Fuzzy search edge case testing", () => {
|
||||
// Test edge case where aspiring pages (Infinity orderId) sort last without NaN
|
||||
const array: FilterOption[] = [
|
||||
{ name: "Existing", orderId: 1 },
|
||||
{ name: "Aspiring A", orderId: Infinity },
|
||||
{ name: "Aspiring B", orderId: Infinity },
|
||||
];
|
||||
const results = fuzzySearchAndSort(array, "");
|
||||
expect(results.length).toEqual(3);
|
||||
expect(results[0].name).toEqual("Existing");
|
||||
// Both aspiring pages last; relative order between them is stable (no NaN)
|
||||
expect(results[1].name).toEqual("Aspiring A");
|
||||
expect(results[2].name).toEqual("Aspiring B");
|
||||
});
|
||||
@@ -1,95 +0,0 @@
|
||||
import Fuse from "fuse.js";
|
||||
import type { FilterOption } from "@silverbulletmd/silverbullet/type/client";
|
||||
import { fileName } from "@silverbulletmd/silverbullet/lib/resolve";
|
||||
|
||||
type FuseOption = FilterOption & {
|
||||
baseName: string;
|
||||
};
|
||||
|
||||
/**
|
||||
* Compare orderIds; aspiring pages (Infinity) sort last. Avoids NaN when both are Infinity.
|
||||
* @returns -1 if a is less than b, 1 if a is greater than b, 0 if they are equal
|
||||
*/
|
||||
function compareOrderId(a: number | undefined, b: number | undefined): number {
|
||||
const aOrder = a ?? 0;
|
||||
const bOrder = b ?? 0;
|
||||
if (aOrder === Infinity && bOrder === Infinity) return 0;
|
||||
if (aOrder === Infinity) return 1;
|
||||
if (bOrder === Infinity) return -1;
|
||||
return aOrder - bOrder;
|
||||
}
|
||||
|
||||
export const fuzzySearchAndSort = (
|
||||
arr: FilterOption[],
|
||||
searchPhrase: string,
|
||||
): FilterOption[] => {
|
||||
if (!searchPhrase) {
|
||||
return arr.sort((a, b) => compareOrderId(a.orderId, b.orderId));
|
||||
}
|
||||
|
||||
const enrichedArr: FuseOption[] = arr.map((item) => {
|
||||
return {
|
||||
...item,
|
||||
// Only relevant for pages and or documents and not commands
|
||||
baseName: fileName(item.name),
|
||||
displayName: item?.meta?.displayName,
|
||||
aliases: item?.meta?.aliases?.join(" "),
|
||||
};
|
||||
});
|
||||
|
||||
const fuse = new Fuse(enrichedArr, {
|
||||
keys: [
|
||||
{
|
||||
name: "name",
|
||||
weight: 2,
|
||||
},
|
||||
{
|
||||
name: "baseName",
|
||||
weight: 3,
|
||||
},
|
||||
{
|
||||
name: "displayName",
|
||||
weight: 2,
|
||||
},
|
||||
{
|
||||
name: "aliases",
|
||||
weight: 2,
|
||||
},
|
||||
{
|
||||
name: "description",
|
||||
weight: 1,
|
||||
},
|
||||
],
|
||||
isCaseSensitive: false,
|
||||
ignoreDiacritics: true,
|
||||
shouldSort: true,
|
||||
threshold: 0.3,
|
||||
includeScore: true,
|
||||
sortFn: (a, b): number => {
|
||||
const aItem = enrichedArr[a.idx];
|
||||
const bItem = enrichedArr[b.idx];
|
||||
|
||||
// If either orderId is Infinity (aspiring page), put it last (compareOrderId avoids NaN)
|
||||
if (aItem.orderId === Infinity || bItem.orderId === Infinity) {
|
||||
return compareOrderId(aItem.orderId, bItem.orderId);
|
||||
}
|
||||
|
||||
// If scores are the same, use orderId for sorting
|
||||
if (a.score === b.score) {
|
||||
const aOrder = aItem.orderId ?? 0;
|
||||
const bOrder = bItem.orderId ?? 0;
|
||||
if (aOrder !== bOrder) {
|
||||
return aOrder - bOrder;
|
||||
}
|
||||
}
|
||||
return a.score - b.score;
|
||||
},
|
||||
});
|
||||
|
||||
const results = fuse.search(searchPhrase);
|
||||
const enhancedResults = results.map((r) => ({
|
||||
...r.item,
|
||||
fuseScore: r.score,
|
||||
}));
|
||||
return enhancedResults;
|
||||
};
|
||||
@@ -0,0 +1,314 @@
|
||||
import { expect, test } from "vitest";
|
||||
import type { FilterOption } from "@silverbulletmd/silverbullet/type/client";
|
||||
import {
|
||||
fuzzySearchAndSort,
|
||||
scoreCandidate,
|
||||
scoreToken,
|
||||
} from "./fuzzy_search.ts";
|
||||
|
||||
type Expect = {
|
||||
topMatch?: string;
|
||||
mustIncludeInTopN?: string[];
|
||||
n?: number;
|
||||
mustExclude?: string[];
|
||||
resultCount?: number;
|
||||
};
|
||||
|
||||
function assertSearch(
|
||||
corpus: FilterOption[],
|
||||
query: string,
|
||||
expected: Expect,
|
||||
): void {
|
||||
const results = fuzzySearchAndSort(corpus, query);
|
||||
const names = results.map((r) => r.name);
|
||||
const detail = `\nQuery: ${JSON.stringify(query)}\nRanked: ${
|
||||
JSON.stringify(names, null, 2)
|
||||
}`;
|
||||
if (expected.resultCount !== undefined) {
|
||||
expect(results.length, "result count" + detail).toEqual(
|
||||
expected.resultCount,
|
||||
);
|
||||
}
|
||||
if (expected.topMatch !== undefined) {
|
||||
expect(names[0], "top match" + detail).toEqual(expected.topMatch);
|
||||
}
|
||||
if (expected.mustIncludeInTopN) {
|
||||
const n = expected.n ?? expected.mustIncludeInTopN.length;
|
||||
const topN = names.slice(0, n);
|
||||
for (const name of expected.mustIncludeInTopN) {
|
||||
expect(topN, `expected ${name} in top ${n}` + detail).toContain(name);
|
||||
}
|
||||
}
|
||||
if (expected.mustExclude) {
|
||||
for (const name of expected.mustExclude) {
|
||||
expect(names, `expected ${name} excluded` + detail).not.toContain(name);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
test("motivating bug: 'silv todo' matches 'SilverBullet/TODO'", () => {
|
||||
const corpus: FilterOption[] = [
|
||||
{ name: "SilverBullet/TODO" },
|
||||
{ name: "Silver" },
|
||||
{ name: "TODO" },
|
||||
{ name: "Silver/Other" },
|
||||
];
|
||||
assertSearch(corpus, "silv todo", { topMatch: "SilverBullet/TODO" });
|
||||
});
|
||||
|
||||
test("scoreToken: exact case-insensitive equality scores 1.0", () => {
|
||||
expect(scoreToken("todo", "TODO")).toBeCloseTo(1.0, 5);
|
||||
expect(scoreToken("Silver", "silver")).toBeCloseTo(1.0, 5);
|
||||
});
|
||||
|
||||
test("scoreToken: prefix at start scores 0.95 (always word boundary)", () => {
|
||||
expect(scoreToken("silv", "silverbullet")).toBeCloseTo(0.95, 5);
|
||||
expect(scoreToken("silv", "Silver/Other")).toBeCloseTo(0.95, 5);
|
||||
});
|
||||
|
||||
test("scoreToken: substring scores 0.75 (plain) or 0.80 (word boundary)", () => {
|
||||
expect(scoreToken("ver", "Silver")).toBeCloseTo(0.75, 5);
|
||||
expect(scoreToken("silver", "Some/Silver/Thing")).toBeCloseTo(0.80, 5);
|
||||
});
|
||||
|
||||
test("scoreToken: no match returns 0 (before tiers 4/5 implemented)", () => {
|
||||
expect(scoreToken("xyz", "Silver")).toEqual(0);
|
||||
});
|
||||
|
||||
test("scoreToken: subsequence with gaps scores between 0.30 and 0.65", () => {
|
||||
const s = scoreToken("slvr", "SilverBullet");
|
||||
expect(s).toBeGreaterThan(0.3);
|
||||
expect(s).toBeLessThanOrEqual(0.65);
|
||||
});
|
||||
|
||||
test("scoreToken: subsequence with word-boundary starts ranks higher than scattered", () => {
|
||||
const boundary = scoreToken("pae", "PageEditor"); // P-a-...-e at word starts
|
||||
const scattered = scoreToken("pae", "appendage"); // scattered chars
|
||||
expect(boundary).toBeGreaterThan(scattered);
|
||||
});
|
||||
|
||||
test("scoreToken: contiguous subsequence scores higher than gapped", () => {
|
||||
const contiguous = scoreToken("slvr", "slvrabcdef");
|
||||
const gapped = scoreToken("slvr", "s_l_v_r_abc");
|
||||
expect(contiguous).toBeGreaterThan(gapped);
|
||||
});
|
||||
|
||||
test("scoreToken: typo tolerance matches when token length >= 4 and edits <= 2", () => {
|
||||
// 'slverbullet' vs 'silverbullet' = 1 deletion
|
||||
const s = scoreToken("slverbullet", "silverbullet");
|
||||
expect(s).toBeGreaterThan(0); // some typo-tier match
|
||||
// But it must be lower than an exact match would be
|
||||
expect(s).toBeLessThan(scoreToken("silverbullet", "silverbullet"));
|
||||
});
|
||||
|
||||
test("scoreToken: short tokens (<4 chars) do NOT get typo tolerance", () => {
|
||||
// 'cot' is not a subsequence of 'cat' (different middle char), edit distance 1
|
||||
// For length 3, no typo tolerance.
|
||||
expect(scoreToken("cot", "cat")).toEqual(0);
|
||||
});
|
||||
|
||||
test("scoreToken: typo with too many edits returns 0", () => {
|
||||
// 'abcde' vs 'xyzab' has edit distance > 2 — should not match
|
||||
expect(scoreToken("abcde", "xyzab")).toEqual(0);
|
||||
});
|
||||
|
||||
test("scoreCandidate: multi-token AND requires every token to match", () => {
|
||||
const opt: FilterOption = { name: "SilverBullet/TODO" };
|
||||
expect(scoreCandidate("silv todo", opt)).not.toBeNull();
|
||||
// 'xyz' fails — whole candidate fails
|
||||
expect(scoreCandidate("silv xyz", opt)).toBeNull();
|
||||
});
|
||||
|
||||
test("scoreCandidate: scores against displayName and aliases", () => {
|
||||
const opt: FilterOption = {
|
||||
name: "internal/x",
|
||||
meta: { displayName: "Project Tasks", aliases: ["todos"] },
|
||||
};
|
||||
expect(scoreCandidate("project", opt)).not.toBeNull();
|
||||
expect(scoreCandidate("todos", opt)).not.toBeNull();
|
||||
});
|
||||
|
||||
test("scoreCandidate: description is NOT searched", () => {
|
||||
const opt: FilterOption = {
|
||||
name: "internal/x",
|
||||
description: "some description text",
|
||||
};
|
||||
expect(scoreCandidate("description", opt)).toBeNull();
|
||||
});
|
||||
|
||||
test("scoreCandidate: basename match outscores name-only match", () => {
|
||||
const basenameMatch: FilterOption = { name: "Other/Steve" };
|
||||
const nameOnly: FilterOption = { name: "Steve/Other" };
|
||||
// 'Steve' is basename in first, not in second
|
||||
const a = scoreCandidate("Steve", basenameMatch)!;
|
||||
const b = scoreCandidate("Steve", nameOnly)!;
|
||||
expect(a).toBeGreaterThan(b);
|
||||
});
|
||||
|
||||
test("seed 1: 'silv todo' matches SilverBullet/TODO", () => {
|
||||
assertSearch(
|
||||
[
|
||||
{ name: "SilverBullet/TODO" },
|
||||
{ name: "Silver" },
|
||||
{ name: "TODO" },
|
||||
{ name: "Silver/Other" },
|
||||
],
|
||||
"silv todo",
|
||||
{ topMatch: "SilverBullet/TODO" },
|
||||
);
|
||||
});
|
||||
|
||||
test("seed 2: 'silverbullet' exact wins, all SilverBullet/* in top-3", () => {
|
||||
assertSearch(
|
||||
[
|
||||
{ name: "SilverBullet/TODO" },
|
||||
{ name: "SilverBullet/Notes" },
|
||||
{ name: "SilverBullet" },
|
||||
],
|
||||
"silverbullet",
|
||||
{
|
||||
topMatch: "SilverBullet",
|
||||
mustIncludeInTopN: ["SilverBullet/TODO", "SilverBullet/Notes"],
|
||||
n: 3,
|
||||
},
|
||||
);
|
||||
});
|
||||
|
||||
test("seed 3: 'pae' matches PageEditor via subsequence", () => {
|
||||
assertSearch(
|
||||
[
|
||||
{ name: "PageEditor" },
|
||||
{ name: "PathEditor" },
|
||||
{ name: "Other" },
|
||||
],
|
||||
"pae",
|
||||
{ mustIncludeInTopN: ["PageEditor", "PathEditor"], n: 2 },
|
||||
);
|
||||
});
|
||||
|
||||
test("seed 4: 'slverbullet' typo matches SilverBullet", () => {
|
||||
assertSearch(
|
||||
[
|
||||
{ name: "SilverBullet" },
|
||||
{ name: "Silver" },
|
||||
{ name: "Bullet" },
|
||||
],
|
||||
"slverbullet",
|
||||
{ topMatch: "SilverBullet" },
|
||||
);
|
||||
});
|
||||
|
||||
test("seed 5: 'Co' prefers basename match", () => {
|
||||
assertSearch(
|
||||
[
|
||||
{ name: "My Company/Hank", orderId: 2 },
|
||||
{ name: "My Company/Hane", orderId: 1 },
|
||||
{ name: "My Company/Steve Co" },
|
||||
{ name: "Other/Steve" },
|
||||
{ name: "Steve" },
|
||||
],
|
||||
"Co",
|
||||
{
|
||||
topMatch: "My Company/Steve Co",
|
||||
mustIncludeInTopN: ["My Company/Hane", "My Company/Hank"],
|
||||
n: 3,
|
||||
},
|
||||
);
|
||||
});
|
||||
|
||||
test("seed 6: 'in' prefix beats subsequence; no typo tolerance on short", () => {
|
||||
assertSearch(
|
||||
[
|
||||
{ name: "Inbox" },
|
||||
{ name: "In Progress" },
|
||||
{ name: "Index" },
|
||||
],
|
||||
"in",
|
||||
{
|
||||
mustIncludeInTopN: ["Inbox", "In Progress", "Index"],
|
||||
n: 3,
|
||||
},
|
||||
);
|
||||
});
|
||||
|
||||
test("seed 7: 'daily 02' matches Daily/2024-02-15", () => {
|
||||
assertSearch(
|
||||
[
|
||||
{ name: "Daily/2024-01-15" },
|
||||
{ name: "Daily/2024-02-15" },
|
||||
],
|
||||
"daily 02",
|
||||
{ topMatch: "Daily/2024-02-15", resultCount: 1 },
|
||||
);
|
||||
});
|
||||
|
||||
test("seed 8: no matches returns empty", () => {
|
||||
assertSearch(
|
||||
[{ name: "Foo" }, { name: "Bar" }],
|
||||
"xyz",
|
||||
{ resultCount: 0 },
|
||||
);
|
||||
});
|
||||
|
||||
test("seed 9: empty query sorts by orderId, aspiring (Infinity) last", () => {
|
||||
assertSearch(
|
||||
[
|
||||
{ name: "Existing", orderId: 1 },
|
||||
{ name: "Aspiring A", orderId: Infinity },
|
||||
{ name: "Aspiring B", orderId: Infinity },
|
||||
],
|
||||
"",
|
||||
{
|
||||
topMatch: "Existing",
|
||||
resultCount: 3,
|
||||
mustIncludeInTopN: ["Existing", "Aspiring A", "Aspiring B"],
|
||||
n: 3,
|
||||
},
|
||||
);
|
||||
});
|
||||
|
||||
test("seed 10: 'readme' case-insensitive matches both", () => {
|
||||
assertSearch(
|
||||
[{ name: "README" }, { name: "readme.md" }],
|
||||
"readme",
|
||||
{
|
||||
mustIncludeInTopN: ["README", "readme.md"],
|
||||
n: 2,
|
||||
},
|
||||
);
|
||||
});
|
||||
|
||||
test("seed 11: alias matching", () => {
|
||||
assertSearch(
|
||||
[
|
||||
{ name: "internal/projects", meta: { aliases: ["todos", "tasks"] } },
|
||||
{ name: "other" },
|
||||
],
|
||||
"todos",
|
||||
{ topMatch: "internal/projects" },
|
||||
);
|
||||
});
|
||||
|
||||
test("seed 12: 1-2 char queries do not match via typo tolerance", () => {
|
||||
assertSearch(
|
||||
[{ name: "cat" }, { name: "dog" }],
|
||||
"co",
|
||||
{ resultCount: 0 },
|
||||
);
|
||||
});
|
||||
|
||||
test("preserves existing testFuzzyFilter behavior", () => {
|
||||
const array: FilterOption[] = [
|
||||
{ name: "My Company/Hank", orderId: 2 },
|
||||
{ name: "My Company/Hane", orderId: 1 },
|
||||
{ name: "My Company/Steve Co" },
|
||||
{ name: "Other/Steve" },
|
||||
{ name: "Steve" },
|
||||
];
|
||||
let results = fuzzySearchAndSort(array, "");
|
||||
expect(results.length).toEqual(array.length);
|
||||
results = fuzzySearchAndSort(array, "Steve");
|
||||
expect(results.length).toEqual(3);
|
||||
results = fuzzySearchAndSort(array, "Co");
|
||||
expect(results[0].name).toEqual("My Company/Steve Co");
|
||||
});
|
||||
@@ -0,0 +1,282 @@
|
||||
import type { FilterOption } from "@silverbulletmd/silverbullet/type/client";
|
||||
import { fileName } from "@silverbulletmd/silverbullet/lib/resolve";
|
||||
|
||||
function normalize(s: string): string {
|
||||
return s.normalize("NFD").replace(/\p{Diacritic}/gu, "").toLowerCase();
|
||||
}
|
||||
|
||||
function tokenize(query: string): string[] {
|
||||
return normalize(query).split(/\s+/).filter((t) => t.length > 0);
|
||||
}
|
||||
|
||||
const WORD_BOUNDARY_CHARS = new Set(["/", "-", "_", " ", ".", ":"]);
|
||||
|
||||
function isWordBoundaryAt(
|
||||
original: string,
|
||||
lowered: string,
|
||||
i: number,
|
||||
): boolean {
|
||||
if (i === 0) return true;
|
||||
const prev = lowered[i - 1];
|
||||
if (WORD_BOUNDARY_CHARS.has(prev)) return true;
|
||||
// CamelCase boundary: previous original char is lowercase, current is uppercase
|
||||
const origPrev = original[i - 1];
|
||||
const origCur = original[i];
|
||||
if (
|
||||
origPrev && origCur &&
|
||||
origPrev === origPrev.toLowerCase() &&
|
||||
origPrev !== origPrev.toUpperCase() &&
|
||||
origCur === origCur.toUpperCase() &&
|
||||
origCur !== origCur.toLowerCase()
|
||||
) {
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
function boundedDamerauLevenshtein(
|
||||
a: string,
|
||||
b: string,
|
||||
max: number,
|
||||
): number {
|
||||
// Returns edit distance if <= max, otherwise max + 1.
|
||||
const al = a.length;
|
||||
const bl = b.length;
|
||||
if (Math.abs(al - bl) > max) return max + 1;
|
||||
|
||||
// Three-row DP with early-exit per row, rotating row references each iteration.
|
||||
let prevPrev = new Array(bl + 1);
|
||||
let prev = new Array(bl + 1);
|
||||
let curr = new Array(bl + 1);
|
||||
for (let j = 0; j <= bl; j++) prev[j] = j;
|
||||
|
||||
for (let i = 1; i <= al; i++) {
|
||||
curr[0] = i;
|
||||
let rowMin = curr[0];
|
||||
for (let j = 1; j <= bl; j++) {
|
||||
const cost = a[i - 1] === b[j - 1] ? 0 : 1;
|
||||
let v = Math.min(
|
||||
prev[j] + 1,
|
||||
curr[j - 1] + 1,
|
||||
prev[j - 1] + cost,
|
||||
);
|
||||
// Damerau transposition
|
||||
if (
|
||||
i > 1 && j > 1 &&
|
||||
a[i - 1] === b[j - 2] && a[i - 2] === b[j - 1]
|
||||
) {
|
||||
v = Math.min(v, prevPrev[j - 2] + 1);
|
||||
}
|
||||
curr[j] = v;
|
||||
if (v < rowMin) rowMin = v;
|
||||
}
|
||||
if (rowMin > max) return max + 1;
|
||||
// Rotate: prevPrev <- prev, prev <- curr, curr <- old prevPrev (reused as scratch)
|
||||
const tmp = prevPrev;
|
||||
prevPrev = prev;
|
||||
prev = curr;
|
||||
curr = tmp;
|
||||
}
|
||||
return prev[bl];
|
||||
}
|
||||
|
||||
function typoScore(token: string, candidate: string): number {
|
||||
if (token.length < 4) return 0;
|
||||
const max = Math.min(2, Math.floor(token.length / 4));
|
||||
if (max < 1) return 0;
|
||||
|
||||
// Try aligning token against each substring of candidate of length token.length +/- max.
|
||||
let bestDist = max + 1;
|
||||
for (
|
||||
let len = Math.max(1, token.length - max);
|
||||
len <= token.length + max;
|
||||
len++
|
||||
) {
|
||||
for (let start = 0; start + len <= candidate.length; start++) {
|
||||
const slice = candidate.slice(start, start + len);
|
||||
const d = boundedDamerauLevenshtein(token, slice, bestDist - 1);
|
||||
if (d < bestDist) bestDist = d;
|
||||
if (bestDist === 0) break;
|
||||
}
|
||||
if (bestDist === 0) break;
|
||||
}
|
||||
if (bestDist > max) return 0;
|
||||
return 0.25 - 0.05 * bestDist; // 0.25 / 0.20 / 0.15
|
||||
}
|
||||
|
||||
function subsequenceScore(
|
||||
token: string,
|
||||
candidate: string,
|
||||
candidateOriginal: string,
|
||||
): number {
|
||||
// Greedy left-to-right match; track contiguous runs and word-boundary hits.
|
||||
let ti = 0;
|
||||
let firstMatch = -1;
|
||||
let lastMatch = -1;
|
||||
let contiguousRunSum = 0;
|
||||
let currentRun = 0;
|
||||
let boundaryHits = 0;
|
||||
let prevMatchCi = -2;
|
||||
|
||||
for (let ci = 0; ci < candidate.length && ti < token.length; ci++) {
|
||||
if (candidate[ci] === token[ti]) {
|
||||
if (firstMatch < 0) firstMatch = ci;
|
||||
lastMatch = ci;
|
||||
if (isWordBoundaryAt(candidateOriginal, candidate, ci)) boundaryHits++;
|
||||
if (ci === prevMatchCi + 1) {
|
||||
currentRun++;
|
||||
} else {
|
||||
contiguousRunSum += currentRun * currentRun;
|
||||
currentRun = 1;
|
||||
}
|
||||
prevMatchCi = ci;
|
||||
ti++;
|
||||
}
|
||||
}
|
||||
contiguousRunSum += currentRun * currentRun;
|
||||
|
||||
if (ti < token.length) return 0; // not a subsequence at all
|
||||
|
||||
// Normalize: token length over span (penalizes wide spans)
|
||||
const span = lastMatch - firstMatch + 1;
|
||||
const density = token.length / span; // in (0, 1]
|
||||
const boundaryBonus = boundaryHits / token.length; // in [0, 1]
|
||||
const contiguityBonus = contiguousRunSum / (token.length * token.length); // in (0, 1]
|
||||
|
||||
// Weighted combination scaled into [0.30, 0.65]
|
||||
const raw = 0.5 * density + 0.3 * contiguityBonus + 0.2 * boundaryBonus;
|
||||
return 0.30 + raw * 0.35;
|
||||
}
|
||||
|
||||
export function scoreToken(token: string, candidate: string): number {
|
||||
if (!token) return 0;
|
||||
const t = normalize(token);
|
||||
const cOrig = candidate;
|
||||
const c = normalize(candidate);
|
||||
|
||||
// Tier 1: exact equality
|
||||
if (c === t) return 1.0;
|
||||
|
||||
// Tier 2: prefix
|
||||
if (c.startsWith(t)) {
|
||||
return isWordBoundaryAt(cOrig, c, 0) ? 0.95 : 0.9;
|
||||
}
|
||||
|
||||
// Tier 3: substring. Search for the best occurrence — a word-boundary
|
||||
// match wins even if it appears later than a non-boundary match.
|
||||
{
|
||||
let bestIdx = -1;
|
||||
let bestBoundary = false;
|
||||
let idx = c.indexOf(t);
|
||||
while (idx >= 0) {
|
||||
const boundary = isWordBoundaryAt(cOrig, c, idx);
|
||||
if (bestIdx < 0 || (boundary && !bestBoundary)) {
|
||||
bestIdx = idx;
|
||||
bestBoundary = boundary;
|
||||
if (boundary) break; // can't get better
|
||||
}
|
||||
idx = c.indexOf(t, idx + 1);
|
||||
}
|
||||
if (bestIdx >= 0) {
|
||||
// For very short tokens (<=2 chars), only count substring matches at
|
||||
// word boundaries — otherwise short tokens generate too much noise.
|
||||
if (t.length > 2 || bestBoundary) {
|
||||
return bestBoundary ? 0.80 : 0.75;
|
||||
}
|
||||
// fall through for short non-boundary substring matches
|
||||
}
|
||||
}
|
||||
|
||||
// Tier 4: subsequence (short tokens skip this tier; substring at boundary
|
||||
// is the only way for them to match)
|
||||
if (t.length > 2) {
|
||||
const subseq = subsequenceScore(t, c, cOrig);
|
||||
if (subseq > 0) return subseq;
|
||||
}
|
||||
|
||||
// Tier 5: bounded typo
|
||||
const typo = typoScore(t, c);
|
||||
if (typo > 0) return typo;
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
type Field = { value: string; weight: number };
|
||||
|
||||
function fieldsOf(opt: FilterOption): Field[] {
|
||||
const fields: Field[] = [];
|
||||
const name = opt.name ?? "";
|
||||
fields.push({ value: name, weight: 0.85 });
|
||||
const base = fileName(name);
|
||||
if (base) fields.push({ value: base, weight: 1.0 });
|
||||
const displayName = opt.meta?.displayName;
|
||||
if (displayName) fields.push({ value: displayName, weight: 0.9 });
|
||||
const aliases = opt.meta?.aliases;
|
||||
if (aliases && aliases.length > 0) {
|
||||
fields.push({ value: aliases.join(" "), weight: 0.85 });
|
||||
}
|
||||
return fields;
|
||||
}
|
||||
|
||||
export function scoreCandidate(
|
||||
query: string,
|
||||
option: FilterOption,
|
||||
): number | null {
|
||||
const tokens = tokenize(query);
|
||||
if (tokens.length === 0) return 0;
|
||||
const fields = fieldsOf(option);
|
||||
if (fields.length === 0) return null;
|
||||
|
||||
const perTokenBest: number[] = [];
|
||||
for (const token of tokens) {
|
||||
let best = 0;
|
||||
for (const f of fields) {
|
||||
const raw = scoreToken(token, f.value);
|
||||
const weighted = raw * f.weight;
|
||||
if (weighted > best) best = weighted;
|
||||
}
|
||||
if (best === 0) return null; // any token fails ⇒ candidate excluded
|
||||
perTokenBest.push(best);
|
||||
}
|
||||
// Geometric mean
|
||||
let logSum = 0;
|
||||
for (const s of perTokenBest) logSum += Math.log(s);
|
||||
return Math.exp(logSum / perTokenBest.length);
|
||||
}
|
||||
|
||||
function compareOrderId(
|
||||
a: number | undefined,
|
||||
b: number | undefined,
|
||||
): number {
|
||||
const aOrder = a ?? 0;
|
||||
const bOrder = b ?? 0;
|
||||
if (aOrder === Infinity && bOrder === Infinity) return 0;
|
||||
if (aOrder === Infinity) return 1;
|
||||
if (bOrder === Infinity) return -1;
|
||||
return aOrder - bOrder;
|
||||
}
|
||||
|
||||
type Scored = { item: FilterOption; score: number };
|
||||
|
||||
export function fuzzySearchAndSort(
|
||||
arr: FilterOption[],
|
||||
query: string,
|
||||
): FilterOption[] {
|
||||
if (!query || query.trim() === "") {
|
||||
return [...arr].sort((a, b) => compareOrderId(a.orderId, b.orderId));
|
||||
}
|
||||
const scored: Scored[] = [];
|
||||
for (const item of arr) {
|
||||
const s = scoreCandidate(query, item);
|
||||
if (s !== null && s > 0) scored.push({ item, score: s });
|
||||
}
|
||||
scored.sort((a, b) => {
|
||||
if (a.score !== b.score) return b.score - a.score;
|
||||
const orderCmp = compareOrderId(a.item.orderId, b.item.orderId);
|
||||
if (orderCmp !== 0) return orderCmp;
|
||||
const lenCmp = (a.item.name?.length ?? 0) - (b.item.name?.length ?? 0);
|
||||
if (lenCmp !== 0) return lenCmp;
|
||||
return (a.item.name ?? "").localeCompare(b.item.name ?? "");
|
||||
});
|
||||
return scored.map((s) => ({ ...s.item, score: s.score }));
|
||||
}
|
||||
Generated
-10
@@ -37,7 +37,6 @@
|
||||
"esbuild": "^0.27.3",
|
||||
"fast-diff": "1.3.0",
|
||||
"fast-glob": "^3.3.3",
|
||||
"fuse.js": "7.1.0",
|
||||
"gitignore-parser": "0.0.2",
|
||||
"idb": "8.0.3",
|
||||
"js-yaml": "^4.1.1",
|
||||
@@ -2731,15 +2730,6 @@
|
||||
"node": "^8.16.0 || ^10.6.0 || >=11.0.0"
|
||||
}
|
||||
},
|
||||
"node_modules/fuse.js": {
|
||||
"version": "7.1.0",
|
||||
"resolved": "https://registry.npmjs.org/fuse.js/-/fuse.js-7.1.0.tgz",
|
||||
"integrity": "sha512-trLf4SzuuUxfusZADLINj+dE8clK1frKdmqiJNb1Es75fmI5oY6X2mxLVUciLLjxqw/xr72Dhy+lER6dGd02FQ==",
|
||||
"license": "Apache-2.0",
|
||||
"engines": {
|
||||
"node": ">=10"
|
||||
}
|
||||
},
|
||||
"node_modules/gensync": {
|
||||
"version": "1.0.0-beta.2",
|
||||
"resolved": "https://registry.npmjs.org/gensync/-/gensync-1.0.0-beta.2.tgz",
|
||||
|
||||
@@ -95,7 +95,6 @@
|
||||
"esbuild": "^0.27.3",
|
||||
"fast-diff": "1.3.0",
|
||||
"fast-glob": "^3.3.3",
|
||||
"fuse.js": "7.1.0",
|
||||
"gitignore-parser": "0.0.2",
|
||||
"idb": "8.0.3",
|
||||
"js-yaml": "^4.1.1",
|
||||
|
||||
@@ -3,6 +3,7 @@ An attempt at documenting the changes/new features introduced in each release.
|
||||
## Edge
|
||||
Whenever a commit is pushed to the `main` branch, within ~5 minutes, it will be released as a docker image with the `:v2` tag, and a binary in the [edge release](https://github.com/silverbulletmd/silverbullet/releases/tag/edge). If you want to live on the bleeding edge of SilverBullet goodness (or regression) this is where to do it.
|
||||
|
||||
* Picker fuzzy search: replaced Fuse.js with a custom scorer that supports multi-token queries (e.g. "silv todo" now matches "SilverBullet/TODO"), path-aware ranking, and bounded typo tolerance.
|
||||
* New `relation` indexed object capturing generalized object-to-object relationships: typed edges from frontmatter, inline `[key: value]` attributes, and `#tag` fenced data blocks; untyped mentions; and co-mention edges between refs co-occurring in the same item, nested item, or paragraph. The (now) legacy `link` is reimplemented as a virtual collection on top of `relation` now and should keep acting as before.
|
||||
* **Important**: Run `Space: Reindex` after upgrading (just once) to this version to make linked mentions and other features work again (this should be automatic, but just in case)
|
||||
* Fix: forced space reindex handling
|
||||
|
||||
Reference in New Issue
Block a user