anchor extraction and per-type indexer wiring

Paragraphs, list items, tasks, headers, and fenced-data blocks now adopt the anchor name as the host's `ref` field when present. For fenced-data blocks the anchor comes from a YAML  field named `$ref:` (the markdown anchor syntax doesn't apply inside YAML).
This commit is contained in:
Zef Hemel
2026-04-30 17:57:10 +02:00
parent 3670016408
commit 05d3ba9333
10 changed files with 333 additions and 12 deletions
+50
View File
@@ -0,0 +1,50 @@
import { describe, expect, test } from "vitest";
import { parseMarkdown } from "../../client/markdown_parser/parser.ts";
import {
cleanAnchor,
collectAnchor,
isValidAnchorName,
} from "./anchor.ts";
import { renderToText } from "@silverbulletmd/silverbullet/lib/tree";
describe("anchor helpers", () => {
test("collectAnchor returns null when no anchor", () => {
const tree = parseMarkdown("Just some text.");
expect(collectAnchor(tree)).toBeNull();
});
test("collectAnchor returns the single anchor", () => {
const tree = parseMarkdown("A line with $pete in it.");
const a = collectAnchor(tree);
expect(a?.name).toBe("pete");
expect(typeof a?.from).toBe("number");
expect(typeof a?.to).toBe("number");
});
test("collectAnchor with two anchors returns the first and reports duplicate", () => {
const tree = parseMarkdown("A $first and $second on same line.");
const a = collectAnchor(tree);
expect(a?.name).toBe("first");
expect(a?.duplicateInHost).toBe(true);
});
test("cleanAnchor strips NamedAnchor nodes from a clone", () => {
const tree = parseMarkdown("Hello $toc1 world");
cleanAnchor(tree);
expect(renderToText(tree).trim()).toBe("Hello world".trim());
});
test("isValidAnchorName", () => {
expect(isValidAnchorName("pete")).toBe(true);
expect(isValidAnchorName("a/b")).toBe(true);
expect(isValidAnchorName("a:b")).toBe(true);
expect(isValidAnchorName("a-b")).toBe(true);
// `.` is excluded so a sentence-ending period is not consumed.
expect(isValidAnchorName("a.b")).toBe(false);
expect(isValidAnchorName("_foo")).toBe(true);
expect(isValidAnchorName("1abc")).toBe(false);
expect(isValidAnchorName("a b")).toBe(false);
expect(isValidAnchorName("a!b")).toBe(false);
expect(isValidAnchorName("")).toBe(false);
});
});
+56
View File
@@ -0,0 +1,56 @@
import {
collectNodesOfType,
type ParseTree,
renderToText,
replaceNodesMatching,
} from "@silverbulletmd/silverbullet/lib/tree";
const anchorNameRegex = /^[A-Za-z_][A-Za-z0-9_/:-]*$/;
export function isValidAnchorName(name: string): boolean {
return anchorNameRegex.test(name);
}
export type CollectedAnchor = {
name: string;
from: number;
to: number;
// True when the host contained more than one NamedAnchor node.
duplicateInHost: boolean;
};
/**
* Returns the first NamedAnchor inside `n`, marking `duplicateInHost`
* if more than one was present. Returns null if none.
*/
export function collectAnchor(n: ParseTree): CollectedAnchor | null {
const nodes = collectNodesOfType(n, "NamedAnchor");
if (nodes.length === 0) {
return null;
}
const first = nodes[0];
// The NamedAnchor's rendered text is the literal "$name". the leading
// `$` lives in a NamedAnchorMark child node so live-preview can style
// the sigil distinctly. We strip it here for the bare name.
const literal = renderToText(first);
const name = literal.slice(1);
return {
name,
from: first.from!,
to: first.to!,
duplicateInHost: nodes.length > 1,
};
}
/**
* Strips NamedAnchor nodes from the (typically cloned) tree, mirroring
* `cleanTags`. Mutates in place.
*/
export function cleanAnchor(n: ParseTree) {
return replaceNodesMatching(n, (node) => {
if (node.type === "NamedAnchor") {
return null;
}
return;
});
}
+53 -1
View File
@@ -1,4 +1,4 @@
import { expect, test } from "vitest";
import { describe, expect, test } from "vitest";
import { parseMarkdown } from "../../client/markdown_parser/parser.ts";
import { createMockSystem } from "../../plug-api/system_mock.ts";
import type {
@@ -22,6 +22,25 @@ age: 101
\`\`\`
`.trim();
const defaultPageMeta: PageMeta = {
ref: "Page",
name: "Page",
tag: "page",
created: "",
lastModified: "",
perm: "rw",
};
async function indexDataForTest(
markdown: string,
pageName = "Page",
): Promise<ObjectValue<any>[]> {
const meta: PageMeta = { ...defaultPageMeta, ref: pageName, name: pageName };
const tree = parseMarkdown(markdown);
const frontmatter = extractFrontMatter(tree);
return indexData(meta, frontmatter, tree);
}
test("Test indexers", async () => {
createMockSystem();
const tree = parseMarkdown(testPage);
@@ -56,3 +75,36 @@ test("Test indexers", async () => {
expect(datas[1].name).toEqual("Hank");
expect(datas[1].age).toEqual(101);
});
describe("$ref anchor in fenced data blocks", () => {
test("$ref field becomes ref and is stripped from the object", async () => {
createMockSystem();
const md = `\`\`\`#person\nname: Pete\n$ref: pete\n\`\`\``;
const results = await indexDataForTest(md);
const person = results.find((o) => o.tag === "person")!;
expect(person).toBeTruthy();
expect(person.tag).toBe("person");
expect(person.ref).toBe("pete");
expect(person.name).toBe("Pete");
expect("$ref" in person).toBe(false);
});
test("regression: block without $ref retains Page@docStart ref", async () => {
createMockSystem();
const md = `\`\`\`#person\nname: Alice\n\`\`\``;
const results = await indexDataForTest(md, "Page");
const person = results.find((o) => o.tag === "person")!;
expect(person).toBeTruthy();
expect(person.ref).toMatch(/^Page@\d+$/);
});
test("invalid $ref (digit-leading) is ignored; ref falls back to Page@docStart", async () => {
createMockSystem();
const md = `\`\`\`#person\nname: Bob\n$ref: 1bad\n\`\`\``;
const results = await indexDataForTest(md, "Page");
const person = results.find((o) => o.tag === "person")!;
expect(person).toBeTruthy();
expect(person.ref).toMatch(/^Page@\d+$/);
expect("$ref" in person).toBe(false);
});
});
+15 -1
View File
@@ -11,6 +11,7 @@ import type {
ObjectValue,
PageMeta,
} from "@silverbulletmd/silverbullet/type/index";
import { isValidAnchorName } from "./anchor.ts";
type DataObject = ObjectValue<
{
@@ -56,8 +57,21 @@ export function indexData(
cursor += docs[i].length;
continue;
}
// Extract $ref anchor from the YAML doc. Lint surfaces duplicate
// names; the indexer accepts the first valid value here.
let anchorName: string | undefined;
if (typeof doc === "object") {
const d = doc as Record<string, unknown>;
if ("$ref" in d) {
const candidate = d.$ref;
if (typeof candidate === "string" && isValidAnchorName(candidate)) {
anchorName = candidate;
}
delete d.$ref;
}
}
const dataObj = {
ref: `${pageMeta.name}@${docStart}`,
ref: anchorName ?? `${pageMeta.name}@${docStart}`,
tag: dataType,
itags: ["data"],
pos: docStart,
+37
View File
@@ -34,3 +34,40 @@ test("Test header indexing", async () => {
expect(headers[1].name).toEqual("Header 1.1");
expect(headers[1].level).toEqual(2);
});
function makePageMeta(name = "test"): PageMeta {
return {
ref: name,
name,
tag: "page",
created: "",
lastModified: "",
perm: "rw",
};
}
test("header with $anchor uses anchor as ref", async () => {
createMockSystem();
const src = `# Some heading $sec1`;
const tree = parseMarkdown(src);
const frontmatter = await extractFrontMatter(tree);
const headers = await indexHeaders(makePageMeta(), frontmatter, tree);
expect(headers.length).toEqual(1);
expect(headers[0].ref).toBe("sec1");
// $sec1 must not appear in name or text
expect(headers[0].name).not.toContain("$sec1");
expect(headers[0].text).not.toContain("$sec1");
expect(headers[0].name.trim()).toBe("Some heading");
});
test("header without anchor keeps Page@pos ref shape", async () => {
createMockSystem();
const src = `# Plain header`;
const tree = parseMarkdown(src);
const frontmatter = await extractFrontMatter(tree);
const headers = await indexHeaders(makePageMeta("MyPage"), frontmatter, tree);
expect(headers.length).toEqual(1);
expect(headers[0].ref).toMatch(/^MyPage@\d+$/);
});
+8 -3
View File
@@ -17,6 +17,7 @@ import type { CompleteEvent } from "@silverbulletmd/silverbullet/type/client";
import type { FrontMatter } from "./frontmatter.ts";
import { cleanAttributes, collectAttributes } from "./attribute.ts";
import { cleanTags, collectTags } from "./tags.ts";
import { cleanAnchor, collectAnchor } from "./anchor.ts";
type HeaderObject = ObjectValue<
{
@@ -40,16 +41,20 @@ export function indexHeaders(
(t) => !!t.type?.startsWith("ATXHeading"),
)) {
const level = +n.type!.substring("ATXHeading".length);
const name = renderToText(n).slice(level + 1);
const tags = collectTags(n);
const anchor = collectAnchor(n);
const attributes = collectAttributes(n);
const nClone = cloneTree(n);
cleanTags(nClone);
cleanAttributes(nClone);
const text = renderToText(nClone).slice(level + 1);
cleanAnchor(nClone); // strip $anchor token from clone
// Compute name from the cleaned tree so $anchor is stripped from both name and text.
const name = renderToText(nClone).slice(level + 1);
const text = name;
headers.push({
ref: `${pageMeta.name}@${n.from}`,
// First anchor wins; T12 lint flags duplicates.
ref: anchor ? anchor.name : `${pageMeta.name}@${n.from}`,
tag: "header",
tags: [...tags],
level,
+47
View File
@@ -73,6 +73,53 @@ test("Test item indexing", async () => {
expect(new Set(items[8].ilinks)).toEqual(new Set(["link", "link 2"]));
});
// --- Anchor tests ---
async function indexItemsForTest(
md: string,
pageName = "TestPage",
): Promise<ReturnType<typeof indexItems> extends Promise<infer T> ? T : never> {
createMockSystem();
const tree = parseMarkdown(md);
const frontmatter = extractFrontMatter(tree);
const pageMeta: PageMeta = {
ref: pageName,
name: pageName,
tag: "page",
created: "",
lastModified: "",
perm: "rw",
};
return indexItems(pageMeta, frontmatter, tree);
}
test("list item with $anchor uses anchor as ref", async () => {
const items = await indexItemsForTest("- Item with anchor $foo here\n");
const item = items.find((i) => i.tag === "item")!;
expect(item.ref).toBe("foo");
expect(item.name).not.toContain("$foo");
});
test("task with $anchor uses anchor as ref", async () => {
const items = await indexItemsForTest("- [ ] Task body $bar\n");
const task = items.find((i) => i.tag === "task")!;
expect(task.tag).toBe("task");
expect(task.ref).toBe("bar");
expect(task.name).not.toContain("$bar");
});
test("item without anchor keeps Page@pos ref", async () => {
const items = await indexItemsForTest("- Plain item #tag\n", "MyPage");
const item = items.find((i) => i.tag === "item")!;
expect(item.ref).toMatch(/^MyPage@\d+$/);
});
test("task without anchor keeps Page@pos ref", async () => {
const items = await indexItemsForTest("- [ ] Plain task #tag\n", "MyPage");
const task = items.find((i) => i.tag === "task")!;
expect(task.ref).toMatch(/^MyPage@\d+$/);
});
// Regression test for https://github.com/silverbulletmd/silverbullet/issues/1932
// and https://github.com/silverbulletmd/silverbullet/issues/1903
const nestedListMd = `
+11
View File
@@ -6,6 +6,7 @@ import {
traverseTree,
} from "@silverbulletmd/silverbullet/lib/tree";
import { cleanTags, collectTags, updateITags } from "./tags.ts";
import { cleanAnchor, collectAnchor } from "./anchor.ts";
import type { FrontMatter } from "./frontmatter.ts";
import type {
ObjectValue,
@@ -136,6 +137,10 @@ export function extractItemFromNode(
nameNode = { type: "Paragraph", children: taskNode.children!.slice(1) };
}
// Collect anchor from nameNode only (not the whole itemNode) so that
// child sublist anchors don't bleed into the parent item's ref.
const anchor = nameNode ? collectAnchor(nameNode) : null;
// Now let's extract tags and attributes
const tags = collectTags(itemNode);
const attributes = collectAttributes(itemNode);
@@ -147,11 +152,17 @@ export function extractItemFromNode(
const nameNodeClone = cloneTree(nameNode);
cleanTags(nameNodeClone);
cleanAttributes(nameNodeClone);
cleanAnchor(nameNodeClone);
item.name = renderToText(nameNodeClone).trim();
} else {
item.name = item.text;
}
// First anchor wins; T12 lint flags duplicates.
if (anchor) {
item.ref = anchor.name;
}
if (tags.length > 0) {
item.tags = tags;
}
+44
View File
@@ -5,6 +5,25 @@ import type { PageMeta } from "@silverbulletmd/silverbullet/type/index";
import { extractFrontMatter } from "./frontmatter.ts";
import { indexParagraphs } from "./paragraph.ts";
async function indexParagraphsForTest(
text: string,
pageName = "TestPage",
) {
const { config } = createMockSystem();
config.set("index.paragraph.all", false);
const tree = parseMarkdown(text);
const frontmatter = extractFrontMatter(tree);
const pageMeta: PageMeta = {
ref: pageName,
name: pageName,
tag: "page",
created: "",
lastModified: "",
perm: "rw",
};
return indexParagraphs(pageMeta, frontmatter, tree);
}
const testPage = `
#tag-only
@@ -45,3 +64,28 @@ test("Test paragraph indexing", async () => {
paragraphs = await indexParagraphs(pageMeta, frontmatter, tree);
expect(paragraphs.length).toEqual(1);
});
test("paragraph with $anchor uses anchor as ref", async () => {
const objects = await indexParagraphsForTest(
`Paragraph with anchor $pete here. #marker`,
);
const para = objects.find((o) => o.tag === "paragraph")!;
expect(para.ref).toBe("pete");
expect(para.text).not.toContain("$pete");
});
test("standalone-line $anchor still indexes the paragraph", async () => {
const objects = await indexParagraphsForTest(`$standalone\n`);
const para = objects.find((o) => o.tag === "paragraph");
expect(para).toBeDefined();
expect(para!.ref).toBe("standalone");
});
test("paragraph without anchor keeps Page@pos ref", async () => {
const objects = await indexParagraphsForTest(
`Tagged paragraph without anchor #foo`,
"MyPage",
);
const para = objects.find((o) => o.tag === "paragraph")!;
expect(para.ref).toMatch(/^MyPage@\d+$/);
});
+12 -7
View File
@@ -6,6 +6,7 @@ import {
traverseTree,
} from "@silverbulletmd/silverbullet/lib/tree";
import { cleanTags, collectTags, updateITags } from "./tags.ts";
import { cleanAnchor, collectAnchor } from "./anchor.ts";
import type { FrontMatter } from "./frontmatter.ts";
import type {
ObjectValue,
@@ -46,9 +47,10 @@ export async function indexParagraphs(
// Collect tags
const tags = collectTags(p);
const anchor = collectAnchor(p);
if (tags.length === 0 && !shouldIndexAll) {
// Don't index paragraphs without a hashtag
if (tags.length === 0 && !anchor && !shouldIndexAll) {
// Don't index paragraphs without a hashtag or anchor
return true;
}
@@ -59,18 +61,21 @@ export async function indexParagraphs(
const pClone = cloneTree(p);
cleanTags(pClone);
cleanAttributes(pClone);
const text = renderToText(pClone);
cleanAnchor(pClone);
const cleanedText = renderToText(pClone);
if (!text.trim()) {
// Empty paragraph, just tags and attributes maybe
if (!cleanedText.trim() && !anchor) {
// Empty paragraph, just tags, attributes, and/or anchor maybe
return true;
}
const pos = p.from!;
const paragraph: ParagraphObject = {
tag: "paragraph",
ref: `${pageMeta.name}@${pos}`,
text: fullText,
// "First anchor wins" anchor.duplicateInHost is intentionally ignored
// here. lint flags multi-anchor hosts at edit time.
ref: anchor ? anchor.name : `${pageMeta.name}@${pos}`,
text: anchor ? cleanedText : fullText,
page: pageMeta.name,
pos,
range: [p.from!, p.to!],