diff --git a/plugs/index/anchor.test.ts b/plugs/index/anchor.test.ts new file mode 100644 index 00000000..5cb317f2 --- /dev/null +++ b/plugs/index/anchor.test.ts @@ -0,0 +1,50 @@ +import { describe, expect, test } from "vitest"; +import { parseMarkdown } from "../../client/markdown_parser/parser.ts"; +import { + cleanAnchor, + collectAnchor, + isValidAnchorName, +} from "./anchor.ts"; +import { renderToText } from "@silverbulletmd/silverbullet/lib/tree"; + +describe("anchor helpers", () => { + test("collectAnchor returns null when no anchor", () => { + const tree = parseMarkdown("Just some text."); + expect(collectAnchor(tree)).toBeNull(); + }); + + test("collectAnchor returns the single anchor", () => { + const tree = parseMarkdown("A line with $pete in it."); + const a = collectAnchor(tree); + expect(a?.name).toBe("pete"); + expect(typeof a?.from).toBe("number"); + expect(typeof a?.to).toBe("number"); + }); + + test("collectAnchor with two anchors returns the first and reports duplicate", () => { + const tree = parseMarkdown("A $first and $second on same line."); + const a = collectAnchor(tree); + expect(a?.name).toBe("first"); + expect(a?.duplicateInHost).toBe(true); + }); + + test("cleanAnchor strips NamedAnchor nodes from a clone", () => { + const tree = parseMarkdown("Hello $toc1 world"); + cleanAnchor(tree); + expect(renderToText(tree).trim()).toBe("Hello world".trim()); + }); + + test("isValidAnchorName", () => { + expect(isValidAnchorName("pete")).toBe(true); + expect(isValidAnchorName("a/b")).toBe(true); + expect(isValidAnchorName("a:b")).toBe(true); + expect(isValidAnchorName("a-b")).toBe(true); + // `.` is excluded so a sentence-ending period is not consumed. + expect(isValidAnchorName("a.b")).toBe(false); + expect(isValidAnchorName("_foo")).toBe(true); + expect(isValidAnchorName("1abc")).toBe(false); + expect(isValidAnchorName("a b")).toBe(false); + expect(isValidAnchorName("a!b")).toBe(false); + expect(isValidAnchorName("")).toBe(false); + }); +}); diff --git a/plugs/index/anchor.ts b/plugs/index/anchor.ts new file mode 100644 index 00000000..7555e38e --- /dev/null +++ b/plugs/index/anchor.ts @@ -0,0 +1,56 @@ +import { + collectNodesOfType, + type ParseTree, + renderToText, + replaceNodesMatching, +} from "@silverbulletmd/silverbullet/lib/tree"; + +const anchorNameRegex = /^[A-Za-z_][A-Za-z0-9_/:-]*$/; + +export function isValidAnchorName(name: string): boolean { + return anchorNameRegex.test(name); +} + +export type CollectedAnchor = { + name: string; + from: number; + to: number; + // True when the host contained more than one NamedAnchor node. + duplicateInHost: boolean; +}; + +/** + * Returns the first NamedAnchor inside `n`, marking `duplicateInHost` + * if more than one was present. Returns null if none. + */ +export function collectAnchor(n: ParseTree): CollectedAnchor | null { + const nodes = collectNodesOfType(n, "NamedAnchor"); + if (nodes.length === 0) { + return null; + } + const first = nodes[0]; + // The NamedAnchor's rendered text is the literal "$name". the leading + // `$` lives in a NamedAnchorMark child node so live-preview can style + // the sigil distinctly. We strip it here for the bare name. + const literal = renderToText(first); + const name = literal.slice(1); + return { + name, + from: first.from!, + to: first.to!, + duplicateInHost: nodes.length > 1, + }; +} + +/** + * Strips NamedAnchor nodes from the (typically cloned) tree, mirroring + * `cleanTags`. Mutates in place. + */ +export function cleanAnchor(n: ParseTree) { + return replaceNodesMatching(n, (node) => { + if (node.type === "NamedAnchor") { + return null; + } + return; + }); +} diff --git a/plugs/index/data.test.ts b/plugs/index/data.test.ts index 1c724977..9c26fe8a 100644 --- a/plugs/index/data.test.ts +++ b/plugs/index/data.test.ts @@ -1,4 +1,4 @@ -import { expect, test } from "vitest"; +import { describe, expect, test } from "vitest"; import { parseMarkdown } from "../../client/markdown_parser/parser.ts"; import { createMockSystem } from "../../plug-api/system_mock.ts"; import type { @@ -22,6 +22,25 @@ age: 101 \`\`\` `.trim(); +const defaultPageMeta: PageMeta = { + ref: "Page", + name: "Page", + tag: "page", + created: "", + lastModified: "", + perm: "rw", +}; + +async function indexDataForTest( + markdown: string, + pageName = "Page", +): Promise[]> { + const meta: PageMeta = { ...defaultPageMeta, ref: pageName, name: pageName }; + const tree = parseMarkdown(markdown); + const frontmatter = extractFrontMatter(tree); + return indexData(meta, frontmatter, tree); +} + test("Test indexers", async () => { createMockSystem(); const tree = parseMarkdown(testPage); @@ -56,3 +75,36 @@ test("Test indexers", async () => { expect(datas[1].name).toEqual("Hank"); expect(datas[1].age).toEqual(101); }); + +describe("$ref anchor in fenced data blocks", () => { + test("$ref field becomes ref and is stripped from the object", async () => { + createMockSystem(); + const md = `\`\`\`#person\nname: Pete\n$ref: pete\n\`\`\``; + const results = await indexDataForTest(md); + const person = results.find((o) => o.tag === "person")!; + expect(person).toBeTruthy(); + expect(person.tag).toBe("person"); + expect(person.ref).toBe("pete"); + expect(person.name).toBe("Pete"); + expect("$ref" in person).toBe(false); + }); + + test("regression: block without $ref retains Page@docStart ref", async () => { + createMockSystem(); + const md = `\`\`\`#person\nname: Alice\n\`\`\``; + const results = await indexDataForTest(md, "Page"); + const person = results.find((o) => o.tag === "person")!; + expect(person).toBeTruthy(); + expect(person.ref).toMatch(/^Page@\d+$/); + }); + + test("invalid $ref (digit-leading) is ignored; ref falls back to Page@docStart", async () => { + createMockSystem(); + const md = `\`\`\`#person\nname: Bob\n$ref: 1bad\n\`\`\``; + const results = await indexDataForTest(md, "Page"); + const person = results.find((o) => o.tag === "person")!; + expect(person).toBeTruthy(); + expect(person.ref).toMatch(/^Page@\d+$/); + expect("$ref" in person).toBe(false); + }); +}); diff --git a/plugs/index/data.ts b/plugs/index/data.ts index 515dac1d..5b10b89f 100644 --- a/plugs/index/data.ts +++ b/plugs/index/data.ts @@ -11,6 +11,7 @@ import type { ObjectValue, PageMeta, } from "@silverbulletmd/silverbullet/type/index"; +import { isValidAnchorName } from "./anchor.ts"; type DataObject = ObjectValue< { @@ -56,8 +57,21 @@ export function indexData( cursor += docs[i].length; continue; } + // Extract $ref anchor from the YAML doc. Lint surfaces duplicate + // names; the indexer accepts the first valid value here. + let anchorName: string | undefined; + if (typeof doc === "object") { + const d = doc as Record; + if ("$ref" in d) { + const candidate = d.$ref; + if (typeof candidate === "string" && isValidAnchorName(candidate)) { + anchorName = candidate; + } + delete d.$ref; + } + } const dataObj = { - ref: `${pageMeta.name}@${docStart}`, + ref: anchorName ?? `${pageMeta.name}@${docStart}`, tag: dataType, itags: ["data"], pos: docStart, diff --git a/plugs/index/header.test.ts b/plugs/index/header.test.ts index 1e39d7ac..46cdcafd 100644 --- a/plugs/index/header.test.ts +++ b/plugs/index/header.test.ts @@ -34,3 +34,40 @@ test("Test header indexing", async () => { expect(headers[1].name).toEqual("Header 1.1"); expect(headers[1].level).toEqual(2); }); + +function makePageMeta(name = "test"): PageMeta { + return { + ref: name, + name, + tag: "page", + created: "", + lastModified: "", + perm: "rw", + }; +} + +test("header with $anchor uses anchor as ref", async () => { + createMockSystem(); + const src = `# Some heading $sec1`; + const tree = parseMarkdown(src); + const frontmatter = await extractFrontMatter(tree); + const headers = await indexHeaders(makePageMeta(), frontmatter, tree); + + expect(headers.length).toEqual(1); + expect(headers[0].ref).toBe("sec1"); + // $sec1 must not appear in name or text + expect(headers[0].name).not.toContain("$sec1"); + expect(headers[0].text).not.toContain("$sec1"); + expect(headers[0].name.trim()).toBe("Some heading"); +}); + +test("header without anchor keeps Page@pos ref shape", async () => { + createMockSystem(); + const src = `# Plain header`; + const tree = parseMarkdown(src); + const frontmatter = await extractFrontMatter(tree); + const headers = await indexHeaders(makePageMeta("MyPage"), frontmatter, tree); + + expect(headers.length).toEqual(1); + expect(headers[0].ref).toMatch(/^MyPage@\d+$/); +}); diff --git a/plugs/index/header.ts b/plugs/index/header.ts index be453936..8c66870a 100644 --- a/plugs/index/header.ts +++ b/plugs/index/header.ts @@ -17,6 +17,7 @@ import type { CompleteEvent } from "@silverbulletmd/silverbullet/type/client"; import type { FrontMatter } from "./frontmatter.ts"; import { cleanAttributes, collectAttributes } from "./attribute.ts"; import { cleanTags, collectTags } from "./tags.ts"; +import { cleanAnchor, collectAnchor } from "./anchor.ts"; type HeaderObject = ObjectValue< { @@ -40,16 +41,20 @@ export function indexHeaders( (t) => !!t.type?.startsWith("ATXHeading"), )) { const level = +n.type!.substring("ATXHeading".length); - const name = renderToText(n).slice(level + 1); const tags = collectTags(n); + const anchor = collectAnchor(n); const attributes = collectAttributes(n); const nClone = cloneTree(n); cleanTags(nClone); cleanAttributes(nClone); - const text = renderToText(nClone).slice(level + 1); + cleanAnchor(nClone); // strip $anchor token from clone + // Compute name from the cleaned tree so $anchor is stripped from both name and text. + const name = renderToText(nClone).slice(level + 1); + const text = name; headers.push({ - ref: `${pageMeta.name}@${n.from}`, + // First anchor wins; T12 lint flags duplicates. + ref: anchor ? anchor.name : `${pageMeta.name}@${n.from}`, tag: "header", tags: [...tags], level, diff --git a/plugs/index/item.test.ts b/plugs/index/item.test.ts index 0c09b7dd..f64aba13 100644 --- a/plugs/index/item.test.ts +++ b/plugs/index/item.test.ts @@ -73,6 +73,53 @@ test("Test item indexing", async () => { expect(new Set(items[8].ilinks)).toEqual(new Set(["link", "link 2"])); }); +// --- Anchor tests --- + +async function indexItemsForTest( + md: string, + pageName = "TestPage", +): Promise extends Promise ? T : never> { + createMockSystem(); + const tree = parseMarkdown(md); + const frontmatter = extractFrontMatter(tree); + const pageMeta: PageMeta = { + ref: pageName, + name: pageName, + tag: "page", + created: "", + lastModified: "", + perm: "rw", + }; + return indexItems(pageMeta, frontmatter, tree); +} + +test("list item with $anchor uses anchor as ref", async () => { + const items = await indexItemsForTest("- Item with anchor $foo here\n"); + const item = items.find((i) => i.tag === "item")!; + expect(item.ref).toBe("foo"); + expect(item.name).not.toContain("$foo"); +}); + +test("task with $anchor uses anchor as ref", async () => { + const items = await indexItemsForTest("- [ ] Task body $bar\n"); + const task = items.find((i) => i.tag === "task")!; + expect(task.tag).toBe("task"); + expect(task.ref).toBe("bar"); + expect(task.name).not.toContain("$bar"); +}); + +test("item without anchor keeps Page@pos ref", async () => { + const items = await indexItemsForTest("- Plain item #tag\n", "MyPage"); + const item = items.find((i) => i.tag === "item")!; + expect(item.ref).toMatch(/^MyPage@\d+$/); +}); + +test("task without anchor keeps Page@pos ref", async () => { + const items = await indexItemsForTest("- [ ] Plain task #tag\n", "MyPage"); + const task = items.find((i) => i.tag === "task")!; + expect(task.ref).toMatch(/^MyPage@\d+$/); +}); + // Regression test for https://github.com/silverbulletmd/silverbullet/issues/1932 // and https://github.com/silverbulletmd/silverbullet/issues/1903 const nestedListMd = ` diff --git a/plugs/index/item.ts b/plugs/index/item.ts index 9c8a89e9..a0bdc4f1 100644 --- a/plugs/index/item.ts +++ b/plugs/index/item.ts @@ -6,6 +6,7 @@ import { traverseTree, } from "@silverbulletmd/silverbullet/lib/tree"; import { cleanTags, collectTags, updateITags } from "./tags.ts"; +import { cleanAnchor, collectAnchor } from "./anchor.ts"; import type { FrontMatter } from "./frontmatter.ts"; import type { ObjectValue, @@ -136,6 +137,10 @@ export function extractItemFromNode( nameNode = { type: "Paragraph", children: taskNode.children!.slice(1) }; } + // Collect anchor from nameNode only (not the whole itemNode) so that + // child sublist anchors don't bleed into the parent item's ref. + const anchor = nameNode ? collectAnchor(nameNode) : null; + // Now let's extract tags and attributes const tags = collectTags(itemNode); const attributes = collectAttributes(itemNode); @@ -147,11 +152,17 @@ export function extractItemFromNode( const nameNodeClone = cloneTree(nameNode); cleanTags(nameNodeClone); cleanAttributes(nameNodeClone); + cleanAnchor(nameNodeClone); item.name = renderToText(nameNodeClone).trim(); } else { item.name = item.text; } + // First anchor wins; T12 lint flags duplicates. + if (anchor) { + item.ref = anchor.name; + } + if (tags.length > 0) { item.tags = tags; } diff --git a/plugs/index/paragraph.test.ts b/plugs/index/paragraph.test.ts index e208f837..21ea1efc 100644 --- a/plugs/index/paragraph.test.ts +++ b/plugs/index/paragraph.test.ts @@ -5,6 +5,25 @@ import type { PageMeta } from "@silverbulletmd/silverbullet/type/index"; import { extractFrontMatter } from "./frontmatter.ts"; import { indexParagraphs } from "./paragraph.ts"; +async function indexParagraphsForTest( + text: string, + pageName = "TestPage", +) { + const { config } = createMockSystem(); + config.set("index.paragraph.all", false); + const tree = parseMarkdown(text); + const frontmatter = extractFrontMatter(tree); + const pageMeta: PageMeta = { + ref: pageName, + name: pageName, + tag: "page", + created: "", + lastModified: "", + perm: "rw", + }; + return indexParagraphs(pageMeta, frontmatter, tree); +} + const testPage = ` #tag-only @@ -45,3 +64,28 @@ test("Test paragraph indexing", async () => { paragraphs = await indexParagraphs(pageMeta, frontmatter, tree); expect(paragraphs.length).toEqual(1); }); + +test("paragraph with $anchor uses anchor as ref", async () => { + const objects = await indexParagraphsForTest( + `Paragraph with anchor $pete here. #marker`, + ); + const para = objects.find((o) => o.tag === "paragraph")!; + expect(para.ref).toBe("pete"); + expect(para.text).not.toContain("$pete"); +}); + +test("standalone-line $anchor still indexes the paragraph", async () => { + const objects = await indexParagraphsForTest(`$standalone\n`); + const para = objects.find((o) => o.tag === "paragraph"); + expect(para).toBeDefined(); + expect(para!.ref).toBe("standalone"); +}); + +test("paragraph without anchor keeps Page@pos ref", async () => { + const objects = await indexParagraphsForTest( + `Tagged paragraph without anchor #foo`, + "MyPage", + ); + const para = objects.find((o) => o.tag === "paragraph")!; + expect(para.ref).toMatch(/^MyPage@\d+$/); +}); diff --git a/plugs/index/paragraph.ts b/plugs/index/paragraph.ts index a27e62f2..86237515 100644 --- a/plugs/index/paragraph.ts +++ b/plugs/index/paragraph.ts @@ -6,6 +6,7 @@ import { traverseTree, } from "@silverbulletmd/silverbullet/lib/tree"; import { cleanTags, collectTags, updateITags } from "./tags.ts"; +import { cleanAnchor, collectAnchor } from "./anchor.ts"; import type { FrontMatter } from "./frontmatter.ts"; import type { ObjectValue, @@ -46,9 +47,10 @@ export async function indexParagraphs( // Collect tags const tags = collectTags(p); + const anchor = collectAnchor(p); - if (tags.length === 0 && !shouldIndexAll) { - // Don't index paragraphs without a hashtag + if (tags.length === 0 && !anchor && !shouldIndexAll) { + // Don't index paragraphs without a hashtag or anchor return true; } @@ -59,18 +61,21 @@ export async function indexParagraphs( const pClone = cloneTree(p); cleanTags(pClone); cleanAttributes(pClone); - const text = renderToText(pClone); + cleanAnchor(pClone); + const cleanedText = renderToText(pClone); - if (!text.trim()) { - // Empty paragraph, just tags and attributes maybe + if (!cleanedText.trim() && !anchor) { + // Empty paragraph, just tags, attributes, and/or anchor maybe return true; } const pos = p.from!; const paragraph: ParagraphObject = { tag: "paragraph", - ref: `${pageMeta.name}@${pos}`, - text: fullText, + // "First anchor wins" anchor.duplicateInHost is intentionally ignored + // here. lint flags multi-anchor hosts at edit time. + ref: anchor ? anchor.name : `${pageMeta.name}@${pos}`, + text: anchor ? cleanedText : fullText, page: pageMeta.name, pos, range: [p.from!, p.to!],