From 667af69594eda4860add9e9c7972c7e2ad5d27f2 Mon Sep 17 00:00:00 2001 From: Zef Hemel Date: Thu, 21 May 2026 16:50:17 +0200 Subject: [PATCH] Add index-stats script --- package.json | 1 + plugs/index/index.bench.ts | 95 +++++++------------------ plugs/index/index_stats.ts | 137 +++++++++++++++++++++++++++++++++++++ plugs/index/test_corpus.ts | 44 ++++++++++++ 4 files changed, 208 insertions(+), 69 deletions(-) create mode 100644 plugs/index/index_stats.ts create mode 100644 plugs/index/test_corpus.ts diff --git a/package.json b/package.json index 4cf069e2..b470daaa 100644 --- a/package.json +++ b/package.json @@ -61,6 +61,7 @@ "fmt": "biome format --write .", "fmt:check": "biome format .", "bench": "vitest bench", + "index-stats": "tsx plugs/index/index_stats.ts", "test:e2e": "npx playwright test", "test:e2e:headed": "npx playwright test --headed" }, diff --git a/plugs/index/index.bench.ts b/plugs/index/index.bench.ts index ab56af89..8bdd1c99 100644 --- a/plugs/index/index.bench.ts +++ b/plugs/index/index.bench.ts @@ -1,80 +1,47 @@ import { bench, describe } from "vitest"; -import { readdirSync, readFileSync, statSync } from "node:fs"; -import { join } from "node:path"; -import { fileURLToPath } from "node:url"; import { parseMarkdown } from "../../client/markdown_parser/parser.ts"; import { createMockSystem } from "../../plug-api/system_mock.ts"; -import { extractFrontMatter } from "./frontmatter.ts"; +import { extractFrontMatter, type FrontMatter } from "./frontmatter.ts"; import { indexPage as pageIndexPage } from "./page.ts"; import { indexData } from "./data.ts"; import { indexItems } from "./item.ts"; import { indexHeaders } from "./header.ts"; import { indexParagraphs } from "./paragraph.ts"; -import { indexLinks } from "./link.ts"; +import { indexRelations } from "./relation.ts"; import { indexTables } from "./table.ts"; import { indexSpaceLua } from "./space_lua.ts"; import { indexSpaceStyle } from "./space_style.ts"; import { indexTags } from "./tags.ts"; +import { allIndexers } from "./indexer.ts"; +import { + type CorpusPage, + loadMarkdownFiles, + stubPageMeta, + websiteDir, +} from "./test_corpus.ts"; import type { PageMeta } from "@silverbulletmd/silverbullet/type/index"; import type { ParseTree } from "@silverbulletmd/silverbullet/lib/tree"; -import type { FrontMatter } from "./frontmatter.ts"; -import { allIndexers } from "./indexer.ts"; - -// --- Load all website markdown files --- -const __dirname = fileURLToPath(new URL(".", import.meta.url)); -const websiteDir = join(__dirname, "../../website"); - -interface PageData { - name: string; - text: string; -} - -function loadMarkdownFiles(dir: string, base = ""): PageData[] { - const pages: PageData[] = []; - for (const entry of readdirSync(dir)) { - const fullPath = join(dir, entry); - const relativePath = base ? `${base}/${entry}` : entry; - const stat = statSync(fullPath); - if (stat.isDirectory()) { - pages.push(...loadMarkdownFiles(fullPath, relativePath)); - } else if (entry.endsWith(".md")) { - pages.push({ - name: relativePath.replace(/\.md$/, ""), - text: readFileSync(fullPath, "utf-8"), - }); - } - } - return pages; -} const pages = loadMarkdownFiles(websiteDir); -// Pre-parse trees for indexer-only benchmarks -interface ParsedPage extends PageData { +// Pre-parse trees for indexer-only benchmarks. +type ParsedPage = CorpusPage & { tree: ParseTree; frontmatter: FrontMatter; pageMeta: PageMeta; -} +}; -function buildParsedPages(): ParsedPage[] { - return pages.map((p) => { - const tree = parseMarkdown(p.text); - const frontmatter = extractFrontMatter(tree); - const pageMeta: PageMeta = { - ref: p.name, - name: p.name, - tag: "page", - created: "", - lastModified: "", - perm: "rw", - }; - return { ...p, tree, frontmatter, pageMeta }; - }); -} - -// Setup mock system (registers syscalls globally) createMockSystem(); -const parsedPages = buildParsedPages(); + +const parsedPages: ParsedPage[] = pages.map((p) => { + const tree = parseMarkdown(p.text); + return { + ...p, + tree, + frontmatter: extractFrontMatter(tree), + pageMeta: stubPageMeta(p.name), + }; +}); // --- Benchmarks --- @@ -93,12 +60,11 @@ describe("Page Indexing Benchmarks", () => { } }); - bench("indexLinks (all pages)", async () => { + bench("indexRelations (all pages)", async () => { for (const p of parsedPages) { - // Re-parse to get a fresh tree (extractFrontMatter adds parent pointers) const tree = parseMarkdown(p.text); const fm = extractFrontMatter(tree); - await indexLinks(p.pageMeta, fm, tree, p.text); + await indexRelations(p.pageMeta, fm, tree, p.text); } }); @@ -178,18 +144,9 @@ describe("Page Indexing Benchmarks", () => { for (const p of pages) { const tree = parseMarkdown(p.text); const frontmatter = extractFrontMatter(tree); - const pageMeta: PageMeta = { - ref: p.name, - name: p.name, - tag: "page", - created: "", - lastModified: "", - perm: "rw", - }; + const meta = stubPageMeta(p.name); await Promise.all( - allIndexers.map((indexer) => - indexer(pageMeta, frontmatter, tree, p.text) - ), + allIndexers.map((indexer) => indexer(meta, frontmatter, tree, p.text)), ); } }); diff --git a/plugs/index/index_stats.ts b/plugs/index/index_stats.ts new file mode 100644 index 00000000..6665b404 --- /dev/null +++ b/plugs/index/index_stats.ts @@ -0,0 +1,137 @@ +// Run with `npm run index-stats`. +// +// Indexes every Markdown page under `silverbullet/website/` and reports +// total indexer wall time (mean over a few runs) and object counts +// grouped by tag, with a sub-breakdown of relation records by kind. +// +// Parsing is hoisted out of the timing loop so the reported time +// reflects the indexer pipeline itself, not Lezer parsing. + +import { parseMarkdown } from "../../client/markdown_parser/parser.ts"; +import { createMockSystem } from "../../plug-api/system_mock.ts"; +import { extractFrontMatter, type FrontMatter } from "./frontmatter.ts"; +import { allIndexers } from "./indexer.ts"; +import { + type CorpusPage, + loadMarkdownFiles, + stubPageMeta, + websiteDir, +} from "./test_corpus.ts"; +import type { + ObjectValue, + PageMeta, +} from "@silverbulletmd/silverbullet/type/index"; +import type { ParseTree } from "@silverbulletmd/silverbullet/lib/tree"; + +createMockSystem(); + +// Suppress link.ts's per-broken-link `console.info` noise for the +// duration of the script. (Global mute is fine here — the process +// exits when main() returns.) +console.info = () => {}; + +type ParsedPage = CorpusPage & { + tree: ParseTree; + frontmatter: FrontMatter; + meta: PageMeta; +}; + +type RunResult = { + elapsedMs: number; + total: number; + byTag: Record; + relationByKind: Record; +}; + +async function runOnce(parsed: ParsedPage[]): Promise { + const start = performance.now(); + let total = 0; + const byTag: Record = {}; + const relationByKind: Record = {}; + for (const p of parsed) { + const results = await Promise.all( + allIndexers.map((idx) => idx(p.meta, p.frontmatter, p.tree, p.text)), + ); + for (const arr of results) { + for (const o of arr as ObjectValue[]) { + total++; + byTag[o.tag] = (byTag[o.tag] ?? 0) + 1; + if (o.tag === "relation") { + const k = (o as any).kind ?? "?"; + relationByKind[k] = (relationByKind[k] ?? 0) + 1; + } + } + } + } + return { + elapsedMs: performance.now() - start, + total, + byTag, + relationByKind, + }; +} + +async function main() { + const pages = loadMarkdownFiles(websiteDir); + console.log(`Loaded ${pages.length} pages from ${websiteDir}`); + + const parsed: ParsedPage[] = pages.map((p) => { + const tree = parseMarkdown(p.text); + return { + ...p, + tree, + frontmatter: extractFrontMatter(tree), + meta: stubPageMeta(p.name), + }; + }); + + const RUNS = 5; + let sum = 0; + let last: RunResult | undefined; + for (let i = 0; i < RUNS; i++) { + // Re-parse outside the timed region: extractFrontMatter mutates + // the tree (adds parent pointers) and the indexer pipeline uses + // addParentPointers, so each run wants a fresh tree. We don't want + // to measure that parsing in the indexer timing. + for (const p of parsed) { + p.tree = parseMarkdown(p.text); + p.frontmatter = extractFrontMatter(p.tree); + } + last = await runOnce(parsed); + sum += last.elapsedMs; + } + const r = last!; + + console.log(`\n=== TIMING ===`); + console.log(`Mean over ${RUNS} runs: ${(sum / RUNS).toFixed(1)}ms`); + + console.log(`\n=== INDEX SIZE ===`); + console.log(`Pages indexed: ${pages.length}`); + console.log(`Total objects: ${r.total}`); + console.log(`Per page mean: ${(r.total / pages.length).toFixed(1)}`); + + console.log(`\n=== OBJECTS BY TAG ===`); + for ( + const [tag, count] of Object.entries(r.byTag).sort((a, b) => b[1] - a[1]) + ) { + const pct = ((count / r.total) * 100).toFixed(1); + console.log(` ${tag.padEnd(20)} ${String(count).padStart(5)} (${pct}%)`); + } + + const relTotal = r.byTag["relation"] ?? 0; + if (relTotal > 0) { + console.log(`\n=== RELATION RECORDS BY KIND ===`); + for ( + const [kind, count] of Object.entries(r.relationByKind).sort( + (a, b) => b[1] - a[1], + ) + ) { + const pct = ((count / relTotal) * 100).toFixed(1); + console.log( + ` ${kind.padEnd(14)} ${String(count).padStart(5)} (${pct}%)`, + ); + } + } +} + +main(); diff --git a/plugs/index/test_corpus.ts b/plugs/index/test_corpus.ts new file mode 100644 index 00000000..8c9e78c7 --- /dev/null +++ b/plugs/index/test_corpus.ts @@ -0,0 +1,44 @@ +// Shared test/benchmark helpers for loading the silverbullet/website +// markdown corpus and synthesizing PageMeta stubs. + +import { readdirSync, readFileSync, statSync } from "node:fs"; +import { join } from "node:path"; +import { fileURLToPath } from "node:url"; +import type { PageMeta } from "@silverbulletmd/silverbullet/type/index"; + +export type CorpusPage = { + name: string; + text: string; +}; + +export const websiteDir = fileURLToPath( + new URL("../../website", import.meta.url), +); + +export function loadMarkdownFiles(dir: string, base = ""): CorpusPage[] { + const out: CorpusPage[] = []; + for (const entry of readdirSync(dir)) { + const full = join(dir, entry); + const rel = base ? `${base}/${entry}` : entry; + if (statSync(full).isDirectory()) { + out.push(...loadMarkdownFiles(full, rel)); + } else if (entry.endsWith(".md")) { + out.push({ + name: rel.replace(/\.md$/, ""), + text: readFileSync(full, "utf-8"), + }); + } + } + return out; +} + +export function stubPageMeta(name: string): PageMeta { + return { + ref: name, + name, + tag: "page", + created: "", + lastModified: "", + perm: "rw", + }; +}