Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions docs-site/package.json
Original file line number Diff line number Diff line change
Expand Up @@ -10,6 +10,7 @@
"build": "pnpm run sync-docs && pnpm run build:llms-txt && docusaurus build",
"sync-docs": "tsx scripts/sync-docs.ts",
"build:llms-txt": "tsx scripts/build-llms-txt.ts",
"test:search-index-reproducibility": "node scripts/test-search-index-reproducibility.cjs",
"validate:frontmatter": "tsx scripts/validate-frontmatter.ts",
"validate:frontmatter:strict": "tsx scripts/validate-frontmatter.ts --strict",
"typecheck": "tsc --noEmit",
Expand Down
96 changes: 85 additions & 11 deletions docs-site/plugins/docusaurus-plugin-search-index/index.js
Original file line number Diff line number Diff line change
Expand Up @@ -8,6 +8,7 @@
const fs = require("fs");
const path = require("path");
const matter = require("gray-matter");
const { createHash } = require("node:crypto");

/** Simple glob matching for exclude patterns */
function matchGlob(glob, filePath) {
Expand Down Expand Up @@ -73,11 +74,17 @@ function extractSections(content) {
let currentContent = [];
let currentLevel = 0;

// Every heading is recorded here, even one immediately followed by another
// heading (empty body) — anchor disambiguation must see every heading a
// real Markdown slugger would, not just the ones with indexable content.
// Whether a section has enough body to actually be indexed is decided by
// the caller, which reads `section.content`.
for (const line of lines) {
const headingMatch = line.match(/^(#{1,6})\s+(.+)/);
if (headingMatch) {
// Save previous section
if (currentContent.length > 0) {
// Save previous section (skip the pre-first-heading preamble, which
// has no heading at all)
if (currentHeading || currentContent.length > 0) {
sections.push({
heading: currentHeading,
level: currentLevel,
Expand All @@ -96,7 +103,7 @@ function extractSections(content) {
}

// Save last section
if (currentContent.length > 0) {
if (currentHeading || currentContent.length > 0) {
sections.push({
heading: currentHeading,
level: currentLevel,
Expand All @@ -107,17 +114,55 @@ function extractSections(content) {
return sections;
}

/**
* Codepoint total order; locale collation differs across Node/ICU builds.
* Plain `<`/`>` on strings compares UTF-16 code units, which diverges from
* code point order for astral characters (surrogate pairs, U+10000+) vs BMP
* characters above the surrogate range (U+E000-U+FFFF): a surrogate pair's
* leading unit (U+D800-U+DBFF) always sorts below those BMP units even when
* its actual code point is numerically larger. Step through code points
* explicitly instead.
*/
function byCodepoint(a, b) {
let i = 0;
let j = 0;
while (i < a.length && j < b.length) {
const ca = a.codePointAt(i);
const cb = b.codePointAt(j);
if (ca !== cb) {
return ca < cb ? -1 : 1;
}
i += ca > 0xffff ? 2 : 1;
j += cb > 0xffff ? 2 : 1;
}
if (i < a.length) {
return 1;
}
if (j < b.length) {
return -1;
}
return 0;
}

/** Stable Algolia/MiniSearch object id for one URL. */
function objectIdForUrl(url) {
return createHash("sha256").update(url).digest("hex");
}

/** Recursively find all markdown files */
function findMarkdownFiles(dir, baseDir = dir) {
function findMarkdownFiles(dir) {
const files = [];
if (!fs.existsSync(dir)) {
return files;
}

for (const entry of fs.readdirSync(dir, { withFileTypes: true })) {
const entries = fs
.readdirSync(dir, { withFileTypes: true })
.sort((a, b) => byCodepoint(a.name, b.name));
for (const entry of entries) {
const fullPath = path.join(dir, entry.name);
if (entry.isDirectory()) {
files.push(...findMarkdownFiles(fullPath, baseDir));
files.push(...findMarkdownFiles(fullPath));
} else if (/\.(md|mdx)$/.test(entry.name)) {
files.push(fullPath);
}
Expand Down Expand Up @@ -177,9 +222,17 @@ module.exports = function searchIndexPlugin(context, options = {}) {
async function generateIndex(docsDir, outDir, isExcluded) {
const files = findMarkdownFiles(docsDir);
const documents = [];
let id = 0;
let skipped = 0;

// URL is derived from file path only (frontmatter `slug:` overrides are
// not consulted), so two files can legitimately compute the same URL —
// e.g. `tutorials.md` and `tutorials/index.md` both landing on
// `/docs/tutorials`. That predates this file and is not fixed here; what
// must hold regardless is that every entry still gets a distinct,
// order-independent objectID, so a repeat occurrence is disambiguated the
// same way a repeated heading anchor is below.
const urlOccurrences = new Map();

for (const filePath of files) {
try {
const relativePath = path.relative(docsDir, filePath).replace(/\\/g, "/");
Expand All @@ -206,6 +259,14 @@ async function generateIndex(docsDir, outDir, isExcluded) {
const url = `/docs/${urlPath}`;
const title = frontmatter.title || path.basename(urlPath) || "Untitled";

const urlOccurrence = urlOccurrences.get(url) ?? 0;
urlOccurrences.set(url, urlOccurrence + 1);
// Only the objectID input is suffixed on a repeat — the visible `url`
// field is untouched, so this changes nothing about what's indexed or
// where a result links, only how its id is derived.
const idSource = (forUrl) =>
urlOccurrence === 0 ? forUrl : `${forUrl}::dup${urlOccurrence}`;

// Extract hierarchy from path
const pathParts = urlPath.split("/");
const lvl0 =
Expand All @@ -216,7 +277,7 @@ async function generateIndex(docsDir, outDir, isExcluded) {
// Add main document entry — index full content for better search recall
const plainContent = stripMarkdown(content);
documents.push({
objectID: String(id++),
objectID: objectIdForUrl(idSource(url)),
title,
url,
content: plainContent.slice(0, 5000),
Expand All @@ -230,19 +291,32 @@ async function generateIndex(docsDir, outDir, isExcluded) {

// Add section entries
const sections = extractSections(content);
const anchorCounts = new Map();
for (const section of sections) {
if (!section.heading) {
continue;
}
const anchor = section.heading
const baseAnchor = section.heading
.toLowerCase()
.replace(/[^\w\s-]/g, "")
.replace(/\s+/g, "-");
const occurrence = anchorCounts.get(baseAnchor) ?? 0;
Comment thread
Tara-ag marked this conversation as resolved.
Comment thread
Tara-ag marked this conversation as resolved.
anchorCounts.set(baseAnchor, occurrence + 1);
const anchor =
occurrence === 0 ? baseAnchor : `${baseAnchor}-${occurrence}`;

// Every heading counts toward anchor disambiguation above (matching
// Docusaurus's own slugger), but a heading with no body text isn't
// worth indexing as a search result.
if (!section.content) {
continue;
}

const sectionUrl = `${url}#${anchor}`;
documents.push({
objectID: String(id++),
objectID: objectIdForUrl(idSource(sectionUrl)),
title: section.heading,
url: `${url}#${anchor}`,
url: sectionUrl,
content: section.content,
hierarchy: {
lvl0,
Expand Down
196 changes: 196 additions & 0 deletions docs-site/scripts/test-search-index-reproducibility.cjs
Original file line number Diff line number Diff line change
@@ -0,0 +1,196 @@
#!/usr/bin/env node
/**
* Determinism exception: this drives the real Docusaurus search-index plugin
* against a fixed temporary corpus while varying the filesystem enumeration
* order. A production docs build cannot reliably force both valid readdir
* orders, so the controlled backend is what makes the regression repeatable.
*
* Breaks caught:
* - removing the traversal sort makes two valid directory orders emit
* different bytes;
* - restoring positional object IDs makes an earlier new document renumber
* every unchanged document after it;
* - two files that compute the same URL (e.g. `tutorials.md` and
* `tutorials/index.md` both landing on `/docs/tutorials` because the
* plugin derives URLs from path, not from a `slug:` frontmatter override)
* must still get distinct objectIDs rather than colliding — a real
* instance of this crashed the docs MCP server's MiniSearch index with
* "duplicate ID" on the full corpus.
*/
const assert = require("node:assert/strict");
const fs = require("node:fs");
const os = require("node:os");
const path = require("node:path");

const createSearchIndexPlugin = require("../plugins/docusaurus-plugin-search-index/index.js");

function writeDoc(docsDir, relativePath, title, heading) {
const filePath = path.join(docsDir, relativePath);
fs.mkdirSync(path.dirname(filePath), { recursive: true });
fs.writeFileSync(
filePath,
`---\ntitle: ${title}\n---\n\n# ${heading}\n\n${title} body.\n`,
);
}

async function generate(docsDir, outDir, reverseTraversal = false) {
const originalReaddirSync = fs.readdirSync;
if (reverseTraversal) {
fs.readdirSync = function reversedReaddirSync(dir, options) {
const entries = originalReaddirSync.call(fs, dir, options);
const resolved = path.resolve(String(dir));
if (
(resolved === docsDir || resolved.startsWith(`${docsDir}${path.sep}`)) &&
Array.isArray(entries)
) {
return [...entries].reverse();
}
return entries;
};
}

try {
const plugin = createSearchIndexPlugin(
{ siteDir: path.dirname(docsDir) },
{ docsDir, exclude: [] },
);
await plugin.postBuild({ outDir });
return fs.readFileSync(path.join(outDir, "search-index.json"), "utf8");
} finally {
fs.readdirSync = originalReaddirSync;
}
}

async function main() {
const root = fs.mkdtempSync(path.join(os.tmpdir(), "neurolink-search-index-"));
const docsDir = path.join(root, "docs");
try {
writeDoc(docsDir, "zeta/second.md", "Second", "Second section");
writeDoc(docsDir, "alpha/first.md", "First", "First section");

const normal = await generate(docsDir, path.join(root, "normal"));
const reversed = await generate(docsDir, path.join(root, "reversed"), true);
const before = JSON.parse(normal);
const secondIdsBefore = before
.filter((entry) => entry.url.startsWith("/docs/zeta/second"))
.map((entry) => entry.objectID);
assert.equal(secondIdsBefore.length, 2, "fixture must index the document and section");

writeDoc(docsDir, "aardvark/new.md", "New", "New section");
const after = JSON.parse(
await generate(docsDir, path.join(root, "after-insert")),
);
const secondIdsAfter = after
.filter((entry) => entry.url.startsWith("/docs/zeta/second"))
.map((entry) => entry.objectID);
assert.deepEqual(
secondIdsAfter,
secondIdsBefore,
"adding an earlier document must not renumber an unchanged document",
);
assert.equal(
reversed,
normal,
"two valid filesystem enumeration orders must emit byte-identical indexes",
);

const repeatedHeadingPath = path.join(docsDir, "duplicates.md");
fs.writeFileSync(
repeatedHeadingPath,
"---\ntitle: Duplicates\n---\n\n# Repeated\n\nFirst.\n\n# Repeated\n\nSecond.\n",
);
const withRepeatedHeadings = JSON.parse(
await generate(docsDir, path.join(root, "repeated-headings")),
);
const ids = withRepeatedHeadings.map((entry) => entry.objectID);
assert.equal(
new Set(ids).size,
ids.length,
"repeated headings must still produce unique objectIDs",
);

// A slug override (or any other path-vs-URL mismatch) can make two
// different files compute the identical URL. writeDoc doesn't emit
// frontmatter slugs, so fake the collision directly: a file and a
// same-named directory's index file both resolve to /docs/collide.
writeDoc(docsDir, "collide.md", "Collide File", "Collide file heading");
writeDoc(docsDir, "collide/index.md", "Collide Index", "Collide index heading");
const withUrlCollision = JSON.parse(
await generate(docsDir, path.join(root, "url-collision")),
);
const collideEntries = withUrlCollision.filter(
(entry) => entry.url === "/docs/collide",
);
assert.equal(
collideEntries.length,
2,
"both files sharing a computed URL must still be indexed",
);
assert.notEqual(
collideEntries[0].objectID,
collideEntries[1].objectID,
"two entries sharing a URL must not collide on objectID",
);
const collisionIds = withUrlCollision.map((entry) => entry.objectID);
assert.equal(
new Set(collisionIds).size,
collisionIds.length,
"a URL collision must not produce any duplicate objectID in the full index",
);

// A heading immediately followed by another heading (no body text
// between them) must still count toward anchor disambiguation, the same
// way Docusaurus's own slugger counts every heading regardless of body.
// The first "Same" has no body and is never indexed as a section, but it
// must still claim the bare "#same" anchor, pushing the second — indexed
// — "Same" to "#same-1".
const emptyHeadingPath = path.join(docsDir, "anchor-heading.md");
fs.writeFileSync(
emptyHeadingPath,
"---\ntitle: AnchorHeading\n---\n\n# Same\n# Same\n\nBody text.\n",
);
const withEmptyHeading = JSON.parse(
await generate(docsDir, path.join(root, "anchor-heading")),
);
const sameSection = withEmptyHeading.find(
(entry) => entry.url.startsWith("/docs/anchor-heading#"),
);
assert.equal(
sameSection?.url,
"/docs/anchor-heading#same-1",
"a heading with no body must still occupy its anchor slot, pushing the next same-text heading to '-1'",
);

// File traversal order must follow true Unicode code point order, not
// UTF-16 code unit order. U+E000 (BMP, Private Use Area) and U+10000
// (astral, encoded as a surrogate pair starting at U+D800) diverge under
// plain `<`/`>`: comparing the surrogate pair's leading unit (0xD800)
// against 0xE000 sorts the astral character first, even though its real
// code point (0x10000 = 65536) is numerically larger than 0xE000 (57344).
const astral = "\u{10000}";
const pua = "\u{E000}";
writeDoc(docsDir, `${astral}-astral.md`, "Astral", "Astral heading");
writeDoc(docsDir, `${pua}-pua.md`, "Pua", "Pua heading");
const withAstralAndPua = JSON.parse(
await generate(docsDir, path.join(root, "codepoint-order")),
);
const topLevelTitles = withAstralAndPua
.filter((entry) => !entry.url.includes("#"))
.map((entry) => entry.title);
const astralIndex = topLevelTitles.indexOf("Astral");
const puaIndex = topLevelTitles.indexOf("Pua");
assert.ok(
puaIndex >= 0 && astralIndex >= 0 && puaIndex < astralIndex,
"true code point order must place U+E000 before U+10000",
);

console.log("search-index reproducibility: PASS");
} finally {
fs.rmSync(root, { recursive: true, force: true });
}
}

main().catch((error) => {
console.error(error);
process.exitCode = 1;
});
2 changes: 1 addition & 1 deletion docs-site/static/search-index.json

Large diffs are not rendered by default.

Loading