From a9d9634678bb3c81a11a0080bc244ace95fab036 Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Sat, 8 Aug 2026 05:39:32 -0700 Subject: [PATCH] perf(docs-site): split search index by section and stop indexing code blocks MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The client-side search index had grown to 16.3 MB raw / ~4.5 MB wire (4,500+ docs), giving 25-30 s of blank UI before first results — users read the dead window as 'search is broken entirely' (the engine itself returned correct results). Three levers, all config-only: - searchContextByPaths: per-section index chunks. Searching from any docs page now fetches only that section's index (e.g. Reference: 1.4 MB / 387 KB gz) instead of the 16 MB monolith. The search page gains a section dropdown; landing-page searches still cover everything via useAllContextsWithNoSearchContext. - ignoreCssSelectors: ['pre']: fenced code blocks no longer indexed. YAML/shell samples were generating huge high-cardinality lunr token dictionaries; inline code in prose stays searchable. - ignoreFiles: user-stories excluded (527 index docs of scraped community quotes rendered by a React component). Measured (npm run build, both locales): - root 'Everywhere' index: 16.3 MB -> 13.2 MB raw (4.42 -> 3.57 MB gz) - per-section indexes: 0.4-8.3 MB raw (103 KB-2.2 MB gz) - zh-Hans root: 14.6 -> 12.6 MB raw - local serve: first dropdown results in ~175-205 ms for both scoped and Everywhere queries ('telegram' -> 8 options, search page -> 100) --- website/docusaurus.config.ts | 24 ++++++++++++++++++++++++ 1 file changed, 24 insertions(+) diff --git a/website/docusaurus.config.ts b/website/docusaurus.config.ts index a10987be10269..0a5792866d59b 100644 --- a/website/docusaurus.config.ts +++ b/website/docusaurus.config.ts @@ -61,7 +61,31 @@ const config: Config = { ignoreFiles: [ /^user-guide\/skills\/bundled\//, /^user-guide\/skills\/optional\//, + // Community-quote collage page: 527 index documents of scraped + // testimonials that pollute results and bloat the index. The page + // renders a React component; its text isn't reference material. + /^user-stories/, ], + // Split the lunr index into per-section chunks so the browser only + // fetches+hydrates the index for the section being searched instead + // of one monolithic multi-MB file (16.3 MB by Aug 2026 — 25-30 s of + // blank UI on first search). Searches started outside these paths + // (e.g. the docs landing page) fall back to all contexts combined. + searchContextByPaths: [ + { label: 'User Guide', path: 'user-guide' }, + { label: 'Developer Guide', path: 'developer-guide' }, + { label: 'Guides', path: 'guides' }, + { label: 'Reference', path: 'reference' }, + { label: 'Getting Started', path: 'getting-started' }, + { label: 'Integrations', path: 'integrations' }, + ], + useAllContextsWithNoSearchContext: true, + // Don't index fenced code blocks (
). Config/YAML/shell examples
+        // generate huge high-cardinality token dictionaries in lunr (the
+        // biggest single contributor to index size) and nobody searches for
+        // a literal line of a code sample. Inline  (command and config
+        // key names in prose) stays indexed.
+        ignoreCssSelectors: ['pre'],
         // Exact-or-prefix matching only (default is edit distance 1).
         // With fuzzy distance 1, "keet" matched "meetings"/"keep" (one
         // edit away after stemming), and multi-word typo queries against