From e38491980bd98dd7594f04415588678277ada97d Mon Sep 17 00:00:00 2001 From: William Laugesen Date: Mon, 17 Aug 2026 16:34:24 +1200 Subject: [PATCH 01/14] Replace the search.json index with Pagefind Indexes the built HTML at the end of the build and puts Pagefind behind the SearchEngine seam. Body text is indexed for the first time, so a phrase that appears in an article but not its title or headings is now findable. The index goes to dist/docs/pagefind, before pruneDist would delete it, and is built from dist/docs so stored URLs match the /docs/ prefix the client sets as its basePath. data-pagefind-body on the article is what scopes indexing and, as a side effect, excludes all 1,409 redirect stubs: they render through a minimal layout that has no article at all. 1,250 pages are indexed, matching the eligible count the markdown emitter reports. Removes the search.json endpoint and the client scoring it fed. Co-Authored-By: Claude Opus 5 (1M context) --- astro.config.mjs | 4 + package.json | 3 +- pnpm-lock.yaml | 139 +++++++------- src/components/DocsSearch.astro | 2 +- src/integrations/pagefind-index.ts | 82 ++++++++ src/layouts/Default.astro | 55 +++++- src/pages/docs/search.json.ts | 93 --------- src/scripts/docs-search.ts | 4 +- src/scripts/modules/stemmer.js | 197 ------------------- src/scripts/modules/string.js | 83 -------- src/scripts/search-engine-legacy.ts | 265 -------------------------- src/scripts/search-engine-pagefind.ts | 117 ++++++++++++ src/scripts/synonyms.js | 17 -- 13 files changed, 332 insertions(+), 729 deletions(-) create mode 100644 src/integrations/pagefind-index.ts delete mode 100644 src/pages/docs/search.json.ts delete mode 100644 src/scripts/modules/stemmer.js delete mode 100644 src/scripts/modules/string.js delete mode 100644 src/scripts/search-engine-legacy.ts create mode 100644 src/scripts/search-engine-pagefind.ts delete mode 100644 src/scripts/synonyms.js diff --git a/astro.config.mjs b/astro.config.mjs index a5f2c67a89..c8625946b3 100644 --- a/astro.config.mjs +++ b/astro.config.mjs @@ -3,6 +3,7 @@ import { satteri } from '@astrojs/markdown-satteri'; import mdx from '@astrojs/mdx'; import { attributeMarkdown, wrapTables } from '/src/themes/octopus/utilities/custom-markdown.mjs'; import llmMdEmitter from './src/integrations/llm-md-emitter.ts'; +import pagefindIndex from './src/integrations/pagefind-index.ts'; import pruneDist from './src/integrations/prune-dist.ts'; import satteriHeadingId from './src/plugins/satteri-heading-id.js'; import satteriApiExamples, { apiExampleDirective } from './src/plugins/satteri-api-examples.js'; @@ -22,6 +23,9 @@ export default defineConfig({ integrations: [ mdx(), llmMdEmitter(), + // Indexes the HTML the build just wrote, so it lands after the page + // emitters and before the prune that would delete its output + pagefindIndex(), // Must run last: strips build output that can't be served under /docs/ pruneDist() ], diff --git a/package.json b/package.json index 8ea81c1723..e9a2c57a10 100644 --- a/package.json +++ b/package.json @@ -33,8 +33,6 @@ "astro-accelerator-utils": "^0.3.84", "glob": "^13.0.6", "gray-matter": "^4.0.3", - "html-to-text": "^10.0.0", - "keyword-extractor": "^0.0.28", "optional": "^0.1.4", "satteri": "^0.9.5", "sharp": "^0.34.5" @@ -47,6 +45,7 @@ "linkinator": "7.0.0", "npm-run-all": "^4.1.5", "onchange": "^7.1.0", + "pagefind": "^1.5.2", "prettier": "^3.8.3", "prettier-plugin-astro": "^0.14.1" }, diff --git a/pnpm-lock.yaml b/pnpm-lock.yaml index 715655b317..ab0d4d3873 100644 --- a/pnpm-lock.yaml +++ b/pnpm-lock.yaml @@ -39,12 +39,6 @@ importers: gray-matter: specifier: ^4.0.3 version: 4.0.3 - html-to-text: - specifier: ^10.0.0 - version: 10.0.0 - keyword-extractor: - specifier: ^0.0.28 - version: 0.0.28 optional: specifier: ^0.1.4 version: 0.1.4 @@ -76,6 +70,9 @@ importers: onchange: specifier: ^7.1.0 version: 7.1.0 + pagefind: + specifier: ^1.5.2 + version: 1.5.2 prettier: specifier: ^3.8.3 version: 3.8.3 @@ -821,6 +818,41 @@ packages: '@oxc-project/types@0.144.0': resolution: {integrity: sha512-nuhZIOLuI6TFQ32I/WnUx+SCPY7SdSKwgnFHydAuoS1+Z4BRcaP+RRJmGzl9lw+0OFF7UmaESf7KQRXaNLHypg==} + '@pagefind/darwin-arm64@1.5.2': + resolution: {integrity: sha512-MXpI+7HsAdPkvJ0gk9xj9g541BCqBZOBbdwj9g6lB5LCj6kSV6nqDSjzcAJwvOsfu0fjwvC8hQU+ecfhp+MpiQ==} + cpu: [arm64] + os: [darwin] + + '@pagefind/darwin-x64@1.5.2': + resolution: {integrity: sha512-IojxFWMEJe0RQ7PQ3KXQsPIImNsbpPYpoZ+QUDrL8fAl/O27IX+LVLs74/UzEZy5uA2LD8Nz1AiwKr72vrkZQw==} + cpu: [x64] + os: [darwin] + + '@pagefind/freebsd-x64@1.5.2': + resolution: {integrity: sha512-7EVzo9+0w+2cbe671BtMj10UlNo83I+HrLVLfRxO731svHRJKUfJ/mo05gU14pe9PCfpKNQT8FS3Xc/oDN6pOA==} + cpu: [x64] + os: [freebsd] + + '@pagefind/linux-arm64@1.5.2': + resolution: {integrity: sha512-Ovt9+K35sqzn8H3ZMXGwls4TD/wMJuvRtShHIsmUQREmaxjrDEX7gHckRCrwYJ4XE1H1p6HkLz3wukrAnsfXQw==} + cpu: [arm64] + os: [linux] + + '@pagefind/linux-x64@1.5.2': + resolution: {integrity: sha512-V+tFqHKXhQKq/WqPBD67AFy7scn1/aZID00ws4fSDd+1daSi5UHR9VVlRrOUYKxn3VuFQYRD7lYXdZK1WED1YA==} + cpu: [x64] + os: [linux] + + '@pagefind/windows-arm64@1.5.2': + resolution: {integrity: sha512-hN9Nh90fNW61nNRCW9ZyQrAj/mD0eRvmJ8NlTUzkbuW8kIzGJUi3cxjFkEcMZ5h/8FsKWD/VcouZl4yo1F7B6g==} + cpu: [arm64] + os: [win32] + + '@pagefind/windows-x64@1.5.2': + resolution: {integrity: sha512-Fa2Iyw7kaDRzGMfNYNUXNW2zbL5FQVDgSOcbDHdzBrDEdpqOqg8TcZ68F22ol6NJ9IGzvUdmeyZypLW5dyhqsg==} + cpu: [x64] + os: [win32] + '@pkgjs/parseargs@0.11.0': resolution: {integrity: sha512-+1VkjdD0QBLPodGrJUeqarH8VAIvQODIbwh9XpP5Syisf7YoQgsJKPNFoqqLQlu+VQ/tVSshMR6loPMn8U+dPg==} engines: {node: '>=14'} @@ -1008,11 +1040,6 @@ packages: '@rolldown/pluginutils@1.0.1': resolution: {integrity: sha512-2j9bGt5Jh8hj+vPtgzPtl72j0yRxHAyumoo6TNfAjsLB04UtpSvPbPcDcBMxz7n+9CYB0c1GxQFxYRg2jimqGw==} - '@selderee/plugin-htmlparser2@0.12.0': - resolution: {integrity: sha512-oELmoyA6ML9jDRMV3kgcMQFKxUfBU0yFVn6yTctVaLT5ygXnxH52I3TZEgV9EhXJC68/uFvE5Daj1/25c0Xa/A==} - peerDependencies: - selderee: ~0.12.0 - '@shikijs/core@4.4.3': resolution: {integrity: sha512-QCR4q2ZO/ILJEuwiBMel4wdcTDb1JGwfjKTxPDF6x8ixOaluPrVqIn06C99AcRPhmYlBR56d/Fb+GN58GzExpg==} engines: {node: '>=20'} @@ -1398,10 +1425,6 @@ packages: decode-named-character-reference@1.3.0: resolution: {integrity: sha512-GtpQYB283KrPp6nRw50q3U9/VfOutZOe103qlN7BPP6Ad27xYnOIWv4lPzo8HCAL+mMZofJ9KEy30fq6MfaK6Q==} - deepmerge-ts@7.1.5: - resolution: {integrity: sha512-HOJkrhaYsweh+W+e74Yn7YStZOilkoPb6fycpwNLKzSPtruFs48nYis0zy5yJz1+ktUhHxoRDJ27RQAWLIJVJw==} - engines: {node: '>=16.0.0'} - define-data-property@1.1.4: resolution: {integrity: sha512-rBMvIzlpA8v6E+SJZoo++HAYqsLrkg7MSfIinMPFhmkorw7X+dOXVJQs+QT69zGkzMyfDnIMN2Wid1+NbL3T+A==} engines: {node: '>= 0.4'} @@ -1778,10 +1801,6 @@ packages: html-escaper@3.0.3: resolution: {integrity: sha512-RuMffC89BOWQoY0WKGpIhn5gX3iI54O6nRA0yC124NYVtzjmFWBIiFd8M0x+ZdX0P9R4lADg1mgP8C7PxGOWuQ==} - html-to-text@10.0.0: - resolution: {integrity: sha512-2OH59Gtprdczel+7Rxgpz9hGVJREaf8Lt1H4kZwWHpEn70VQKRuMNGsb2eDbwaTzrYzb0hheiOG1P7Dim0B4dQ==} - engines: {node: '>=20.19.0'} - html-void-elements@3.0.0: resolution: {integrity: sha512-bEqo66MRXsUGxWHV5IP0PUiAWwoEjba4VCzg0LjFJBpchPaTfyfCKTG6bc5F8ucKec3q5y6qOdGyYTSBEvhCrg==} @@ -1979,17 +1998,10 @@ packages: jsonc-parser@3.3.1: resolution: {integrity: sha512-HUgH65KyejrUFPvHFPbqOY0rsFip3Bo5wb4ngvdi1EpCYWUQDC5V+Y7mZws+DLkr4M//zQJoanu1SP+87Dv1oQ==} - keyword-extractor@0.0.28: - resolution: {integrity: sha512-oi7dSPpYtW/3fE0vZiqQgZ8mW3F1V9K4+rBJ0FcVrdXBEQuhZ0zKj7sX74eqGASuepLHf9aYdeonyKHWhYpHQA==} - engines: {node: '>= 0.10.0'} - kind-of@6.0.3: resolution: {integrity: sha512-dcS1ul+9tmeD95T+x28/ehLgd9mENa3LsvDTtzm3vyBEO7RPptvAD+t44WVXaUjTBRcrpFeFlC8WCruUR456hw==} engines: {node: '>=0.10.0'} - leac@0.7.0: - resolution: {integrity: sha512-qMrZeyEekgdRQ9o6a4NAB2EQZrv827GJdn1vnapwSJ90hWRB4TzUSunvacPkxQ2TnNqHNI1/zSt0hlo0crG8Jw==} - lightningcss-android-arm64@1.33.0: resolution: {integrity: sha512-gEpRTalKdosp4Bb8qWtc2iOgE5SeIHlpS1up9bFq2wAyYhl1UdTObYiHe98zEM9SQvSoqQZ1IQD0JNpg3Ml5pg==} engines: {node: '>= 12.0.0'} @@ -2392,6 +2404,10 @@ packages: package-manager-detector@1.8.0: resolution: {integrity: sha512-yQA4H19AmPEoMUeavPMDIe1higySl/gH/yaQrkT/s07Qp+7pp2hYz30N3z2l5BkjVkF9Ow6o0wjJamm2y7Sn0A==} + pagefind@1.5.2: + resolution: {integrity: sha512-XTUaK0hXMCu2jszWE584JGQT7y284TmMV9l/HX3rnG5uo3rHI/uHU56XTyyyPFjeWEBxECbAi0CaFDJOONtG0Q==} + hasBin: true + parse-entities@4.0.2: resolution: {integrity: sha512-GG2AQYWoLgL877gQIKeRPGO1xF9+eG1ujIb5soS5gPvLQ1y2o8FL90w2QWNdf9I361Mpp7726c+lj3U0qK1uGw==} @@ -2405,9 +2421,6 @@ packages: parse5@7.3.0: resolution: {integrity: sha512-IInvU7fabl34qmi9gY8XOVxhYyMyuH2xUNpb2q8/Y+7552KlejkRvqvD19nMoUW/uQGGbqNpA6Tufu5FL5BZgw==} - parseley@0.13.1: - resolution: {integrity: sha512-uNBJZzmb60l6p6VWLTmevizNAGnE0xoSf1n0B4q3ntegDNzcS68NRCcBDZTcyXHxt2XhBChsCuqj4M+nChvE/A==} - path-key@2.0.1: resolution: {integrity: sha512-fEHGKCSmUSDPv4uoj8AlD+joPlq3peND+HRYyxFz4KPw4z926S/b8rIuFs2FYJg3BwsxJf6A9/3eIdLaYC+9Dw==} engines: {node: '>=4'} @@ -2431,9 +2444,6 @@ packages: resolution: {integrity: sha512-T2ZUsdZFHgA3u4e5PfPbjd7HDDpxPnQb5jN0SrDsjNSuVXHJqtwTnWqG0B1jZrgmJ/7lj1EmVIByWt1gxGkWvg==} engines: {node: '>=4'} - peberminta@0.10.0: - resolution: {integrity: sha512-80B2AsU+I4Qdb0ZAPSfe9UwvGzwkM37IKIFEvdS3D/3Ndgv2bsuJ0bfG1+iEYO+l7Gfd4EUJmuRyq7efLgRMzQ==} - piccolore@0.1.3: resolution: {integrity: sha512-o8bTeDWjE086iwKrROaDf31K0qC/BENdm15/uH9usSC/uZjJOKb2YGiVHfLY4GhwsERiPI1jmwI2XrA7ACOxVw==} @@ -2634,9 +2644,6 @@ packages: resolution: {integrity: sha512-vfD3pmTzGpufjScBh50YHKzEu2lxBWhVEHsNGoEXmCmn2hKGfeNLYMzCJpe8cD7gqX7TJluOVpBkAequ6dgMmA==} engines: {node: '>=4'} - selderee@0.12.0: - resolution: {integrity: sha512-b1YMh3+DHZp59DLna3qVwQ5iOla/nrI6mLBNW02XxU77M3046Df6VLkoaJyFz20VsGIG5kkp+FK0kg4K4HnUFw==} - semver@5.7.2: resolution: {integrity: sha512-cBznnQ9KjJqU67B52RMC65CMarK2600WFnbkcaiwWq3xy/5haFJlshgnpjovMVJ+Hff49d8GEn0b87C5pDQ10g==} hasBin: true @@ -3796,6 +3803,27 @@ snapshots: '@oxc-project/types@0.144.0': {} + '@pagefind/darwin-arm64@1.5.2': + optional: true + + '@pagefind/darwin-x64@1.5.2': + optional: true + + '@pagefind/freebsd-x64@1.5.2': + optional: true + + '@pagefind/linux-arm64@1.5.2': + optional: true + + '@pagefind/linux-x64@1.5.2': + optional: true + + '@pagefind/windows-arm64@1.5.2': + optional: true + + '@pagefind/windows-x64@1.5.2': + optional: true + '@pkgjs/parseargs@0.11.0': optional: true @@ -3916,12 +3944,6 @@ snapshots: '@rolldown/pluginutils@1.0.1': {} - '@selderee/plugin-htmlparser2@0.12.0(selderee@0.12.0)': - dependencies: - domelementtype: 2.3.0 - domhandler: 5.0.3 - selderee: 0.12.0 - '@shikijs/core@4.4.3': dependencies: '@shikijs/primitive': 4.4.3 @@ -4437,8 +4459,6 @@ snapshots: dependencies: character-entities: 2.0.2 - deepmerge-ts@7.1.5: {} - define-data-property@1.1.4: dependencies: es-define-property: 1.0.1 @@ -4998,14 +5018,6 @@ snapshots: html-escaper@3.0.3: {} - html-to-text@10.0.0: - dependencies: - '@selderee/plugin-htmlparser2': 0.12.0(selderee@0.12.0) - deepmerge-ts: 7.1.5 - dom-serializer: 2.0.0 - htmlparser2: 10.1.0 - selderee: 0.12.0 - html-void-elements@3.0.0: {} htmlparser2@10.1.0: @@ -5193,12 +5205,8 @@ snapshots: jsonc-parser@3.3.1: {} - keyword-extractor@0.0.28: {} - kind-of@6.0.3: {} - leac@0.7.0: {} - lightningcss-android-arm64@1.33.0: optional: true @@ -5851,6 +5859,16 @@ snapshots: package-manager-detector@1.8.0: {} + pagefind@1.5.2: + optionalDependencies: + '@pagefind/darwin-arm64': 1.5.2 + '@pagefind/darwin-x64': 1.5.2 + '@pagefind/freebsd-x64': 1.5.2 + '@pagefind/linux-arm64': 1.5.2 + '@pagefind/linux-x64': 1.5.2 + '@pagefind/windows-arm64': 1.5.2 + '@pagefind/windows-x64': 1.5.2 + parse-entities@4.0.2: dependencies: '@types/unist': 2.0.11 @@ -5879,11 +5897,6 @@ snapshots: dependencies: entities: 6.0.1 - parseley@0.13.1: - dependencies: - leac: 0.7.0 - peberminta: 0.10.0 - path-key@2.0.1: {} path-key@3.1.1: {} @@ -5904,8 +5917,6 @@ snapshots: dependencies: pify: 3.0.0 - peberminta@0.10.0: {} - piccolore@0.1.3: {} picocolors@1.1.1: {} @@ -6198,10 +6209,6 @@ snapshots: extend-shallow: 2.0.1 kind-of: 6.0.3 - selderee@0.12.0: - dependencies: - parseley: 0.13.1 - semver@5.7.2: {} semver@7.8.0: {} diff --git a/src/components/DocsSearch.astro b/src/components/DocsSearch.astro index 4957e1b73c..10de1ccd35 100644 --- a/src/components/DocsSearch.astro +++ b/src/components/DocsSearch.astro @@ -44,7 +44,7 @@ if (facets.length === 0) { throw new Error('DocsSearch needs at least one facet for its tab strip'); } -const indexUrl = `${SITE.subfolder}/search.json`; +const indexUrl = `${SITE.subfolder}/pagefind/`; const listboxId = `${name}-search-listbox`; --- diff --git a/src/integrations/pagefind-index.ts b/src/integrations/pagefind-index.ts new file mode 100644 index 0000000000..ce6206cb1a --- /dev/null +++ b/src/integrations/pagefind-index.ts @@ -0,0 +1,82 @@ +// Builds the Pagefind index from the HTML the site just emitted. +// +// Two things about this repo shape the setup. The index has to land inside +// `dist/docs/`, because `prune-dist.ts` deletes every top-level entry outside +// its allowlist and only `/docs/` is served on octopus.com. And this has to be +// registered before `pruneDist()` in `astro.config.mjs`, since integration hooks +// run in the order they are listed. +// +// What gets indexed is decided in `Default.astro` by `data-pagefind-body`, +// `data-pagefind-ignore` and `data-pagefind-filter`, not here. + +import type { AstroIntegration } from 'astro'; +import { fileURLToPath } from 'node:url'; +import * as path from 'node:path'; +import * as fs from 'node:fs'; +// Statically, because `astro:build:done` fires after Vite's module runner has +// closed and a dynamic import from inside the hook cannot be resolved. +import { createIndex, close } from 'pagefind'; + +// Roughly the number of pages that pass the search predicate. Well under it +// means the body scoping or the exclusions have broken, which is a silent +// failure otherwise — search still works, it is just full of the wrong pages. +const EXPECTED_PAGES = 1000; + +export default function pagefindIndex(): AstroIntegration { + return { + name: 'pagefind-index', + hooks: { + 'astro:build:done': async ({ dir, logger }) => { + const distDir = fileURLToPath(dir); + + const { index, errors } = await createIndex({ keepIndexUrl: false }); + + if (!index) { + logger.error(`could not start Pagefind: ${errors.join(', ')}`); + return; + } + + // `dist/docs` rather than `dist`, so stored URLs are relative to the + // /docs/ prefix the site is proxied under. The client sets a `basePath` + // of `/docs/`, which Pagefind prepends back on at query time — indexing + // from `dist` would put `/docs/` on twice. + const added = await index.addDirectory({ + path: path.join(distDir, 'docs'), + glob: '**/*.html', + }); + + if (added.errors.length > 0) { + logger.error(added.errors.join(', ')); + } + + const written = await index.writeFiles({ + outputPath: path.join(distDir, 'docs', 'pagefind'), + }); + + if (written.errors.length > 0) { + logger.error(written.errors.join(', ')); + } + + await close(); + + // `page_count` is files scanned, not files indexed — it counts the + // redirect stubs that `data-pagefind-body` then drops. One fragment is + // written per indexed page, so that is the number worth checking. + const outDir = path.join(distDir, 'docs', 'pagefind'); + const indexed = ( + await fs.promises.readdir(path.join(outDir, 'fragment')) + ).length; + + logger.info( + `indexed ${indexed} of ${added.page_count} pages into docs/pagefind` + ); + + if (indexed < EXPECTED_PAGES) { + logger.warn( + `only ${indexed} pages were indexed, expected at least ${EXPECTED_PAGES} — check data-pagefind-body is still on the article` + ); + } + }, + }, + }; +} diff --git a/src/layouts/Default.astro b/src/layouts/Default.astro index a4e7ec2bdc..2ecc3e45a4 100644 --- a/src/layouts/Default.astro +++ b/src/layouts/Default.astro @@ -59,6 +59,39 @@ const showSearch = !isSearchPage; // The footer prints this; JSON-LD carries it as dateModified. const lastUpdated = frontmatter.modDate ?? frontmatter.pubDate ?? null; + +// Search indexing. The same predicate `PostFiltering.showInSearch` applies at +// the point the page is rendered — redirect stubs never reach this layout at +// all, so only the frontmatter opt-outs have to be reproduced here. +const indexable = + frontmatter.navSearch !== false && + frontmatter.listable !== false && + frontmatter.draft !== true; + +// Which tab a result lands under. Longest match first: every CLI page also sits +// under the REST API tree. Mirrors `classify()` in `src/scripts/search-engine.ts`, +// which is what the old index derives the same value from at query time. +const sections: [string, RegExp][] = [ + [ + 'cli', + /^\/docs\/octopus-rest-api\/(cli|octopus-cli|[a-z.]+-command-line)(\/|$)/, + ], + ['api', /^\/docs\/octopus-rest-api(\/|$)/], + ['integrations', /^\/docs\/api-and-integration(\/|$)/], +]; +const section = + sections.find(([, prefix]) => prefix.test(Astro.url.pathname))?.[0] ?? 'docs'; + +// The trail the result row prints, taken from the path so it matches whatever +// the page is actually filed under. +const trail = Astro.url.pathname + .split('/') + .filter(Boolean) + .slice(1, -1) + .map((segment) => + segment.replace(/-/g, ' ').replace(/^./, (c) => c.toUpperCase()) + ) + .join(' / '); ---
-
+ { + /* Scoping the index to the article is what keeps every sidebar link + and footer word out of every page's search terms. Without it the + index is still built and search still "works" — it is just bad, + which is why the integration asserts on the page count. + + The article rather than `.page-content`, so the heading inside it is + there for Pagefind to take the result title from. */ + } +
-
+
- + {/* Same control on every page, so it is noise in every result. */} +
+ +
diff --git a/src/pages/docs/search.json.ts b/src/pages/docs/search.json.ts deleted file mode 100644 index 6ef6c53a15..0000000000 --- a/src/pages/docs/search.json.ts +++ /dev/null @@ -1,93 +0,0 @@ -/** @format */ - -// warning: This file is overwritten by Astro Accelerator - -import { accelerator } from '@lib/accelerator'; -import { PostFiltering } from 'astro-accelerator-utils'; -import type { MarkdownInstance } from 'astro'; -import { SITE } from '@config'; -import { convert } from 'html-to-text'; -import keywordExtractor from 'keyword-extractor'; -import { isUnderConstruction } from '@lib/underConstruction'; - -const getData = async () => { - //@ts-ignore - const allPages = import.meta.glob(['./**/*.md', './**/*.mdx']); - const items = []; - - for (const path in allPages) { - // Temporary - see src/lib/underConstruction.ts. - if (isUnderConstruction(path)) { - continue; - } - - const page = (await allPages[path]()) as MarkdownInstance< - Record - >; - - if (!PostFiltering.showInSearch(page)) { - continue; - } - - let url = page.url ?? ''; - - if (page.frontmatter.paged) { - url += '/1/'; - } - - const headings = await page.getHeadings(); - const title = await accelerator.markdown.getTextFrom( - page.frontmatter?.title - ); - const content = page.compiledContent ? await page.compiledContent() : ''; - let counted: { word: string; count: number }[] = []; - - if (content) { - const options = { - wordwrap: false, - selectors: [ - { selector: 'a', options: { ignoreHref: true } }, - { selector: 'img', format: 'skip' }, - { selector: 'h1', options: { uppercase: false } }, - ], - }; - const text = convert(content, options); - - const words = keywordExtractor.extract(text, { - language: 'english', - return_changed_case: true, - remove_duplicates: true, - }); - - counted = words - .map((w) => { - return { - word: w, - count: words.filter((wd) => wd === w).length, - }; - }) - .filter((e) => e.word.replace(/[^a-z]+/g, '').length > 1); - } - - items.push({ - title: title, - headings: headings.map((h) => { - return { text: h.text, slug: h.slug }; - }), - description: page.frontmatter.description ?? '', - keywords: counted.map((c) => c.word).join(' '), - tags: page.frontmatter.tags ?? [], - url: SITE.url + accelerator.urlFormatter.formatAddress(url), - date: page.frontmatter.pubDate ?? '', - }); - } - - return new Response(JSON.stringify(items), { - status: 200, - headers: { - 'Content-Type': 'application/json', - }, - }); -}; - -export const GET = getData; diff --git a/src/scripts/docs-search.ts b/src/scripts/docs-search.ts index ef88eb5758..be62b753a8 100644 --- a/src/scripts/docs-search.ts +++ b/src/scripts/docs-search.ts @@ -14,7 +14,7 @@ import { type SearchEngine, type SearchResult, } from './search-engine'; -import { legacyEngine } from './search-engine-legacy'; +import { pagefindEngine } from './search-engine-pagefind'; const DEBOUNCE_MS = 150; const SITE_SEARCH = 'site'; @@ -41,7 +41,7 @@ function setup(dialog: HTMLDialogElement) { const demo = dialog.dataset.demoResults; const engine: SearchEngine = demo ? fixtureEngine(JSON.parse(demo)) - : legacyEngine(dialog.dataset.indexUrl ?? '/docs/search.json'); + : pagefindEngine(dialog.dataset.indexUrl ?? '/docs/pagefind/'); let facet = 'all'; let active = -1; diff --git a/src/scripts/modules/stemmer.js b/src/scripts/modules/stemmer.js deleted file mode 100644 index 778ebcc9c0..0000000000 --- a/src/scripts/modules/stemmer.js +++ /dev/null @@ -1,197 +0,0 @@ -/** - * Removes "morphological and inflexional endings" from words - * See: http://www.tartarus.org/~martin/PorterStemmer - */ -export const stemmer = (function () { - const step2list = { - ational: 'ate', - tional: 'tion', - enci: 'ence', - anci: 'ance', - izer: 'ize', - bli: 'ble', - alli: 'al', - entli: 'ent', - eli: 'e', - ousli: 'ous', - ization: 'ize', - ation: 'ate', - ator: 'ate', - alism: 'al', - iveness: 'ive', - fulness: 'ful', - ousness: 'ous', - aliti: 'al', - iviti: 'ive', - biliti: 'ble', - logi: 'log', - }; - - const step3list = { - icate: 'ic', - ative: '', - alize: 'al', - iciti: 'ic', - ical: 'ic', - ful: '', - ness: '', - }; - - const c = '[^aeiou]', // consonant - v = '[aeiouy]', // vowel - C = c + '[^aeiouy]*', // consonant sequence - V = v + '[aeiou]*', // vowel sequence - mgr0 = '^(' + C + ')?' + V + C, // [C]VC... is m>0 - meq1 = '^(' + C + ')?' + V + C + '(' + V + ')?$', // [C]VC[V] is m=1 - mgr1 = '^(' + C + ')?' + V + C + V + C, // [C]VCVC... is m>1 - s_v = '^(' + C + ')?' + v; // vowel in stem - - /** - * @param {string} w - * @returns {string} - */ - return function (w) { - var stem, - suffix, - firstch, - re, - re2, - re3, - re4, - origword = w; - - if (w.length < 3) { - return w; - } - - firstch = w.substring(0, 1); - - if (firstch == 'y') { - w = firstch.toUpperCase() + w.substring(1, w.length); - } - - // Step 1a - re = /^(.+?)(ss|i)es$/; - re2 = /^(.+?)([^s])s$/; - - if (re.test(w)) { - w = w.replace(re, '$1$2'); - } else if (re2.test(w)) { - w = w.replace(re2, '$1$2'); - } - - // Step 1b - re = /^(.+?)eed$/; - re2 = /^(.+?)(ed|ing)$/; - if (re.test(w)) { - var fp = re.exec(w); - re = new RegExp(mgr0); - if (re.test(fp[1])) { - re = /.$/; - w = w.replace(re, ''); - } - } else if (re2.test(w)) { - var fp = re2.exec(w); - stem = fp[1]; - re2 = new RegExp(s_v); - if (re2.test(stem)) { - w = stem; - re2 = /(at|bl|iz)$/; - re3 = new RegExp('([^aeiouylsz])\\1$'); - re4 = new RegExp('^' + C + v + '[^aeiouwxy]$'); - if (re2.test(w)) { - w = w + 'e'; - } else if (re3.test(w)) { - re = /.$/; - w = w.replace(re, ''); - } else if (re4.test(w)) { - w = w + 'e'; - } - } - } - - // Step 1c - re = /^(.+?)y$/; - if (re.test(w)) { - var fp = re.exec(w); - stem = fp[1]; - re = new RegExp(s_v); - if (re.test(stem)) { - w = stem + 'i'; - } - } - - // Step 2 - re = - /^(.+?)(ational|tional|enci|anci|izer|bli|alli|entli|eli|ousli|ization|ation|ator|alism|iveness|fulness|ousness|aliti|iviti|biliti|logi)$/; - if (re.test(w)) { - var fp = re.exec(w); - stem = fp[1]; - suffix = fp[2]; - re = new RegExp(mgr0); - if (re.test(stem)) { - w = stem + step2list[suffix]; - } - } - - // Step 3 - re = /^(.+?)(icate|ative|alize|iciti|ical|ful|ness)$/; - if (re.test(w)) { - var fp = re.exec(w); - stem = fp[1]; - suffix = fp[2]; - re = new RegExp(mgr0); - if (re.test(stem)) { - w = stem + step3list[suffix]; - } - } - - // Step 4 - re = - /^(.+?)(al|ance|ence|er|ic|able|ible|ant|ement|ment|ent|ou|ism|ate|iti|ous|ive|ize)$/; - re2 = /^(.+?)(s|t)(ion)$/; - if (re.test(w)) { - var fp = re.exec(w); - stem = fp[1]; - re = new RegExp(mgr1); - if (re.test(stem)) { - w = stem; - } - } else if (re2.test(w)) { - var fp = re2.exec(w); - stem = fp[1] + fp[2]; - re2 = new RegExp(mgr1); - if (re2.test(stem)) { - w = stem; - } - } - - // Step 5 - re = /^(.+?)e$/; - if (re.test(w)) { - var fp = re.exec(w); - stem = fp[1]; - re = new RegExp(mgr1); - re2 = new RegExp(meq1); - re3 = new RegExp('^' + C + v + '[^aeiouwxy]$'); - if (re.test(stem) || (re2.test(stem) && !re3.test(stem))) { - w = stem; - } - } - - re = /ll$/; - re2 = new RegExp(mgr1); - if (re.test(w) && re2.test(w)) { - re = /.$/; - w = w.replace(re, ''); - } - - // and turn initial Y back to y - - if (firstch == 'y') { - w = firstch.toLowerCase() + w.substr(1); - } - - return w; - }; -})(); diff --git a/src/scripts/modules/string.js b/src/scripts/modules/string.js deleted file mode 100644 index dea7142565..0000000000 --- a/src/scripts/modules/string.js +++ /dev/null @@ -1,83 +0,0 @@ -// @ts-check - -/** - * Looks for a search within a string - * - * @param {string} string - * @param {string} search - * @returns - */ -function contains(string, search) { - return string.indexOf(search) > -1; -} - -/** - * Looks for a search within a string - * - * @param {string} string - * @param {string} search - * @returns - */ -function containsWord(string, search) { - return string.split(' ').indexOf(search) > -1; -} - -/** - * - * @param {string} string - * @param {string[]} terms - * @returns - */ -function highlight(string, terms) { - terms.forEach((term) => { - const regEx = new RegExp(term, 'ig'); - const matches = string.match(regEx); - if (matches) { - string = string.replace(regEx, `${matches[0]}`); - } - }); - return string; -} - -/** - * Simplifies a string to plain lower case, removing diacritic characters and hyphens - * This means a search for "co-op" will be found in "COOP" and "Café" will be found in "cafe" - * @param {string} string - * @returns {string} - */ -function sanitise(string) { - // @ts-ignore - if (String.prototype.normalize) { - // Reduces diacritic characters to plain characters - string - .trim() - .normalize('NFD') - .replace(/\./g, ' ') - .replace(/[\u0300-\u036f]/g, '') - .toLowerCase() - .replace(/-/g, ''); - } - - // Some browsers can't normalise strings - return string.trim().toLowerCase().replace(/-/g, ''); -} - -/** - * Sets a minimum length for a search - * @param {string} string - * @returns - */ -function isLongEnough(string) { - return string.length > 1; -} - -/** - * - * @param {string} string - * @returns {string[]} - */ -function explode(string) { - return string.split(' ').filter(isLongEnough).map(sanitise); -} - -export { contains, containsWord, sanitise, explode, highlight }; diff --git a/src/scripts/search-engine-legacy.ts b/src/scripts/search-engine-legacy.ts deleted file mode 100644 index 398c758757..0000000000 --- a/src/scripts/search-engine-legacy.ts +++ /dev/null @@ -1,265 +0,0 @@ -// The current search, behind the `SearchEngine` seam: one build-time JSON of -// titles, headings, descriptions, tags and an extracted keyword bag, scored in -// the browser. The scoring is carried over unchanged from the results list this -// replaces, so the overlay returns exactly what the old dropdown returned. -// -// Body text is not in the index, which is the ceiling on how good this can be -// and the reason the Pagefind and Orama spikes exist. - -import { contains, explode, highlight, sanitise } from './modules/string.js'; -import { stemmer } from './modules/stemmer.js'; -import { - breadcrumbFrom, - classify, - countByFacet, - type SearchEngine, - type SearchResult, -} from './search-engine'; - -type Heading = { text: string; slug: string; safeText: string }; - -type Entry = { - title: string; - safeTitle: string; - description: string; - keywords: string; - tags: string[]; - headings: Heading[]; - url: string; - depth: number; -}; - -type Scored = Entry & { - score: number; - foundWords: number; - foundTerms: string[]; -}; - -// Phrase and per-term weights, verbatim from the list this replaces. -const SCORING = { - depth: 5, - phraseTitle: 60, - phraseHeading: 20, - phraseDescription: 20, - termTitle: 40, - termHeading: 15, - termDescription: 15, - termTags: 15, - termKeywords: 15, -}; - -const WORD_SCORES = { - titleExact: 20, - titleContains: 15, - headingContains: 10, - contentContains: 1, -}; - -const RESULT_LIMIT = 30; - -async function getSynonyms(): Promise> { - try { - return (await import('./synonyms.js')).synonyms; - } catch { - return {}; - } -} - -/** Terms reach `highlight()` as a regular expression, so a query like `c++` - * would throw before anything was drawn. */ -function escapeForRegExp(term: string) { - return term.replace(/[.*+?^${}()|[\]\\]/g, '\\$&'); -} - -/** The excerpt is set as HTML so the `` around each hit survives. Escaping - * first means those marks are the only tags that can ever reach the DOM. */ -function escapeHtml(text: string) { - return text - .replace(/&/g, '&') - .replace(//g, '>'); -} - -async function expand(query: string) { - const synonyms = await getSynonyms(); - const terms: string[] = []; - - for (const term of explode(query)) { - const synonym = synonyms[term]; - // An empty synonym is a stop word: drop the term rather than expanding it. - if (synonym === '') continue; - terms.push(term); - if (synonym) terms.push(...synonym.split(' ')); - } - - const stemmed = terms - .map((term) => stemmer(term)) - .filter((stem) => !terms.includes(stem)); - - return { terms, allTerms: [...terms, ...stemmed] }; -} - -function score(entry: Entry, query: string, allTerms: string[]): Scored { - const scored: Scored = { ...entry, score: 0, foundWords: 0, foundTerms: [] }; - - // Phrase matches: the whole query, found intact - if (scored.safeTitle === query) scored.foundWords += WORD_SCORES.titleExact; - - if (contains(scored.safeTitle, query)) { - scored.score += SCORING.phraseTitle; - scored.foundWords += WORD_SCORES.titleContains; - } - - for (const heading of scored.headings) { - if (contains(heading.safeText, query)) { - scored.score += SCORING.phraseHeading; - scored.foundWords += WORD_SCORES.headingContains; - } - } - - if (contains(scored.description, query)) { - scored.score += SCORING.phraseDescription; - scored.foundWords += WORD_SCORES.contentContains; - } - - // Term matches: each word of the query, found anywhere - for (const term of allTerms) { - let found = false; - - if (contains(scored.safeTitle, term)) { - scored.score += SCORING.termTitle; - scored.foundWords += WORD_SCORES.headingContains / 2; - found = true; - } - - for (const heading of scored.headings) { - if (contains(heading.safeText, term)) { - scored.score += SCORING.termHeading; - found = true; - } - } - - if (contains(scored.description, term)) { - scored.score += SCORING.termDescription; - found = true; - } - - for (const tag of scored.tags) { - if (contains(tag, term)) { - scored.score += SCORING.termTags; - found = true; - } - } - - if (contains(scored.keywords, term)) { - scored.score += SCORING.termKeywords; - found = true; - } - - if (found) { - scored.foundWords++; - if (!scored.foundTerms.includes(term)) scored.foundTerms.push(term); - } - } - - // Shallow pages win ties: /docs/features over /docs/features/a/b - if (scored.score > 0) { - if (scored.depth < 5) { - scored.score += SCORING.depth; - scored.foundWords++; - } - if (scored.depth < 4) { - scored.score += SCORING.depth; - scored.foundWords++; - } - } - - return scored; -} - -function byRelevance(a: Scored, b: Scored) { - if (b.foundTerms.length !== a.foundTerms.length) { - return b.foundTerms.length - a.foundTerms.length; - } - if (b.foundWords !== a.foundWords) return b.foundWords - a.foundWords; - return b.score - a.score; -} - -export function legacyEngine(indexUrl: string): SearchEngine { - // One fetch, on the first search rather than on page load, shared by every - // search after it. The old list fetched 1.86MB the moment the script ran. - let loading: Promise | null = null; - - function load() { - loading ??= fetch(indexUrl) - .then((response) => response.json()) - .then((data: Record[]) => - data.map((item) => ({ - title: item.title, - safeTitle: sanitise(item.title), - description: item.description ?? '', - keywords: item.keywords ?? '', - tags: (item.tags ?? []).map((tag: string) => sanitise(tag)), - headings: (item.headings ?? []).map((heading: Heading) => ({ - ...heading, - safeText: sanitise(heading.text), - })), - url: item.url, - depth: item.url.match(/\//g)?.length ?? 0, - })) - ) - .catch(() => { - // A failed load must not poison every later search. - loading = null; - return [] as Entry[]; - }); - - return loading; - } - - return { - warm: load, - - async search(rawQuery, facet) { - // Chained words are joined, so `System.Text` searches as `systemtext`. - const query = sanitise(rawQuery.replace(/\./g, ' ')); - if (!query) return { results: [], counts: countByFacet([]) }; - - const [haystack, { terms, allTerms }] = await Promise.all([ - load(), - expand(query), - ]); - - const highlightTerms = terms.map(escapeForRegExp); - - const matched = haystack - .map((entry) => score(entry, query, allTerms)) - .filter((entry) => entry.score > 0) - .sort(byRelevance) - .slice(0, RESULT_LIMIT) - .map((entry): SearchResult => { - // The index stores absolute production URLs, and every entry in it is - // a page of this site. Keeping only the path is what makes a result - // stay on whatever host is serving it — an ephemeral environment or - // localhost would otherwise send you to production. - const path = new URL(entry.url, window.location.origin).pathname; - - return { - url: path, - title: entry.title, - excerpt: highlight(escapeHtml(entry.description), highlightTerms), - breadcrumb: breadcrumbFrom(path), - ...classify(path), - }; - }); - - return { - counts: countByFacet(matched), - results: - facet && facet !== 'all' - ? matched.filter((result) => result.facet === facet) - : matched, - }; - }, - }; -} diff --git a/src/scripts/search-engine-pagefind.ts b/src/scripts/search-engine-pagefind.ts new file mode 100644 index 0000000000..bdfbeb3eae --- /dev/null +++ b/src/scripts/search-engine-pagefind.ts @@ -0,0 +1,117 @@ +// Pagefind behind the `SearchEngine` seam. +// +// The index is chunked and content-hashed, and Pagefind fetches only the chunks +// a query actually touches. That is the whole reason for the spike: the old +// index is a single 1.86MB download, and this one's per-query cost stays flat as +// the corpus grows. +// +// Two calls per search: `search()` returns lightweight stubs, then each result's +// `data()` fetches its own fragment. Only the page of results being shown is +// loaded, so the fragments fetched scale with the result limit, not the corpus. + +import { + breadcrumbFrom, + classify, + type SearchEngine, + type SearchResult, +} from './search-engine'; + +const RESULT_LIMIT = 30; + +type PagefindFragment = { + url: string; + meta?: Record; + excerpt: string; +}; + +type PagefindResultStub = { data(): Promise }; + +type PagefindApi = { + options(config: Record): Promise; + filters(): Promise>>; + search( + query: string, + options?: { filters?: Record } + ): Promise<{ + results: PagefindResultStub[]; + totalFilters: Record>; + unfilteredResultCount: number; + }>; +}; + +export function pagefindEngine(bundlePath: string): SearchEngine { + let loading: Promise | null = null; + + function load() { + loading ??= import(/* @vite-ignore */ `${bundlePath}pagefind.js`) + .then(async (api: PagefindApi) => { + // Pagefind resolves its chunk URLs against this, and prepends it to + // every result URL; the site is proxied under /docs/ rather than served + // from the root, and the index is built relative to that same prefix. + await api.options({ basePath: bundlePath }); + // The filter index is a separate chunk, and a search returns empty + // filter counts until it has been pulled down. Loading it here rather + // than per-search means the tab strip is populated on the first result. + await api.filters(); + return api; + }) + .catch(() => { + // A failed load must not poison every later search. + loading = null; + return null; + }); + + return loading; + } + + return { + warm: load, + + async search(rawQuery, facet) { + const query = rawQuery.trim(); + const empty = { results: [], counts: { all: 0 } }; + if (!query) return empty; + + const api = await load(); + if (!api) return empty; + + const response = await api.search(query, { + filters: facet && facet !== 'all' ? { section: [facet] } : undefined, + }); + + // Counts come from the whole match set rather than the filtered one, so + // the strip keeps showing what the other tabs hold while one is selected. + const counts: Record = { + all: response.unfilteredResultCount, + ...(response.totalFilters?.section ?? {}), + }; + + const fragments = await Promise.all( + response.results.slice(0, RESULT_LIMIT).map((stub) => stub.data()) + ); + + const results = fragments.map((fragment): SearchResult => { + const path = new URL(fragment.url, window.location.origin).pathname; + + return { + url: path, + // Pagefind takes the title from the first heading inside the indexed + // body, which is why the article rather than `.page-content` carries + // `data-pagefind-body`. + title: fragment.meta?.title ?? path, + // Already carries around the hits, and Pagefind escapes the + // surrounding text itself. + excerpt: fragment.excerpt, + // Written into the page by the layout; the path is the fallback for + // anything built before that attribute existed. + breadcrumb: fragment.meta?.trail + ? fragment.meta.trail.split(' / ') + : breadcrumbFrom(path), + ...classify(path), + }; + }); + + return { results, counts }; + }, + }; +} diff --git a/src/scripts/synonyms.js b/src/scripts/synonyms.js deleted file mode 100644 index 489fc14133..0000000000 --- a/src/scripts/synonyms.js +++ /dev/null @@ -1,17 +0,0 @@ -const synonyms = { - // Keep me alphabetical - licence: 'license', - logs: 'log', - regex: 'regular expression', - // Join words - a: '', - and: '', - for: '', - if: '', - the: '', - to: '', - via: '', - with: '', -}; - -export { synonyms }; From 8278e5a8ca62004c0589374cea791ed77eaf2128 Mon Sep 17 00:00:00 2001 From: William Laugesen Date: Tue, 18 Aug 2026 09:23:55 +1200 Subject: [PATCH 02/14] Add pagefind to the project dictionary Co-Authored-By: Claude Opus 5 (1M context) --- dictionary-octopus.txt | 1 + 1 file changed, 1 insertion(+) diff --git a/dictionary-octopus.txt b/dictionary-octopus.txt index a94359012f..8d53e3bf57 100644 --- a/dictionary-octopus.txt +++ b/dictionary-octopus.txt @@ -405,6 +405,7 @@ Ozar PAAS packageid packageversion +pagefind passout PathtoPublish pemstore From 02c15f87e7aed426da02416f1f1786bcc2c3b634 Mon Sep 17 00:00:00 2001 From: William Laugesen Date: Tue, 18 Aug 2026 15:15:29 +1200 Subject: [PATCH 03/14] Preload while typing, warm on page load, and set the excerpt length MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Three configuration corrections from the Pagefind audit. Its indexing and ranking were already idiomatic, so none of this touches either. preload() now runs on each keystroke ahead of the debounced search, which is what Pagefind documents as the way to fetch the chunks a query needs while the reader is still typing. Warming moves to page load for Pagefind only. That is 118KB of runtime and WASM against a cold first result measured at 2239ms. Which side of that trade is right differs by engine, so `SearchEngine` carries an `eager` flag and the engine states its own answer rather than the overlay assuming one — Orama pays for its whole index on every navigation and must stay lazy. excerptLength drops from its 30-word default to 20, to fit the single line the result row gives it. Sub-results are deliberately not used. They render heading-scoped matches with their own anchors, which is real, but the corpus fights it: the CLI tree has 511 headings across only 49 distinct titles, 213 of them "Learn more" and 176 "Examples", and the API tree 291 across 100. Only documentation pages are distinctive, at 4160 titles across 5302 headings. Suppressing the boilerplate needs a threshold nobody has justified yet, and the audit is clear that sub-results move neither Success@1 nor Success@5 — so this stays out of a comparison it cannot decide, and can be added later against whichever engine wins. Co-Authored-By: Claude Opus 5 (1M context) --- src/scripts/docs-search.ts | 11 ++++++++++- src/scripts/search-engine-pagefind.ts | 22 +++++++++++++++++++++- src/scripts/search-engine.ts | 13 +++++++++++++ 3 files changed, 44 insertions(+), 2 deletions(-) diff --git a/src/scripts/docs-search.ts b/src/scripts/docs-search.ts index be62b753a8..c24299a665 100644 --- a/src/scripts/docs-search.ts +++ b/src/scripts/docs-search.ts @@ -212,7 +212,12 @@ function setup(dialog: HTMLDialogElement) { // --- Inside the dialog --------------------------------------------------- - input.addEventListener('input', schedule); + input.addEventListener('input', () => { + // Ahead of the debounce, so the chunks this query needs are already on + // their way by the time the search runs. + engine.preload?.(input!.value); + schedule(); + }); dialog.addEventListener('keydown', (event) => { // Up and Down only. Home and End belong to the query, which is what an @@ -328,6 +333,10 @@ function setup(dialog: HTMLDialogElement) { open(); }); + // Engines that are cheap to start say so; the rest wait for the overlay. See + // `eager` on `SearchEngine` for why this is not the same answer for both. + if (engine.eager) engine.warm?.(); + // A shared `?q=` link opens straight into results. const q = new URLSearchParams(window.location.search).get('q'); if (q) open(q); diff --git a/src/scripts/search-engine-pagefind.ts b/src/scripts/search-engine-pagefind.ts index bdfbeb3eae..f07e9ff3ed 100644 --- a/src/scripts/search-engine-pagefind.ts +++ b/src/scripts/search-engine-pagefind.ts @@ -29,6 +29,10 @@ type PagefindResultStub = { data(): Promise }; type PagefindApi = { options(config: Record): Promise; filters(): Promise>>; + preload( + query: string, + options?: { filters?: Record } + ): void; search( query: string, options?: { filters?: Record } @@ -48,7 +52,12 @@ export function pagefindEngine(bundlePath: string): SearchEngine { // Pagefind resolves its chunk URLs against this, and prepends it to // every result URL; the site is proxied under /docs/ rather than served // from the root, and the index is built relative to that same prefix. - await api.options({ basePath: bundlePath }); + await api.options({ + basePath: bundlePath, + // The default is 30 words, which overruns the single line the result + // row gives it. Sized to the row instead. + excerptLength: 20, + }); // The filter index is a separate chunk, and a search returns empty // filter counts until it has been pulled down. Loading it here rather // than per-search means the tab strip is populated on the first result. @@ -65,8 +74,19 @@ export function pagefindEngine(bundlePath: string): SearchEngine { } return { + // 118KB of runtime and WASM, and it shortens the cold path to the first + // result. The per-query chunks are still fetched on demand. + eager: true, warm: load, + // Pulls the chunks this query needs while the reader is still typing. + // Without it every keystroke pays for its own chunk fetches. + preload(query) { + const term = query.trim(); + if (!term) return; + void load().then((api) => api?.preload(term)); + }, + async search(rawQuery, facet) { const query = rawQuery.trim(); const empty = { results: [], counts: { all: 0 } }; diff --git a/src/scripts/search-engine.ts b/src/scripts/search-engine.ts index c8d1800b2b..cdc8744b41 100644 --- a/src/scripts/search-engine.ts +++ b/src/scripts/search-engine.ts @@ -22,6 +22,19 @@ export type SearchEngine = { search(query: string, facet?: string): Promise; /** Optional: start loading the index before the first query needs it. */ warm?(): void; + /** + * Whether `warm()` is cheap enough to run on page load rather than when the + * overlay opens. Pagefind pays about 118KB for its runtime and fetches the + * rest per query; Orama pays for its whole index parsed and restored, on + * every navigation, for every visitor including the majority who never + * search. The answer differs by engine, so the engine states it. + */ + eager?: boolean; + /** + * Optional: fetch what this query will need without running it. Called while + * the reader is still typing, ahead of the debounced search. + */ + preload?(query: string): void; }; export type Facet = { key: string; label: string }; From 5b663677730936e125db44f54e03c1f8f831dcd1 Mon Sep 17 00:00:00 2001 From: William Laugesen Date: Wed, 19 Aug 2026 09:55:20 +1200 Subject: [PATCH 04/14] Use the rest of what Pagefind offers MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Six capabilities that were available and unused. Indexing. A depth-derived data-pagefind-weight on the page header, so a bare section name ranks the section above the pages inside it — the largest single source of missed traffic on the search terms readers actually type. Image alt text becomes searchable via data-pagefind-index-attrs, which has no inheritance and so needs a hast plugin to reach every markdown image. A date sort key is indexed, and a frontmatter title is registered as fallback metadata for any page whose heading comes back empty. Querying. highlightParam so result links carry the query for the destination page to highlight, and sub_results rendered as their own options beneath their page, so a match inside a long page can be arrowed onto and lands on that heading. The weight is the one worth measuring: applied as a client-side reorder it took Pagefind from 42% to 62% weighted Success@5 on real search terms, and from 73% to 95% on top-visited pages. At index time it has no top-30 reach limit. Co-Authored-By: Claude Opus 5 (1M context) --- astro.config.mjs | 3 ++ src/components/ArticleHeader.astro | 7 ++-- src/components/DocsSearch.astro | 48 +++++++++++++++++++++++++ src/components/Image.astro | 7 +++- src/layouts/Default.astro | 40 ++++++++++++++++++++- src/plugins/pagefind-image-attrs.js | 39 +++++++++++++++++++++ src/scripts/docs-search.ts | 50 +++++++++++++++++++++++++-- src/scripts/search-engine-pagefind.ts | 44 +++++++++++++++++++++++ src/scripts/search-engine.ts | 16 +++++++++ 9 files changed, 248 insertions(+), 6 deletions(-) create mode 100644 src/plugins/pagefind-image-attrs.js diff --git a/astro.config.mjs b/astro.config.mjs index c8625946b3..c340c496d1 100644 --- a/astro.config.mjs +++ b/astro.config.mjs @@ -9,6 +9,7 @@ import satteriHeadingId from './src/plugins/satteri-heading-id.js'; import satteriApiExamples, { apiExampleDirective } from './src/plugins/satteri-api-examples.js'; import { endpointDirective } from './src/plugins/satteri-endpoint.js'; import satteriWbr from './src/plugins/satteri-wbr.js'; +import pagefindImageAttrs from './src/plugins/pagefind-image-attrs.js'; import shikiCodeBlock from './src/plugins/shiki-code-block.js'; // https://astro.build/config @@ -67,6 +68,8 @@ export default defineConfig({ ], hastPlugins: [ satteriWbr, + // Marks image alt text for Pagefind to index. Inert without it. + pagefindImageAttrs, satteriApiExamples ], }), diff --git a/src/components/ArticleHeader.astro b/src/components/ArticleHeader.astro index 85f387e823..e7f5349c96 100644 --- a/src/components/ArticleHeader.astro +++ b/src/components/ArticleHeader.astro @@ -12,8 +12,11 @@ type Props = { lang: string; subtitle?: string | null; title: string; + /** Passed to `data-pagefind-weight`. Omitted leaves Pagefind's own heading + * weighting alone, which is what every page outside the docs layout wants. */ + searchWeight?: number; }; -const { lang, subtitle, title } = Astro.props satisfies Props; +const { lang, subtitle, title, searchWeight } = Astro.props satisfies Props; // Language const _ = Lang(lang); @@ -22,7 +25,7 @@ const _ = Lang(lang); stats.stop(); --- -
+

{title}

{subtitle &&

{subtitle}

}
diff --git a/src/components/DocsSearch.astro b/src/components/DocsSearch.astro index 10de1ccd35..091314b9ae 100644 --- a/src/components/DocsSearch.astro +++ b/src/components/DocsSearch.astro @@ -145,6 +145,21 @@ const listboxId = `${name}-search-listbox`; + { + /* A heading inside the result above that matched on its own. It is a + separate `role="option"` rather than a link inside the parent row, + because it goes somewhere different — the reader can arrow onto it and + land on that section instead of the top of the page. */ + } + +