From d5e85aa5a675ec44f39e7eedbe19d8826c7bbc0c Mon Sep 17 00:00:00 2001 From: kevin9327 <5299031+kevin9327@users.noreply.github.com> Date: Mon, 28 Sep 2026 20:05:19 +0900 Subject: [PATCH] Cut an MkDocs query into words the way titles and bodies are 'oc astral uv pyproject.toml' answered that nothing in the docs' own index matches, though uv has a section titled "The pyproject.toml". The query was split on spaces only, and titles and bodies at every character outside [a-z0-9_], so a query word holding a dot or a dash (pyproject.toml, uv.lock, --group) could never equal a word of the index, and a question's closing "?" left its last word unmatched, which dropped "how do I add a dependency?" to the partial-match list. An accented word was cut apart on both sides and matched nothing. The query, titles, and bodies now go through one wordsOf, which cuts at any character that is not a letter, digit, or underscore in any script, and drops a contraction's tail first so "what's" is "what" rather than "what" and a stray "s" every section would have to match. Co-Authored-By: Claude Opus 5.5 --- src/mkdocs.js | 11 +++++++++-- tests/mkdocs.test.js | 23 +++++++++++++++++++++++ 2 files changed, 32 insertions(+), 2 deletions(-) diff --git a/src/mkdocs.js b/src/mkdocs.js index 59308d2..8fa3fed 100644 --- a/src/mkdocs.js +++ b/src/mkdocs.js @@ -69,7 +69,14 @@ const STOP = new Set(['a', 'an', 'the', 'to', 'of', 'in', 'on', 'for', 'and', 'o // "dependency" should find "Managing dependencies", so a word also matches // by its stem once a plural or verb ending is off. const stem = (w) => (w.length > 4 ? w.replace(/(ies|es|s|y|ing|ed)$/, '') : w); -const tokensOf = (s) => new Set(s.split(/[^a-z0-9_]+/)); +// A title, a body, and the query are all cut into words at the same places, +// so "pyproject.toml", "uv.lock", "--group", or a question's closing "?" +// meet the words the index holds, and a letter outside ASCII stays in its +// word. A contraction's tail goes first: "what's" is "what", not "what" and +// a stray "s" every section would have to match. +const wordsOf = (s) => s.replace(/(?<=\p{L})['’](s|t|re|ve|ll|d|m)(?!\p{L})/gu, '') + .split(/[^\p{L}\p{N}_]+/u).filter(Boolean); +const tokensOf = (s) => new Set(wordsOf(s)); // A stem matches a word it starts, give or take an ending: "add" finds // "adding" but not "additional". const rooted = (tokens, root) => [...tokens].some((t) => t.startsWith(root) && t.length - root.length <= 3); @@ -85,7 +92,7 @@ const rooted = (tokens, root) => [...tokens].some((t) => t.startsWith(root) && t * @param {string} query */ export function searchMkdocs(entries, query) { - const all = [...new Set(query.toLowerCase().split(/\s+/).filter(Boolean))]; + const all = [...new Set(wordsOf(query.toLowerCase()))]; const topical = all.filter((w) => !STOP.has(w)); const words = topical.length ? topical : all; const scored = []; diff --git a/tests/mkdocs.test.js b/tests/mkdocs.test.js index 8c6c60e..d355f1d 100644 --- a/tests/mkdocs.test.js +++ b/tests/mkdocs.test.js @@ -63,6 +63,29 @@ test('when no section holds every word, the best partial matches say so', () => assert.equal(searchMkdocs(entries(), 'xyzzy').total, 0); }); +test('a query word is cut into words the way titles and bodies are', () => { + // The query was split on spaces only while titles and bodies were split + // at every non-word character, so "pyproject.toml" could never equal a + // word of the index and found nothing, a question's closing "?" left its + // last word unmatched, and an accented word matched no title at all. + const docs = buildMkdocsEntries(parseMkdocsIndex(JSON.stringify({ docs: [ + { location: 'concepts/projects/layout/', title: 'Project structure and files', text: 'The files uv reads and writes.' }, + { location: 'concepts/projects/layout/#the-pyprojecttoml', title: 'The pyproject.toml', text: 'Project metadata is defined in a pyproject.toml file.' }, + { location: 'concepts/projects/layout/#the-lockfile', title: 'The lockfile', text: 'uv creates a uv.lock file next to the pyproject.toml.' }, + { location: 'concepts/projects/dependencies/#dependency-groups', title: 'Dependency groups', text: 'Use --group to add a dependency to a group.' }, + { location: 'fr/', title: 'Démarrage rapide', text: 'Créer un projet.' }, + ] }))); + const first = (query) => searchMkdocs(docs, query).hits[0]?.text; + assert.equal(first('pyproject.toml'), 'The pyproject.toml'); + assert.equal(first('what is uv.lock?'), 'The lockfile'); + assert.equal(first('--group'), 'Dependency groups'); + assert.equal(first('démarrage'), 'Démarrage rapide'); + const question = searchMkdocs(docs, "what's a dependency group?"); + assert.deepEqual(question.words, ['dependency', 'group']); + assert.equal(question.partial, false); + assert.equal(question.hits[0].text, 'Dependency groups'); +}); + test('the results page links each section, keeps repeated titles apart, and escapes', () => { const html = resultsToHTML(BASE, 'import unused', searchMkdocs(entries(), 'import unused')); assert.match(html, /Example \(rule-a \(A001\)\)<\/a> section: import os <unused>/);