diff --git a/src/mkdocs.js b/src/mkdocs.js index 59308d2..8fa3fed 100644 --- a/src/mkdocs.js +++ b/src/mkdocs.js @@ -69,7 +69,14 @@ const STOP = new Set(['a', 'an', 'the', 'to', 'of', 'in', 'on', 'for', 'and', 'o // "dependency" should find "Managing dependencies", so a word also matches // by its stem once a plural or verb ending is off. const stem = (w) => (w.length > 4 ? w.replace(/(ies|es|s|y|ing|ed)$/, '') : w); -const tokensOf = (s) => new Set(s.split(/[^a-z0-9_]+/)); +// A title, a body, and the query are all cut into words at the same places, +// so "pyproject.toml", "uv.lock", "--group", or a question's closing "?" +// meet the words the index holds, and a letter outside ASCII stays in its +// word. A contraction's tail goes first: "what's" is "what", not "what" and +// a stray "s" every section would have to match. +const wordsOf = (s) => s.replace(/(?<=\p{L})['’](s|t|re|ve|ll|d|m)(?!\p{L})/gu, '') + .split(/[^\p{L}\p{N}_]+/u).filter(Boolean); +const tokensOf = (s) => new Set(wordsOf(s)); // A stem matches a word it starts, give or take an ending: "add" finds // "adding" but not "additional". const rooted = (tokens, root) => [...tokens].some((t) => t.startsWith(root) && t.length - root.length <= 3); @@ -85,7 +92,7 @@ const rooted = (tokens, root) => [...tokens].some((t) => t.startsWith(root) && t * @param {string} query */ export function searchMkdocs(entries, query) { - const all = [...new Set(query.toLowerCase().split(/\s+/).filter(Boolean))]; + const all = [...new Set(wordsOf(query.toLowerCase()))]; const topical = all.filter((w) => !STOP.has(w)); const words = topical.length ? topical : all; const scored = []; diff --git a/tests/mkdocs.test.js b/tests/mkdocs.test.js index 8c6c60e..d355f1d 100644 --- a/tests/mkdocs.test.js +++ b/tests/mkdocs.test.js @@ -63,6 +63,29 @@ test('when no section holds every word, the best partial matches say so', () => assert.equal(searchMkdocs(entries(), 'xyzzy').total, 0); }); +test('a query word is cut into words the way titles and bodies are', () => { + // The query was split on spaces only while titles and bodies were split + // at every non-word character, so "pyproject.toml" could never equal a + // word of the index and found nothing, a question's closing "?" left its + // last word unmatched, and an accented word matched no title at all. + const docs = buildMkdocsEntries(parseMkdocsIndex(JSON.stringify({ docs: [ + { location: 'concepts/projects/layout/', title: 'Project structure and files', text: 'The files uv reads and writes.' }, + { location: 'concepts/projects/layout/#the-pyprojecttoml', title: 'The pyproject.toml', text: 'Project metadata is defined in a pyproject.toml file.' }, + { location: 'concepts/projects/layout/#the-lockfile', title: 'The lockfile', text: 'uv creates a uv.lock file next to the pyproject.toml.' }, + { location: 'concepts/projects/dependencies/#dependency-groups', title: 'Dependency groups', text: 'Use --group to add a dependency to a group.' }, + { location: 'fr/', title: 'Démarrage rapide', text: 'Créer un projet.' }, + ] }))); + const first = (query) => searchMkdocs(docs, query).hits[0]?.text; + assert.equal(first('pyproject.toml'), 'The pyproject.toml'); + assert.equal(first('what is uv.lock?'), 'The lockfile'); + assert.equal(first('--group'), 'Dependency groups'); + assert.equal(first('démarrage'), 'Démarrage rapide'); + const question = searchMkdocs(docs, "what's a dependency group?"); + assert.deepEqual(question.words, ['dependency', 'group']); + assert.equal(question.partial, false); + assert.equal(question.hits[0].text, 'Dependency groups'); +}); + test('the results page links each section, keeps repeated titles apart, and escapes', () => { const html = resultsToHTML(BASE, 'import unused', searchMkdocs(entries(), 'import unused')); assert.match(html, /Example \(rule-a \(A001\)\)<\/a> section: import os <unused>/);