diff --git a/src/act.js b/src/act.js index 7ada8f3..811107b 100644 --- a/src/act.js +++ b/src/act.js @@ -157,7 +157,7 @@ export function read(n, { session = DEFAULT_SESSION, budget = 2000 } = {}) { // budget it still gets cut: 'up to N tokens' is a promise the page must // not be able to break. if (!lines.length && cost > budget) { - lines.push(`${line.slice(0, budget * 4)} ... cut at ~${budget} tokens, raise --budget for the rest`); + lines.push(`${line.slice(0, whole(line, budget * 4))} ... cut at ~${budget} tokens, raise --budget for the rest`); spent += budget; continue; } @@ -211,6 +211,12 @@ export function submit(n) { const BEFORE = 60; const SNIPPET = 200; +// A character past U+FFFF (an emoji, a rare CJK ideograph) is two UTF-16 +// units, and an edge between them prints half of one, which a terminal shows +// as the replacement character. The compact view's cap steps back for the +// same reason (render.js); a snippet's edges and read's cut do the same. +const whole = (s, i) => (/[\uD800-\uDBFF]/.test(s[i - 1] ?? '') ? i - 1 : i); + /** * Where a string appears on the current page. This is the answer to "the page * is long and I only care about one thing in it": one command, no fetch, and @@ -307,8 +313,9 @@ function search(blocks, terms) { // One number, one line: a run of short blocks under the same handle would // otherwise report the same place several times. if (out.length && out.at(-1).n === n) continue; - const start = Math.max(0, Math.min(...found) - BEFORE); - const end = Math.min(block.text.length, start + SNIPPET); + const from = Math.max(0, Math.min(...found) - BEFORE); + const start = whole(block.text, from); + const end = whole(block.text, Math.min(block.text.length, from + SNIPPET)); // One line per match is the promise the list makes, and a code block is the // one kind of block that carries lines of its own. They survive where they // are read rather than indexed: the whole-match mode above, and `read`. diff --git a/tests/act.test.js b/tests/act.test.js index be96b89..fc38073 100644 --- a/tests/act.test.js +++ b/tests/act.test.js @@ -212,6 +212,32 @@ test('find opens the snippet on the match, not on the start of a long block', () assert.ok(!out.includes('x'.repeat(250)), `snippet was not trimmed:\n${out.slice(0, 200)}`); }); +test('neither edge of a find snippet nor a cut read splits an emoji', () => { + // An emoji is two UTF-16 units. The snippet window opens 60 units before + // the match and runs 200, so here it opened inside the first 😀 and closed + // inside the second, and printed a lone half of each. + const text = `ab😀${'x'.repeat(59)}needle${'y'.repeat(133)}😀${'z'.repeat(50)}`; + saveSession('emoji', { + url: 'https://example.test/emoji', + blocks: [1, 2, 3].map((n) => ({ n, type: 'text', text: n === 1 ? text : `needle ${'w'.repeat(1000)}` })), + cursor: null, + }); + const out = find('needle', { session: 'emoji', budget: 40 }); + const line = out.split('\n')[1]; + assert.ok(line.isWellFormed(), `half an emoji in ${JSON.stringify(line)}`); + // The opening edge takes the whole 😀, the closing one leaves it out. + assert.ok(line.startsWith('[1] ... 😀x') && line.endsWith('y ...'), `the window moved: ${line}`); + // read cuts a block over its whole budget at budget * 4 characters; the + // 😀 here straddles character 40, after the "[1] " tag. + saveSession('emoji-read', { + url: 'https://example.test/emoji', + blocks: [{ n: 1, type: 'text', text: `${'x'.repeat(35)}😀${'y'.repeat(100)}` }], + cursor: null, + }); + const cut = read(1, { session: 'emoji-read', budget: 10 }); + assert.ok(cut.isWellFormed(), `half an emoji in ${JSON.stringify(cut)}`); +}); + test('find answers with the whole match when the matches fit', () => { open(); // The point of the whole path: the text an agent would have spent a `read