From 85e78a15a3a8b5f77a42428ae1efd5f2d14128ae Mon Sep 17 00:00:00 2001 From: Danny Avila Date: Fri, 2 Oct 2026 21:38:28 -0400 Subject: [PATCH 1/4] =?UTF-8?q?=F0=9F=93=8D=20feat:=20Quote=20the=20Curren?= =?UTF-8?q?t=20Text=20Where=20a=20Missing=20Workspace=20Edit=20Belongs?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A missing edit's diagnosis now ends with a bounded excerpt of the region old_text most likely meant: the alignment where most of its distinctive lines fall into place (found in one pass with bounded votes), else its unique first line or closest line. Each of at most 8 rows is `|`, with a space for identical lines, `~` for whitespace-only differences and `!` for changed text; lines are shortened to 160 characters and the excerpt to 1,600. The excerpt is a final `; the current text at lines A-B (...) is ""` hint, so hosts that parse the existing grammar drop it and keep every other hint. Excerpts are the first detail dropped to fit the 3,000-character message bound and never shorten another edit's reason. --- packages/code/README.md | 10 +- packages/code/src/edits.test.ts | 153 ++++++++++++++++++++++++++ packages/code/src/edits.ts | 184 ++++++++++++++++++++++++++++---- 3 files changed, 326 insertions(+), 21 deletions(-) diff --git a/packages/code/README.md b/packages/code/README.md index 38a22d5d..ec38033b 100644 --- a/packages/code/README.md +++ b/packages/code/README.md @@ -640,7 +640,15 @@ fails, the worker still checks the rest and rejects the whole batch with one batch messages shorten reasons and source excerpts to stay within the bound, but never omit failing edit positions. A missing edit names the nearest candidate line and flags elided (`...`) or line-numbered `oldText`, a -whitespace-only difference, or CRLF line endings. +whitespace-only difference, or CRLF line endings. When the file has a region +that `oldText` most likely meant (the alignment where the most of its +distinctive lines fall into place, or its unique first line or closest line), +the reason ends with `the current text at lines A-B (~ whitespace differs, ! +text differs) is ""`: a JSON string of at most 8 rows of the form +`|`, each line shortened to 160 characters, where the mark is +a space for a line identical to `oldText` at that position, `~` when only its +whitespace differs and `!` when its text differs. Excerpts are the first detail +dropped to fit the message bound, and never shorten another edit's reason. An ambiguous edit gives its match count and line numbers. Overlapping occurrences count as separate locations. Detailed source-line excerpts require `read_file` or `preview_edit` on the same workspace; edit-only workers return a diff --git a/packages/code/src/edits.test.ts b/packages/code/src/edits.test.ts index 3ef9428d..8b27a222 100644 --- a/packages/code/src/edits.test.ts +++ b/packages/code/src/edits.test.ts @@ -719,3 +719,156 @@ test('short batch diagnostic reasons remain complete when the message fits', () assert.ok(error.message.includes(`\nEdit ${failure.index + 1}: ${failure.reason}.`)); } }); + +const EXCERPT = /the current text at lines? (\d+)(?:-(\d+))? \(~ whitespace differs, ! text differs\) is ("(?:[^"\\]|\\.)*")/; + +function excerptOf(message: string): { first: number; last: number; rows: string[] } | undefined { + const match = EXCERPT.exec(message); + if (!match) return undefined; + const first = Number(match[1]); + return { first, last: Number(match[2] ?? first), rows: (JSON.parse(match[3]) as string).split('\n') }; +} + +test('a missing edit quotes the current text of the region it meant, marking what differs', () => { + const text = 'export function load(user) {\n const id = user.id;\n return fetchUser(id);\n}\n'; + const error = rejection(() => applyTextEdits(text, [{ + oldText: 'export function load(user) {\n const id = user.id;\n return fetchUser(user);\n}', + newText: 'x', + }], 'tolerant')); + assert.match(error.message, /old_text was not found; its first line appears at line 1, but the lines after it differ; the current text/); + assert.deepEqual(excerptOf(error.message), { + first: 1, + last: 4, + rows: [ + '1| export function load(user) {', + '2|~ const id = user.id;', + '3|! return fetchUser(id);', + '4| }', + ], + }); + assert.match(error.message, /\.$/); +}); + +test('the region is found from later lines when the first line itself changed', () => { + const text = 'header();\nfunction save(record) {\n validate(record);\n persist(record);\n}\nfooter();\n'; + const error = rejection(() => applyTextEdits(text, [{ + oldText: 'function save(item) {\n validate(record);\n persist(record);\n}', + newText: 'x', + }])); + assert.deepEqual(excerptOf(error.message), { + first: 2, + last: 5, + rows: ['2|!function save(record) {', '3| validate(record);', '4| persist(record);', '5| }'], + }); +}); + +test('the region with the most aligned lines wins, and ties prefer the earliest', () => { + const block = (name: string) => `function ${name}() {\n prepare();\n run();\n cleanup();\n}\n`; + const text = `${block('first')}${block('second')}`; + const error = rejection(() => applyTextEdits(text, [{ + oldText: 'function second() {\n prepare();\n execute();\n cleanup();\n}', + newText: 'x', + }])); + assert.equal(excerptOf(error.message)?.first, 6); + const tied = rejection(() => applyTextEdits(text, [{ + oldText: 'function third() {\n prepare();\n execute();\n cleanup();\n}', + newText: 'x', + }])); + assert.equal(excerptOf(tied.message)?.first, 1); +}); + +test('a missing one-line edit quotes its closest line', () => { + const error = rejection(() => applyTextEdits('const total = items.length;\nreturn total;\n', [ + { oldText: 'const total = items.size;', newText: 'x' }, + ])); + assert.match(error.message, /the closest line is line 1: "const total = items.length;"/); + assert.deepEqual(excerptOf(error.message), { first: 1, last: 1, rows: ['1|!const total = items.length;'] }); +}); + +test('exact mode quotes the whitespace to copy alongside the whitespace hint', () => { + const error = rejection(() => applyTextEdits('class A {\n a();\n b();\n}\n', [ + { oldText: 'a();\nb();', newText: 'x' }, + ])); + assert.match(error.message, /exists at line 2 with different whitespace \(indentation-flexible\); copy that whitespace exactly; the current text at lines 2-3/); + assert.deepEqual(excerptOf(error.message)?.rows, ['2|~ a();', '3|~ b();']); +}); + +test('a long region is quoted from just before its first difference', () => { + const source = Array.from({ length: 40 }, (_, index) => `step(${index});`); + const needle = source.slice(5, 25); + needle[8] = 'step(changed);'; + const error = rejection(() => applyTextEdits(`${source.join('\n')}\n`, [{ oldText: needle.join('\n'), newText: 'x' }])); + const excerpt = excerptOf(error.message); + assert.equal(excerpt?.first, 13); + assert.equal(excerpt?.last, 20); + assert.equal(excerpt?.rows[1], '14|!step(13);'); + assert.ok(excerpt?.rows.every((row, index) => index === 1 || row.includes('| '))); +}); + +test('excerpts shorten long lines and stay bounded without splitting characters', () => { + const wide = `x${'😀'.repeat(400)}`; + const controls = '\u0001'.repeat(2_000); + const text = Array.from({ length: 10 }, (_, index) => `key_${index} = ${index % 2 === 0 ? wide : controls}`).join('\n'); + const oldText = Array.from({ length: 10 }, (_, index) => `key_${index} = other`).join('\n'); + const error = rejection(() => applyTextEdits(text, [{ oldText, newText: 'x' }])); + const excerpt = excerptOf(error.message); + assert.ok(excerpt != null); + assert.ok(excerpt.rows.length >= 1 && excerpt.rows.length <= 8); + assert.ok(EXCERPT.exec(error.message)![0].length <= 1_600); + for (const row of excerpt.rows) { + assert.ok(row.length <= 175, row.slice(0, 40)); + assert.equal(Buffer.from(row).toString('utf8'), row); + } +}); + +test('an edit that resembles nothing in the file gets no excerpt', () => { + const error = rejection(() => applyTextEdits('alpha\nbeta\n', [ + { oldText: 'completely different text\nwith nothing shared', newText: 'x' }, + ])); + assert.equal(excerptOf(error.message), undefined); + assert.match(error.message, /old_text was not found\.$/); +}); + +test('ambiguous and line-numbered edits keep their messages without an excerpt', () => { + const ambiguous = rejection(() => applyTextEdits('return a;\nreturn a;\n', [{ oldText: 'return a;', newText: 'x' }])); + assert.equal(excerptOf(ambiguous.message), undefined); + const numbered = rejection(() => applyTextEdits('start\nmiddle\nend\n', [{ oldText: '1 | start\n2 | middle', newText: 'x' }])); + assert.equal(excerptOf(numbered.message), undefined); +}); + +test('batch excerpts are granted in edit order only while every position and reason still fits', () => { + const lines = Array.from({ length: 400 }, (_, index) => `value_${index} = compute_${index}(input_${index}, ${'p'.repeat(120)});`); + const edits = Array.from({ length: 12 }, (_, index) => ({ + oldText: `${lines[index * 10]}\n${lines[index * 10 + 1].replace('compute', 'changed')}\n${lines[index * 10 + 2]}`, + newText: 'x', + })); + const error = rejection(() => applyTextEdits(`${lines.join('\n')}\n`, edits)); + assert.ok(error.message.length <= EDIT_DIAGNOSTIC_MAX_CHARS); + assert.deepEqual( + [...error.message.matchAll(/\nEdit (\d+):/g)].map((match) => Number(match[1])), + Array.from({ length: 12 }, (_, index) => index + 1), + ); + const quoted = [...error.message.matchAll(/\nEdit (\d+): [^\n]*the current text at/g)].map((match) => Number(match[1])); + assert.ok(quoted.length >= 1 && quoted.length < 12); + assert.deepEqual(quoted, Array.from({ length: quoted.length }, (_, index) => index + 1)); + for (const failure of error.failures) { + assert.ok(error.message.includes(`\nEdit ${failure.index + 1}: ${failure.reason}`)); + } +}); + +test('a single edit drops its excerpt rather than exceed the bound', () => { + const error = new WorkspaceEditMatchError([ + { index: 0, reason: `old_text was not found${'; x'.repeat(1_200)}`, excerpt: 'the current text at line 1 (~ whitespace differs, ! text differs) is "1|!y"' }, + ], 1); + assert.ok(error.message.length <= EDIT_DIAGNOSTIC_MAX_CHARS); + assert.doesNotMatch(error.message, /the current text/); +}); + +test('region votes stay bounded on highly repetitive files', () => { + const text = 'item = value;\nother = thing;\n'.repeat(150_000); + const oldText = 'item = value;\nother = thing;\nmissing = line;\nitem = value;'; + const started = performance.now(); + const error = rejection(() => applyTextEdits(text, [{ oldText, newText: 'x' }])); + assert.ok(performance.now() - started < 3_000, 'a repetitive file must not cast unbounded votes'); + assert.match(error.message, /did not apply and nothing was written/); +}); diff --git a/packages/code/src/edits.ts b/packages/code/src/edits.ts index a6d54269..66271832 100644 --- a/packages/code/src/edits.ts +++ b/packages/code/src/edits.ts @@ -14,6 +14,14 @@ import type { export const EDIT_DIAGNOSTIC_MAX_CHARS = 3000; const MAX_REPORTED_LINES = 5; const MAX_SNIPPET_CHARS = 120; +/** A missing edit's current-text excerpt: a few lines, each shortened, in one bounded hint. */ +const MAX_EXCERPT_LINES = 8; +const MAX_EXCERPT_LINE_CHARS = 160; +const MAX_EXCERPT_CHARS = 1_600; +/** Alignment votes cast while locating the region a missing edit most likely meant. */ +const MAX_REGION_VOTES = 200_000; +/** Lines too generic to anchor a region on their own (`}`, `);`, blank lines). */ +const ANCHOR_LINE = /[\p{L}\p{N}_]{2}/u; /** A highly repetitive indentation candidate must not monopolize the worker. */ const MAX_LINE_WINDOW_VERIFICATIONS = 100_000; const MAX_REPLACEMENT_CHUNK_CHARS = 16 * 1024; @@ -45,6 +53,12 @@ export interface EditFailure { /** Zero-based position of the edit in the request. */ index: number; reason: string; + /** + * A final `; the current text at ...` hint quoting the region the edit most + * likely meant. It is appended to `reason` only while the whole message stays + * within its bound, and is the first detail dropped when it would not. + */ + excerpt?: string; } export class WorkspaceEditMatchError extends Error { @@ -93,14 +107,16 @@ export function applyTextEdits( matches.push({ strategy: outcome.strategy, occurrences: outcome.occurrences }); return; } + if (outcome.status === 'none') { + failures.push({ index, ...describeMissing(working, edit.oldText, matching, lines) }); + return; + } failures.push({ index, reason: outcome.status === 'ambiguous' ? describeAmbiguous(working, outcome.strategy, outcome.count, outcome.starts) - : outcome.status === 'limit' - ? 'old_text has too many repetitive line-window candidates; include more surrounding lines or use an exact match' - : describeMissing(working, edit.oldText, matching, lines), + : 'old_text has too many repetitive line-window candidates; include more surrounding lines or use an exact match', }); }); if (failures.length > 0) { @@ -604,9 +620,116 @@ function describeMissing( oldText: string, matching: WorkspaceEditMatching, lines: () => LineIndex, +): { reason: string; excerpt?: string } { + const needle = neededLines(oldText).lines; + const nearest = nearestLine(needle.find((line) => line.trim().length > 0), lines); + const reason = describeMissingReason(text, oldText, needle, matching, lines, nearest); + const anchor = nearest != null && (!nearest.exact || nearest.count === 1) ? nearest.index : undefined; + const start = closestRegion(needle, lines()) ?? anchor; + const excerpt = start == null ? undefined : currentTextExcerpt(needle, lines(), start); + return excerpt == null ? { reason } : { reason, excerpt }; +} + +type ExcerptMark = ' ' | '~' | '!'; + +/** + * Finds where a missing multi-line edit most likely belongs: the alignment on + * which the most of its distinctive lines, compared without surrounding + * whitespace, fall into place. One pass over the file with bounded votes. + */ +function closestRegion(needle: readonly string[], lines: LineIndex): number | undefined { + const offsets = new Map(); + let anchors = 0; + needle.forEach((line, offset) => { + const key = line.trim(); + if (!ANCHOR_LINE.test(key)) return; + anchors++; + const known = offsets.get(key); + if (known) known.push(offset); + else offsets.set(key, [offset]); + }); + if (anchors < 2) return undefined; + const votes = new Map(); + let budget = MAX_REGION_VOTES; + let best: number | undefined; + let bestVotes = 0; + for (let index = 0; index < lines.length && budget > 0; index++) { + const hits = offsets.get(lines.text(index).trim()); + if (!hits) continue; + for (const offset of hits) { + const start = index - offset; + if (start < 0) continue; + if (budget-- <= 0) break; + const count = (votes.get(start) ?? 0) + 1; + votes.set(start, count); + if (count > bestVotes || (count === bestVotes && best !== undefined && start < best)) { + best = start; + bestVotes = count; + } + } + } + // One shared line is too weak to claim a region; the first-line anchor covers that case. + return bestVotes >= 2 ? best : undefined; +} + +function excerptMark(current: string, expected: string | undefined): ExcerptMark { + if (expected === undefined) return '!'; + if (current === expected) return ' '; + return current.trim() === expected.trim() ? '~' : '!'; +} + +function shortenLine(line: string): string { + if (line.length <= MAX_EXCERPT_LINE_CHARS) return line; + let end = MAX_EXCERPT_LINE_CHARS; + if (/[\uD800-\uDBFF]/.test(line[end - 1]) && /[\uDC00-\uDFFF]/.test(line[end])) end--; + return `${line.slice(0, end)}…`; +} + +/** + * Quotes the current text where a missing edit most likely belongs, so the next + * attempt can copy it instead of re-reading the file. Each line is marked + * against the `old_text` line at the same position: a space when identical, + * `~` when only whitespace differs and `!` when the text differs. A region + * longer than the excerpt is shown from just before its first difference. + */ +function currentTextExcerpt(needle: readonly string[], lines: LineIndex, start: number): string | undefined { + const span = Math.max(1, Math.min(needle.length, lines.length - start)); + const marks = Array.from({ length: span }, (_, offset) => + excerptMark(lines.text(start + offset), needle[offset]), + ); + const shown = Math.min(span, MAX_EXCERPT_LINES); + const firstDifference = Math.max(0, marks.findIndex((mark) => mark !== ' ')); + const first = start + Math.min(Math.max(0, firstDifference - 1), span - shown); + const rows = marks + .slice(first - start, first - start + shown) + .map((mark, offset) => `${first + offset + 1}|${mark}${shortenLine(lines.text(first + offset))}`); + for (; rows.length > 0; rows.pop()) { + const where = rows.length === 1 ? `line ${first + 1}` : `lines ${first + 1}-${first + rows.length}`; + const excerpt = `the current text at ${where} (~ whitespace differs, ! text differs) is ${JSON.stringify(rows.join('\n'))}`; + if (excerpt.length <= MAX_EXCERPT_CHARS) return excerpt; + } + return undefined; +} + +interface NearestLine { + exact: boolean; + count: number; + /** Line index of the first exact candidate, or of the best token match. */ + index: number; + starts: number[]; + text: string; +} + +function describeMissingReason( + text: string, + oldText: string, + needle: readonly string[], + matching: WorkspaceEditMatching, + lines: () => LineIndex, + nearest: NearestLine | undefined, ): string { const hints: string[] = []; - const nonBlank = neededLines(oldText).lines.filter((line) => line.trim().length > 0); + const nonBlank = needle.filter((line) => line.trim().length > 0); if (nonBlank.some((line) => ELISION_LINE.test(line))) { hints.push('it contains an elision placeholder ("..."); copy the exact lines instead of abbreviating'); } @@ -629,7 +752,6 @@ function describeMissing( hints.push('the file uses CRLF line endings'); } } - const nearest = nearestLine(nonBlank[0], lines); if (nearest != null && hints.length === 0) { hints.push( nearest.exact @@ -644,45 +766,65 @@ function describeMissing( function nearestLine( firstLine: string | undefined, getLines: () => LineIndex, -): { exact: boolean; count: number; starts: number[]; text: string } | undefined { +): NearestLine | undefined { const target = firstLine?.trim(); if (!target) return undefined; const lines = getLines(); const starts: number[] = []; let count = 0; - let firstMatch = ''; + let first = 0; for (let index = 0; index < lines.length; index++) { - const content = lines.text(index); - if (content.trim() !== target) continue; - if (count++ === 0) firstMatch = content; + if (lines.text(index).trim() !== target) continue; + if (count++ === 0) first = index; if (starts.length < MAX_REPORTED_LINES) starts.push(lines.start(index)); } if (count > 0) { - return { exact: true, count, starts, text: firstMatch }; + return { exact: true, count, index: first, starts, text: lines.text(first) }; } const tokens = new Set(target.split(/\W+/).filter((token) => token.length > 1)); if (tokens.size < 2) return undefined; - let best: Line | undefined; + let best: number | undefined; let bestScore = 0; for (let index = 0; index < lines.length; index++) { - const content = lines.text(index); let score = 0; - for (const token of new Set(content.split(/\W+/))) { + for (const token of new Set(lines.text(index).split(/\W+/))) { if (tokens.has(token)) score++; } if (score > bestScore) { - best = { start: lines.start(index), end: lines.end(index), next: lines.next(index), text: content }; + best = index; bestScore = score; } } return best != null && bestScore / tokens.size >= 0.5 - ? { exact: false, count: 1, starts: [best.start], text: best.text } + ? { exact: false, count: 1, index: best, starts: [lines.start(best)], text: lines.text(best) } : undefined; } +/** A failure's reason, followed by its excerpt when the excerpt was granted room. */ +function failureText(failure: EditFailure, withExcerpt: boolean): string { + return withExcerpt && failure.excerpt ? `${failure.reason}; ${failure.excerpt}` : failure.reason; +} + +/** + * Grants excerpts in edit order while the message still fits. Excerpts are the + * most expendable detail, so granting one never shortens another reason. + */ +function grantExcerpts(failures: readonly EditFailure[], available: number): boolean[] { + let remaining = available - failures.reduce((total, failure) => total + failure.reason.length, 0); + return failures.map((failure) => { + const cost = failure.excerpt ? failure.excerpt.length + 2 : 0; + if (cost === 0 || cost > remaining) return false; + remaining -= cost; + return true; + }); +} + function formatEditFailures(failures: readonly EditFailure[], editCount: number): string { if (editCount === 1) { - return `Workspace edit did not apply and nothing was written: ${failures[0]?.reason ?? 'no match'}.`.slice( + const single = (withExcerpt: boolean) => + `Workspace edit did not apply and nothing was written: ${failures[0] ? failureText(failures[0], withExcerpt) : 'no match'}.`; + const detailed = single(true); + return (detailed.length <= EDIT_DIAGNOSTIC_MAX_CHARS ? detailed : single(false)).slice( 0, EDIT_DIAGNOSTIC_MAX_CHARS, ); @@ -696,10 +838,12 @@ function formatEditFailures(failures: readonly EditFailure[], editCount: number) // reasons or source excerpts. Never leave callers guessing which edits failed. const available = EDIT_DIAGNOSTIC_MAX_CHARS - header.length - footer.length - prefixes.reduce((total, prefix) => total + prefix.length + 1, 0); - const reasonLength = failures.reduce((total, failure) => total + failure.reason.length, 0); + const granted = grantExcerpts(failures, available); + const reasons = failures.map((failure, index) => failureText(failure, granted[index])); + const reasonLength = reasons.reduce((total, reason) => total + reason.length, 0); const reasonLimit = reasonLength <= available ? Infinity : Math.floor(available / failures.length); - const details = failures.map((failure, index) => { - let reason = failure.reason; + const details = reasons.map((full, index) => { + let reason = full; if (reason.length > reasonLimit) { let end = reasonLimit - 1; // Do not split a surrogate pair in a shortened source excerpt. From 104b8e301c7aa24818dd40d4f633595f84aceaf1 Mon Sep 17 00:00:00 2001 From: Danny Avila Date: Fri, 2 Oct 2026 21:48:05 -0400 Subject: [PATCH 2/4] =?UTF-8?q?=F0=9F=A7=AD=20fix:=20Bound=20Every=20Align?= =?UTF-8?q?ment=20Offset=20and=20Anchor=20Excerpts=20Where=20old=5Ftext=20?= =?UTF-8?q?Starts?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - Count every inspected anchor offset against the region vote budget, and stop at the first offset that would start before the file (offsets ascend), so a repetitive old_text cannot escape the bound. - Start a first-line or closest-line excerpt at the line where old_text starts, not at its first non-blank line. - Escape U+2028/U+2029 in quoted text, which JSON.stringify leaves raw, so each batch edit stays on one line for hosts that read the message line by line. --- packages/code/src/edits.test.ts | 34 +++++++++++++++++++++++++++++++++ packages/code/src/edits.ts | 32 +++++++++++++++++++++++-------- 2 files changed, 58 insertions(+), 8 deletions(-) diff --git a/packages/code/src/edits.test.ts b/packages/code/src/edits.test.ts index 8b27a222..48c059e8 100644 --- a/packages/code/src/edits.test.ts +++ b/packages/code/src/edits.test.ts @@ -872,3 +872,37 @@ test('region votes stay bounded on highly repetitive files', () => { assert.ok(performance.now() - started < 3_000, 'a repetitive file must not cast unbounded votes'); assert.match(error.message, /did not apply and nothing was written/); }); + +test('an excerpt anchored on a first line after blank lines starts where old_text starts', () => { + const error = rejection(() => applyTextEdits('header();\nconst total = items.length;\nreturn total;\n', [ + { oldText: '\nconst total = items.size;', newText: 'x' }, + ])); + assert.deepEqual(excerptOf(error.message), { + first: 1, + last: 2, + rows: ['1|!header();', '2|!const total = items.length;'], + }); +}); + +test('region votes count every inspected anchor offset against the budget', () => { + const oldText = `${'repeat_line();\n'.repeat(5_000)}missing();`; + const text = 'repeat_line();\n'.repeat(20_000); + const started = performance.now(); + const error = rejection(() => applyTextEdits(text, [{ oldText, newText: 'x' }])); + assert.ok(performance.now() - started < 3_000, 'repeated anchor offsets must not escape the vote budget'); + assert.match(error.message, /did not apply and nothing was written/); +}); + +test('quoted text escapes line separators so every batch edit stays on one line', () => { + const separator = String.fromCharCode(0x2028); + const text = `const label = "first${separator}second";\nconst other = "value";\n`; + const error = rejection(() => applyTextEdits(text, [ + { oldText: `const label = "first${separator}third";`, newText: 'x' }, + { oldText: 'const other = "changed";', newText: 'y' }, + ])); + assert.ok(!error.message.includes(separator)); + const lines = error.message.split('\n'); + assert.match(lines[1], /^Edit 1: old_text was not found; the closest line is line 1: .*\\u2028.*the current text at line 1/); + assert.match(lines[2], /^Edit 2: /); + assert.deepEqual(excerptOf(lines[1])?.rows, [`1|!const label = "first${separator}second";`]); +}); diff --git a/packages/code/src/edits.ts b/packages/code/src/edits.ts index 66271832..d3ce9799 100644 --- a/packages/code/src/edits.ts +++ b/packages/code/src/edits.ts @@ -22,6 +22,7 @@ const MAX_EXCERPT_CHARS = 1_600; const MAX_REGION_VOTES = 200_000; /** Lines too generic to anchor a region on their own (`}`, `);`, blank lines). */ const ANCHOR_LINE = /[\p{L}\p{N}_]{2}/u; +const LINE_SEPARATORS = /[\u2028\u2029]/g; /** A highly repetitive indentation candidate must not monopolize the worker. */ const MAX_LINE_WINDOW_VERIFICATIONS = 100_000; const MAX_REPLACEMENT_CHUNK_CHARS = 16 * 1024; @@ -605,11 +606,21 @@ function describeAmbiguous( return `old_text matched ${count} locations${how} at ${formatLineList(text, starts, count)}; include more surrounding lines so it matches exactly one`; } +/** + * JSON quoting that also escapes U+2028/U+2029, which JSON.stringify leaves + * raw: hosts read a batch diagnostic one line per edit, and a regular + * expression `.` stops at those separators. + */ +function quote(value: string): string { + return JSON.stringify(value).replace( + LINE_SEPARATORS, + (separator) => `\\u${separator.charCodeAt(0).toString(16)}`, + ); +} + function snippet(line: string): string { const trimmed = line.trim(); - return JSON.stringify( - trimmed.length > MAX_SNIPPET_CHARS ? `${trimmed.slice(0, MAX_SNIPPET_CHARS)}…` : trimmed, - ); + return quote(trimmed.length > MAX_SNIPPET_CHARS ? `${trimmed.slice(0, MAX_SNIPPET_CHARS)}…` : trimmed); } const ELISION_LINE = /^\s*(?:(?:\/\/|#|--|\/\*|\*|))?\s*$/; @@ -622,9 +633,13 @@ function describeMissing( lines: () => LineIndex, ): { reason: string; excerpt?: string } { const needle = neededLines(oldText).lines; - const nearest = nearestLine(needle.find((line) => line.trim().length > 0), lines); + const firstOffset = needle.findIndex((line) => line.trim().length > 0); + const nearest = nearestLine(needle[firstOffset], lines); const reason = describeMissingReason(text, oldText, needle, matching, lines, nearest); - const anchor = nearest != null && (!nearest.exact || nearest.count === 1) ? nearest.index : undefined; + // The anchor is the first non-blank line, so the region starts that many lines earlier. + const anchor = nearest != null && (!nearest.exact || nearest.count === 1) + ? Math.max(0, nearest.index - firstOffset) + : undefined; const start = closestRegion(needle, lines()) ?? anchor; const excerpt = start == null ? undefined : currentTextExcerpt(needle, lines(), start); return excerpt == null ? { reason } : { reason, excerpt }; @@ -656,10 +671,11 @@ function closestRegion(needle: readonly string[], lines: LineIndex): number | un for (let index = 0; index < lines.length && budget > 0; index++) { const hits = offsets.get(lines.text(index).trim()); if (!hits) continue; + // Offsets ascend, so every later one would start before the file too. for (const offset of hits) { - const start = index - offset; - if (start < 0) continue; if (budget-- <= 0) break; + const start = index - offset; + if (start < 0) break; const count = (votes.get(start) ?? 0) + 1; votes.set(start, count); if (count > bestVotes || (count === bestVotes && best !== undefined && start < best)) { @@ -705,7 +721,7 @@ function currentTextExcerpt(needle: readonly string[], lines: LineIndex, start: .map((mark, offset) => `${first + offset + 1}|${mark}${shortenLine(lines.text(first + offset))}`); for (; rows.length > 0; rows.pop()) { const where = rows.length === 1 ? `line ${first + 1}` : `lines ${first + 1}-${first + rows.length}`; - const excerpt = `the current text at ${where} (~ whitespace differs, ! text differs) is ${JSON.stringify(rows.join('\n'))}`; + const excerpt = `the current text at ${where} (~ whitespace differs, ! text differs) is ${quote(rows.join('\n'))}`; if (excerpt.length <= MAX_EXCERPT_CHARS) return excerpt; } return undefined; From b1a197e5581ae2d2f7f77f85d28b9916e3f9419b Mon Sep 17 00:00:00 2001 From: Danny Avila Date: Fri, 2 Oct 2026 22:01:18 -0400 Subject: [PATCH 3/4] =?UTF-8?q?=F0=9F=93=8F=20fix:=20Bound=20Edit=20Excerp?= =?UTF-8?q?ts=20by=20Their=20Encoded=20Error-Body=20Size?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Code API returns the diagnostic inside a JSON error body, and LibreChat reads that body through a 4,096-byte buffer, discarding a truncated one. A 3,000-character message with multibyte or escape-heavy excerpts could exceed it. Excerpts are now limited to 1,600 bytes and granted only while the whole message stays within 3,800 bytes, both measured as UTF-8 once JSON-encoded. --- packages/code/README.md | 5 +++- packages/code/src/edits.test.ts | 29 ++++++++++++++++++++++ packages/code/src/edits.ts | 43 ++++++++++++++++++++++++++------- 3 files changed, 67 insertions(+), 10 deletions(-) diff --git a/packages/code/README.md b/packages/code/README.md index ec38033b..378de9cf 100644 --- a/packages/code/README.md +++ b/packages/code/README.md @@ -648,7 +648,10 @@ text differs) is ""`: a JSON string of at most 8 rows of the form `|`, each line shortened to 160 characters, where the mark is a space for a line identical to `oldText` at that position, `~` when only its whitespace differs and `!` when its text differs. Excerpts are the first detail -dropped to fit the message bound, and never shorten another edit's reason. +dropped to fit the message bound, and never shorten another edit's reason. Both +the excerpt (1,600 bytes) and the message that carries it (3,800 bytes) are +measured as UTF-8 once JSON-encoded, so the Code API's error body stays within +the 4,096 bytes hosts such as LibreChat read. An ambiguous edit gives its match count and line numbers. Overlapping occurrences count as separate locations. Detailed source-line excerpts require `read_file` or `preview_edit` on the same workspace; edit-only workers return a diff --git a/packages/code/src/edits.test.ts b/packages/code/src/edits.test.ts index 48c059e8..1f03cbeb 100644 --- a/packages/code/src/edits.test.ts +++ b/packages/code/src/edits.test.ts @@ -906,3 +906,32 @@ test('quoted text escapes line separators so every batch edit stays on one line' assert.match(lines[2], /^Edit 2: /); assert.deepEqual(excerptOf(lines[1])?.rows, [`1|!const label = "first${separator}second";`]); }); + +function errorBodyBytes(message: string): number { + return Buffer.byteLength(JSON.stringify({ error: message, code: 'EDIT_CONFLICT' })); +} + +test('excerpts of multibyte source stay within the 4096-byte error body hosts read', () => { + const text = Array.from({ length: 12 }, (_, index) => `名前_${index} = "${'漢字'.repeat(80)}";`).join('\n'); + const oldText = Array.from({ length: 12 }, (_, index) => `名前_${index} = "changed";`).join('\n'); + const error = rejection(() => applyTextEdits(text, [{ oldText, newText: 'x' }])); + const excerpt = excerptOf(error.message); + assert.ok(excerpt != null && excerpt.rows.length >= 1 && excerpt.rows.length < 8); + assert.ok(Buffer.byteLength(JSON.stringify(EXCERPT.exec(error.message)![0])) - 2 <= 1_600); + assert.ok(errorBodyBytes(error.message) <= 4_096); +}); + +test('batch excerpts are granted by encoded size, so escapes cannot overflow the error body', () => { + const lines = Array.from({ length: 200 }, (_, index) => `path_${index} = "C:\\\\dir\\\\${'"q"'.repeat(48)}";`); + const edits = Array.from({ length: 6 }, (_, index) => ({ + oldText: lines.slice(index * 20, index * 20 + 8).map((line, offset) => (offset === 3 ? 'changed();' : line)).join('\n'), + newText: 'x', + })); + const error = rejection(() => applyTextEdits(`${lines.join('\n')}\n`, edits)); + assert.ok(errorBodyBytes(error.message) <= 4_096, String(errorBodyBytes(error.message))); + assert.match(error.message, /\nEdit 1: [^\n]*the current text at/); + assert.deepEqual( + [...error.message.matchAll(/\nEdit (\d+):/g)].map((match) => Number(match[1])), + [1, 2, 3, 4, 5, 6], + ); +}); diff --git a/packages/code/src/edits.ts b/packages/code/src/edits.ts index d3ce9799..6b4c8b7a 100644 --- a/packages/code/src/edits.ts +++ b/packages/code/src/edits.ts @@ -12,12 +12,19 @@ import type { * callers prefix their own context, so diagnostics stay well below that. */ export const EDIT_DIAGNOSTIC_MAX_CHARS = 3000; +/** + * Code API returns the message inside a JSON error body, and hosts read that + * body through a bounded buffer (4096 bytes in LibreChat). Optional excerpts + * are granted only while the JSON-encoded UTF-8 message stays within this size. + */ +export const EDIT_DIAGNOSTIC_MAX_BODY_BYTES = 3_800; const MAX_REPORTED_LINES = 5; const MAX_SNIPPET_CHARS = 120; /** A missing edit's current-text excerpt: a few lines, each shortened, in one bounded hint. */ const MAX_EXCERPT_LINES = 8; const MAX_EXCERPT_LINE_CHARS = 160; -const MAX_EXCERPT_CHARS = 1_600; +/** Measured like the message bound: UTF-8 bytes once JSON-encoded into the error body. */ +const MAX_EXCERPT_BYTES = 1_600; /** Alignment votes cast while locating the region a missing edit most likely meant. */ const MAX_REGION_VOTES = 200_000; /** Lines too generic to anchor a region on their own (`}`, `);`, blank lines). */ @@ -722,7 +729,7 @@ function currentTextExcerpt(needle: readonly string[], lines: LineIndex, start: for (; rows.length > 0; rows.pop()) { const where = rows.length === 1 ? `line ${first + 1}` : `lines ${first + 1}-${first + rows.length}`; const excerpt = `the current text at ${where} (~ whitespace differs, ! text differs) is ${quote(rows.join('\n'))}`; - if (excerpt.length <= MAX_EXCERPT_CHARS) return excerpt; + if (encodedBytes(excerpt) <= MAX_EXCERPT_BYTES) return excerpt; } return undefined; } @@ -816,6 +823,11 @@ function nearestLine( : undefined; } +/** UTF-8 bytes `value` occupies once JSON-encoded as a string in an error body. */ +function encodedBytes(value: string): number { + return Buffer.byteLength(JSON.stringify(value)) - 2; +} + /** A failure's reason, followed by its excerpt when the excerpt was granted room. */ function failureText(failure: EditFailure, withExcerpt: boolean): string { return withExcerpt && failure.excerpt ? `${failure.reason}; ${failure.excerpt}` : failure.reason; @@ -825,22 +837,34 @@ function failureText(failure: EditFailure, withExcerpt: boolean): string { * Grants excerpts in edit order while the message still fits. Excerpts are the * most expendable detail, so granting one never shortens another reason. */ -function grantExcerpts(failures: readonly EditFailure[], available: number): boolean[] { - let remaining = available - failures.reduce((total, failure) => total + failure.reason.length, 0); +function grantExcerpts( + failures: readonly EditFailure[], + availableChars: number, + availableBytes: number, +): boolean[] { + let chars = availableChars - failures.reduce((total, failure) => total + failure.reason.length, 0); + let bytes = availableBytes - failures.reduce((total, failure) => total + encodedBytes(failure.reason), 0); return failures.map((failure) => { - const cost = failure.excerpt ? failure.excerpt.length + 2 : 0; - if (cost === 0 || cost > remaining) return false; - remaining -= cost; + if (!failure.excerpt) return false; + const addition = `; ${failure.excerpt}`; + const byteCost = encodedBytes(addition); + if (addition.length > chars || byteCost > bytes) return false; + chars -= addition.length; + bytes -= byteCost; return true; }); } +function fitsDiagnosticBounds(message: string): boolean { + return message.length <= EDIT_DIAGNOSTIC_MAX_CHARS && encodedBytes(message) <= EDIT_DIAGNOSTIC_MAX_BODY_BYTES; +} + function formatEditFailures(failures: readonly EditFailure[], editCount: number): string { if (editCount === 1) { const single = (withExcerpt: boolean) => `Workspace edit did not apply and nothing was written: ${failures[0] ? failureText(failures[0], withExcerpt) : 'no match'}.`; const detailed = single(true); - return (detailed.length <= EDIT_DIAGNOSTIC_MAX_CHARS ? detailed : single(false)).slice( + return (fitsDiagnosticBounds(detailed) ? detailed : single(false)).slice( 0, EDIT_DIAGNOSTIC_MAX_CHARS, ); @@ -854,7 +878,8 @@ function formatEditFailures(failures: readonly EditFailure[], editCount: number) // reasons or source excerpts. Never leave callers guessing which edits failed. const available = EDIT_DIAGNOSTIC_MAX_CHARS - header.length - footer.length - prefixes.reduce((total, prefix) => total + prefix.length + 1, 0); - const granted = grantExcerpts(failures, available); + const frame = header + footer + prefixes.join('') + '.'.repeat(failures.length); + const granted = grantExcerpts(failures, available, EDIT_DIAGNOSTIC_MAX_BODY_BYTES - encodedBytes(frame)); const reasons = failures.map((failure, index) => failureText(failure, granted[index])); const reasonLength = reasons.reduce((total, reason) => total + reason.length, 0); const reasonLimit = reasonLength <= available ? Infinity : Math.floor(available / failures.length); From c807ce4e4c1a34c203e9d46d60541daaf8523b87 Mon Sep 17 00:00:00 2001 From: Danny Avila Date: Fri, 2 Oct 2026 22:11:26 -0400 Subject: [PATCH 4/4] =?UTF-8?q?=F0=9F=93=90=20fix:=20Shorten=20Edit=20Diag?= =?UTF-8?q?nostics=20by=20Encoded=20Size=20as=20Well=20as=20Length?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Batch reasons that still exceed the bound after excerpts are declined are now shortened against both the 3,000-character limit and the 3,800-byte encoded budget, and a single edit's reason is shortened the same way instead of being cut by character count. Multibyte closest-line quotes can no longer push the Code API's error body past the 4,096 bytes hosts read. --- packages/code/src/edits.test.ts | 27 +++++++++++++++++ packages/code/src/edits.ts | 53 ++++++++++++++++++++++----------- 2 files changed, 62 insertions(+), 18 deletions(-) diff --git a/packages/code/src/edits.test.ts b/packages/code/src/edits.test.ts index 1f03cbeb..5bcabf49 100644 --- a/packages/code/src/edits.test.ts +++ b/packages/code/src/edits.test.ts @@ -935,3 +935,30 @@ test('batch excerpts are granted by encoded size, so escapes cannot overflow the [1, 2, 3, 4, 5, 6], ); }); + +test('shortened batch reasons also respect the encoded error-body size', () => { + const emojis = '😀'.repeat(200); + const text = Array.from({ length: 20 }, (_, index) => `item_${index} emoji_${index} = "${emojis}";`).join('\n'); + const edits = Array.from({ length: 20 }, (_, index) => ({ + oldText: `item_${index} emoji_${index} = "plain";`, + newText: 'x', + })); + const error = rejection(() => applyTextEdits(text, edits)); + assert.ok(error.message.length <= EDIT_DIAGNOSTIC_MAX_CHARS); + assert.ok(errorBodyBytes(error.message) <= 4_096, String(errorBodyBytes(error.message))); + assert.equal(Buffer.from(error.message).toString('utf8'), error.message); + assert.deepEqual( + [...error.message.matchAll(/\nEdit (\d+):/g)].map((match) => Number(match[1])), + Array.from({ length: 20 }, (_, index) => index + 1), + ); +}); + +test('a single edit with a long multibyte reason stays within both bounds', () => { + const error = new WorkspaceEditMatchError([ + { index: 0, reason: `old_text was not found; the closest line is line 1: "${'😀'.repeat(1_400)}"` }, + ], 1); + assert.ok(error.message.length <= EDIT_DIAGNOSTIC_MAX_CHARS); + assert.ok(errorBodyBytes(error.message) <= 4_096); + assert.equal(Buffer.from(error.message).toString('utf8'), error.message); + assert.match(error.message, /…\.$/); +}); diff --git a/packages/code/src/edits.ts b/packages/code/src/edits.ts index 6b4c8b7a..8205b1d7 100644 --- a/packages/code/src/edits.ts +++ b/packages/code/src/edits.ts @@ -855,19 +855,40 @@ function grantExcerpts( }); } +/** + * The longest prefix of `reason` within both limits, ending in an ellipsis. It + * never splits a surrogate pair in a shortened source excerpt. + */ +function shortenReason(reason: string, charLimit: number, byteLimit: number): string { + if (reason.length <= charLimit && encodedBytes(reason) <= byteLimit) return reason; + const budget = byteLimit - encodedBytes('…'); + let end = 0; + let bytes = 0; + while (end < reason.length) { + const width = /[\uD800-\uDBFF]/.test(reason[end]) && /[\uDC00-\uDFFF]/.test(reason[end + 1] ?? '') ? 2 : 1; + const cost = encodedBytes(reason.slice(end, end + width)); + if (end + width > charLimit - 1 || bytes + cost > budget) break; + bytes += cost; + end += width; + } + return `${reason.slice(0, end)}…`; +} + function fitsDiagnosticBounds(message: string): boolean { return message.length <= EDIT_DIAGNOSTIC_MAX_CHARS && encodedBytes(message) <= EDIT_DIAGNOSTIC_MAX_BODY_BYTES; } function formatEditFailures(failures: readonly EditFailure[], editCount: number): string { if (editCount === 1) { - const single = (withExcerpt: boolean) => - `Workspace edit did not apply and nothing was written: ${failures[0] ? failureText(failures[0], withExcerpt) : 'no match'}.`; - const detailed = single(true); - return (fitsDiagnosticBounds(detailed) ? detailed : single(false)).slice( - 0, - EDIT_DIAGNOSTIC_MAX_CHARS, + const prefix = 'Workspace edit did not apply and nothing was written: '; + const detailed = `${prefix}${failures[0] ? failureText(failures[0], true) : 'no match'}.`; + if (fitsDiagnosticBounds(detailed)) return detailed; + const reason = shortenReason( + failures[0]?.reason ?? 'no match', + EDIT_DIAGNOSTIC_MAX_CHARS - prefix.length - 1, + EDIT_DIAGNOSTIC_MAX_BODY_BYTES - encodedBytes(prefix) - 1, ); + return `${prefix}${reason}.`; } const header = `${failures.length} of ${editCount} workspace edits did not apply, so nothing was written. Every other edit matched.`; const footer = failures.some((failure) => failure.index > 0) @@ -879,19 +900,15 @@ function formatEditFailures(failures: readonly EditFailure[], editCount: number) const available = EDIT_DIAGNOSTIC_MAX_CHARS - header.length - footer.length - prefixes.reduce((total, prefix) => total + prefix.length + 1, 0); const frame = header + footer + prefixes.join('') + '.'.repeat(failures.length); - const granted = grantExcerpts(failures, available, EDIT_DIAGNOSTIC_MAX_BODY_BYTES - encodedBytes(frame)); + const availableBytes = EDIT_DIAGNOSTIC_MAX_BODY_BYTES - encodedBytes(frame); + const granted = grantExcerpts(failures, available, availableBytes); const reasons = failures.map((failure, index) => failureText(failure, granted[index])); const reasonLength = reasons.reduce((total, reason) => total + reason.length, 0); - const reasonLimit = reasonLength <= available ? Infinity : Math.floor(available / failures.length); - const details = reasons.map((full, index) => { - let reason = full; - if (reason.length > reasonLimit) { - let end = reasonLimit - 1; - // Do not split a surrogate pair in a shortened source excerpt. - if (end > 0 && /[\uD800-\uDBFF]/.test(reason[end - 1]) && /[\uDC00-\uDFFF]/.test(reason[end])) end--; - reason = `${reason.slice(0, end)}…`; - } - return `${prefixes[index]}${reason}.`; - }); + const reasonBytes = reasons.reduce((total, reason) => total + encodedBytes(reason), 0); + const charLimit = reasonLength <= available ? Infinity : Math.floor(available / failures.length); + const byteLimit = reasonBytes <= availableBytes ? Infinity : Math.floor(availableBytes / failures.length); + const details = reasons.map((reason, index) => + `${prefixes[index]}${shortenReason(reason, charLimit, byteLimit)}.`, + ); return header + details.join('') + footer; }