Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
13 changes: 12 additions & 1 deletion packages/code/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -640,7 +640,18 @@ fails, the worker still checks the rest and rejects the whole batch with one
batch messages shorten reasons and source excerpts to stay within the bound,
but never omit failing edit positions. A missing edit names the nearest
candidate line and flags elided (`...`) or line-numbered `oldText`, a
whitespace-only difference, or CRLF line endings.
whitespace-only difference, or CRLF line endings. When the file has a region
that `oldText` most likely meant (the alignment where the most of its
distinctive lines fall into place, or its unique first line or closest line),
the reason ends with `the current text at lines A-B (~ whitespace differs, !
text differs) is "<rows>"`: a JSON string of at most 8 rows of the form
`<line>|<mark><text>`, each line shortened to 160 characters, where the mark is
a space for a line identical to `oldText` at that position, `~` when only its
whitespace differs and `!` when its text differs. Excerpts are the first detail
dropped to fit the message bound, and never shorten another edit's reason. Both
the excerpt (1,600 bytes) and the message that carries it (3,800 bytes) are
measured as UTF-8 once JSON-encoded, so the Code API's error body stays within
the 4,096 bytes hosts such as LibreChat read.
An ambiguous edit gives its match count and line numbers. Overlapping
occurrences count as separate locations. Detailed source-line excerpts require
`read_file` or `preview_edit` on the same workspace; edit-only workers return a
Expand Down
243 changes: 243 additions & 0 deletions packages/code/src/edits.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -719,3 +719,246 @@ test('short batch diagnostic reasons remain complete when the message fits', ()
assert.ok(error.message.includes(`\nEdit ${failure.index + 1}: ${failure.reason}.`));
}
});

const EXCERPT = /the current text at lines? (\d+)(?:-(\d+))? \(~ whitespace differs, ! text differs\) is ("(?:[^"\\]|\\.)*")/;

function excerptOf(message: string): { first: number; last: number; rows: string[] } | undefined {
const match = EXCERPT.exec(message);
if (!match) return undefined;
const first = Number(match[1]);
return { first, last: Number(match[2] ?? first), rows: (JSON.parse(match[3]) as string).split('\n') };
}

test('a missing edit quotes the current text of the region it meant, marking what differs', () => {
const text = 'export function load(user) {\n const id = user.id;\n return fetchUser(id);\n}\n';
const error = rejection(() => applyTextEdits(text, [{
oldText: 'export function load(user) {\n const id = user.id;\n return fetchUser(user);\n}',
newText: 'x',
}], 'tolerant'));
assert.match(error.message, /old_text was not found; its first line appears at line 1, but the lines after it differ; the current text/);
assert.deepEqual(excerptOf(error.message), {
first: 1,
last: 4,
rows: [
'1| export function load(user) {',
'2|~ const id = user.id;',
'3|! return fetchUser(id);',
'4| }',
],
});
assert.match(error.message, /\.$/);
});

test('the region is found from later lines when the first line itself changed', () => {
const text = 'header();\nfunction save(record) {\n validate(record);\n persist(record);\n}\nfooter();\n';
const error = rejection(() => applyTextEdits(text, [{
oldText: 'function save(item) {\n validate(record);\n persist(record);\n}',
newText: 'x',
}]));
assert.deepEqual(excerptOf(error.message), {
first: 2,
last: 5,
rows: ['2|!function save(record) {', '3| validate(record);', '4| persist(record);', '5| }'],
});
});

test('the region with the most aligned lines wins, and ties prefer the earliest', () => {
const block = (name: string) => `function ${name}() {\n prepare();\n run();\n cleanup();\n}\n`;
const text = `${block('first')}${block('second')}`;
const error = rejection(() => applyTextEdits(text, [{
oldText: 'function second() {\n prepare();\n execute();\n cleanup();\n}',
newText: 'x',
}]));
assert.equal(excerptOf(error.message)?.first, 6);
const tied = rejection(() => applyTextEdits(text, [{
oldText: 'function third() {\n prepare();\n execute();\n cleanup();\n}',
newText: 'x',
}]));
assert.equal(excerptOf(tied.message)?.first, 1);
});

test('a missing one-line edit quotes its closest line', () => {
const error = rejection(() => applyTextEdits('const total = items.length;\nreturn total;\n', [
{ oldText: 'const total = items.size;', newText: 'x' },
]));
assert.match(error.message, /the closest line is line 1: "const total = items.length;"/);
assert.deepEqual(excerptOf(error.message), { first: 1, last: 1, rows: ['1|!const total = items.length;'] });
});

test('exact mode quotes the whitespace to copy alongside the whitespace hint', () => {
const error = rejection(() => applyTextEdits('class A {\n a();\n b();\n}\n', [
{ oldText: 'a();\nb();', newText: 'x' },
]));
assert.match(error.message, /exists at line 2 with different whitespace \(indentation-flexible\); copy that whitespace exactly; the current text at lines 2-3/);
assert.deepEqual(excerptOf(error.message)?.rows, ['2|~ a();', '3|~ b();']);
});

test('a long region is quoted from just before its first difference', () => {
const source = Array.from({ length: 40 }, (_, index) => `step(${index});`);
const needle = source.slice(5, 25);
needle[8] = 'step(changed);';
const error = rejection(() => applyTextEdits(`${source.join('\n')}\n`, [{ oldText: needle.join('\n'), newText: 'x' }]));
const excerpt = excerptOf(error.message);
assert.equal(excerpt?.first, 13);
assert.equal(excerpt?.last, 20);
assert.equal(excerpt?.rows[1], '14|!step(13);');
assert.ok(excerpt?.rows.every((row, index) => index === 1 || row.includes('| ')));
});

test('excerpts shorten long lines and stay bounded without splitting characters', () => {
const wide = `x${'πŸ˜€'.repeat(400)}`;
const controls = '\u0001'.repeat(2_000);
const text = Array.from({ length: 10 }, (_, index) => `key_${index} = ${index % 2 === 0 ? wide : controls}`).join('\n');
const oldText = Array.from({ length: 10 }, (_, index) => `key_${index} = other`).join('\n');
const error = rejection(() => applyTextEdits(text, [{ oldText, newText: 'x' }]));
const excerpt = excerptOf(error.message);
assert.ok(excerpt != null);
assert.ok(excerpt.rows.length >= 1 && excerpt.rows.length <= 8);
assert.ok(EXCERPT.exec(error.message)![0].length <= 1_600);
for (const row of excerpt.rows) {
assert.ok(row.length <= 175, row.slice(0, 40));
assert.equal(Buffer.from(row).toString('utf8'), row);
}
});

test('an edit that resembles nothing in the file gets no excerpt', () => {
const error = rejection(() => applyTextEdits('alpha\nbeta\n', [
{ oldText: 'completely different text\nwith nothing shared', newText: 'x' },
]));
assert.equal(excerptOf(error.message), undefined);
assert.match(error.message, /old_text was not found\.$/);
});

test('ambiguous and line-numbered edits keep their messages without an excerpt', () => {
const ambiguous = rejection(() => applyTextEdits('return a;\nreturn a;\n', [{ oldText: 'return a;', newText: 'x' }]));
assert.equal(excerptOf(ambiguous.message), undefined);
const numbered = rejection(() => applyTextEdits('start\nmiddle\nend\n', [{ oldText: '1 | start\n2 | middle', newText: 'x' }]));
assert.equal(excerptOf(numbered.message), undefined);
});

test('batch excerpts are granted in edit order only while every position and reason still fits', () => {
const lines = Array.from({ length: 400 }, (_, index) => `value_${index} = compute_${index}(input_${index}, ${'p'.repeat(120)});`);
const edits = Array.from({ length: 12 }, (_, index) => ({
oldText: `${lines[index * 10]}\n${lines[index * 10 + 1].replace('compute', 'changed')}\n${lines[index * 10 + 2]}`,
newText: 'x',
}));
const error = rejection(() => applyTextEdits(`${lines.join('\n')}\n`, edits));
assert.ok(error.message.length <= EDIT_DIAGNOSTIC_MAX_CHARS);
assert.deepEqual(
[...error.message.matchAll(/\nEdit (\d+):/g)].map((match) => Number(match[1])),
Array.from({ length: 12 }, (_, index) => index + 1),
);
const quoted = [...error.message.matchAll(/\nEdit (\d+): [^\n]*the current text at/g)].map((match) => Number(match[1]));
assert.ok(quoted.length >= 1 && quoted.length < 12);
assert.deepEqual(quoted, Array.from({ length: quoted.length }, (_, index) => index + 1));
for (const failure of error.failures) {
assert.ok(error.message.includes(`\nEdit ${failure.index + 1}: ${failure.reason}`));
}
});

test('a single edit drops its excerpt rather than exceed the bound', () => {
const error = new WorkspaceEditMatchError([
{ index: 0, reason: `old_text was not found${'; x'.repeat(1_200)}`, excerpt: 'the current text at line 1 (~ whitespace differs, ! text differs) is "1|!y"' },
], 1);
assert.ok(error.message.length <= EDIT_DIAGNOSTIC_MAX_CHARS);
assert.doesNotMatch(error.message, /the current text/);
});

test('region votes stay bounded on highly repetitive files', () => {
const text = 'item = value;\nother = thing;\n'.repeat(150_000);
const oldText = 'item = value;\nother = thing;\nmissing = line;\nitem = value;';
const started = performance.now();
const error = rejection(() => applyTextEdits(text, [{ oldText, newText: 'x' }]));
assert.ok(performance.now() - started < 3_000, 'a repetitive file must not cast unbounded votes');
assert.match(error.message, /did not apply and nothing was written/);
});

test('an excerpt anchored on a first line after blank lines starts where old_text starts', () => {
const error = rejection(() => applyTextEdits('header();\nconst total = items.length;\nreturn total;\n', [
{ oldText: '\nconst total = items.size;', newText: 'x' },
]));
assert.deepEqual(excerptOf(error.message), {
first: 1,
last: 2,
rows: ['1|!header();', '2|!const total = items.length;'],
});
});

test('region votes count every inspected anchor offset against the budget', () => {
const oldText = `${'repeat_line();\n'.repeat(5_000)}missing();`;
const text = 'repeat_line();\n'.repeat(20_000);
const started = performance.now();
const error = rejection(() => applyTextEdits(text, [{ oldText, newText: 'x' }]));
assert.ok(performance.now() - started < 3_000, 'repeated anchor offsets must not escape the vote budget');
assert.match(error.message, /did not apply and nothing was written/);
});

test('quoted text escapes line separators so every batch edit stays on one line', () => {
const separator = String.fromCharCode(0x2028);
const text = `const label = "first${separator}second";\nconst other = "value";\n`;
const error = rejection(() => applyTextEdits(text, [
{ oldText: `const label = "first${separator}third";`, newText: 'x' },
{ oldText: 'const other = "changed";', newText: 'y' },
]));
assert.ok(!error.message.includes(separator));
const lines = error.message.split('\n');
assert.match(lines[1], /^Edit 1: old_text was not found; the closest line is line 1: .*\\u2028.*the current text at line 1/);
assert.match(lines[2], /^Edit 2: /);
assert.deepEqual(excerptOf(lines[1])?.rows, [`1|!const label = "first${separator}second";`]);
});

function errorBodyBytes(message: string): number {
return Buffer.byteLength(JSON.stringify({ error: message, code: 'EDIT_CONFLICT' }));
}

test('excerpts of multibyte source stay within the 4096-byte error body hosts read', () => {
const text = Array.from({ length: 12 }, (_, index) => `名前_${index} = "${'ζΌ’ε­—'.repeat(80)}";`).join('\n');
const oldText = Array.from({ length: 12 }, (_, index) => `名前_${index} = "changed";`).join('\n');
const error = rejection(() => applyTextEdits(text, [{ oldText, newText: 'x' }]));
const excerpt = excerptOf(error.message);
assert.ok(excerpt != null && excerpt.rows.length >= 1 && excerpt.rows.length < 8);
assert.ok(Buffer.byteLength(JSON.stringify(EXCERPT.exec(error.message)![0])) - 2 <= 1_600);
assert.ok(errorBodyBytes(error.message) <= 4_096);
});

test('batch excerpts are granted by encoded size, so escapes cannot overflow the error body', () => {
const lines = Array.from({ length: 200 }, (_, index) => `path_${index} = "C:\\\\dir\\\\${'"q"'.repeat(48)}";`);
const edits = Array.from({ length: 6 }, (_, index) => ({
oldText: lines.slice(index * 20, index * 20 + 8).map((line, offset) => (offset === 3 ? 'changed();' : line)).join('\n'),
newText: 'x',
}));
const error = rejection(() => applyTextEdits(`${lines.join('\n')}\n`, edits));
assert.ok(errorBodyBytes(error.message) <= 4_096, String(errorBodyBytes(error.message)));
assert.match(error.message, /\nEdit 1: [^\n]*the current text at/);
assert.deepEqual(
[...error.message.matchAll(/\nEdit (\d+):/g)].map((match) => Number(match[1])),
[1, 2, 3, 4, 5, 6],
);
});

test('shortened batch reasons also respect the encoded error-body size', () => {
const emojis = 'πŸ˜€'.repeat(200);
const text = Array.from({ length: 20 }, (_, index) => `item_${index} emoji_${index} = "${emojis}";`).join('\n');
const edits = Array.from({ length: 20 }, (_, index) => ({
oldText: `item_${index} emoji_${index} = "plain";`,
newText: 'x',
}));
const error = rejection(() => applyTextEdits(text, edits));
assert.ok(error.message.length <= EDIT_DIAGNOSTIC_MAX_CHARS);
assert.ok(errorBodyBytes(error.message) <= 4_096, String(errorBodyBytes(error.message)));
assert.equal(Buffer.from(error.message).toString('utf8'), error.message);
assert.deepEqual(
[...error.message.matchAll(/\nEdit (\d+):/g)].map((match) => Number(match[1])),
Array.from({ length: 20 }, (_, index) => index + 1),
);
});

test('a single edit with a long multibyte reason stays within both bounds', () => {
const error = new WorkspaceEditMatchError([
{ index: 0, reason: `old_text was not found; the closest line is line 1: "${'πŸ˜€'.repeat(1_400)}"` },
], 1);
assert.ok(error.message.length <= EDIT_DIAGNOSTIC_MAX_CHARS);
assert.ok(errorBodyBytes(error.message) <= 4_096);
assert.equal(Buffer.from(error.message).toString('utf8'), error.message);
assert.match(error.message, /…\.$/);
});
Loading
Loading