이전에는 순수 부분 문자열 매칭이라, 파일보다 얕게 들여쓴 SEARCH가 더 깊은 줄
안에서 매칭됐음(" deep()"가 " deep()"에 적중). 한 줄을 노린 편집이
다른 줄에 적용될 수 있는 경로였고, 조용히 잘못된 바이트를 쓰는 fail-open이었음.
"무조건 줄 단위 일치"로 바꾸면 줄 안의 조각을 고치는 정당한 사용
("return 1" → "return 42")이 전부 깨지므로, SEARCH가 스스로 무엇을 주장하는지에
따라 앵커를 조건부로 적용:
- 선행 공백이 있거나 여러 줄 → 줄 시작 정렬 필수
(들여쓰기/구조를 주장하고 있으므로 정렬돼야 함)
- 선행 공백 없는 한 줄 → 기존대로 부분 매칭
(줄 안의 조각일 뿐 들여쓰기를 주장한 적 없음)
거부는 fail-closed — 찾지 못함으로 보고돼 모델이 더 정확한 컨텍스트로 재시도함.
앵커에 걸린 등장은 건너뛰고 뒤쪽의 정렬된 등장을 계속 찾으므로, 같은 텍스트가
줄 중간과 줄 시작에 모두 있으면 후자를 고침.
부수 수정 — 오프셋 치환:
빠른 경로와 폴백 경로 모두 String.replace()를 쓰고 있었는데, replace()는 위치 0
부터 다시 스캔하므로 방금 내린 앵커 판정을 무효화하고 앞쪽의 정렬 안 된 등장을
고칠 수 있었음. 이미 알고 있는 오프셋으로 splice하도록 양쪽 다 변경.
테스트 121개 통과(앵커 규칙 4개 신규). 실디스크 검증: 8칸 들여쓰기 안의
value = 1 → 99 수정 정상, 들여쓰기 불일치 케이스는 파일 무변경으로 거부.
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
170 lines
7.9 KiB
TypeScript
170 lines
7.9 KiB
TypeScript
/**
|
|
* search-replace-edit.test.ts
|
|
*
|
|
* This function rewrites the user's real source files. It has no way to "fail loudly" — a bug
|
|
* writes wrong bytes to disk, or reports "3 edits applied" having applied two, and the user
|
|
* finds out later from broken code. It was inline inside executeTool()'s 1,349-line body until
|
|
* 2026-07-29, so none of this had ever been exercised without touching the filesystem.
|
|
*
|
|
* The properties worth guarding, in rough order of how much damage a regression does:
|
|
* 1. never claim an edit that did not happen (silent corruption of the edit report)
|
|
* 2. never replace the wrong span (silent corruption of the file)
|
|
* 3. preserve bytes outside the matched span (indentation, trailing whitespace)
|
|
* 4. report misses so the model can retry
|
|
*/
|
|
|
|
import { test, describe } from 'node:test';
|
|
import assert from 'node:assert/strict';
|
|
import { applySearchReplaceEdits } from '../src/gateway/chat/search-replace-edit';
|
|
|
|
const block = (search: string, replace: string) =>
|
|
`------- SEARCH\n${search}\n=======\n${replace}\n+++++++ REPLACE`;
|
|
|
|
describe('정확히 일치하는 경우', () => {
|
|
test('한 블록을 적용한다', () => {
|
|
const r = applySearchReplaceEdits('a\nb\nc\n', block('b', 'B'));
|
|
assert.equal(r.content, 'a\nB\nc\n');
|
|
assert.equal(r.editCount, 1);
|
|
assert.deepEqual(r.failedEdits, []);
|
|
});
|
|
|
|
test('여러 블록을 순서대로 적용한다', () => {
|
|
const r = applySearchReplaceEdits('one\ntwo\nthree\n', block('one', '1') + '\n' + block('three', '3'));
|
|
assert.equal(r.content, '1\ntwo\n3\n');
|
|
assert.equal(r.editCount, 2);
|
|
});
|
|
|
|
test('여러 줄 블록', () => {
|
|
const src = 'def f():\n return 1\n\nprint(f())\n';
|
|
const r = applySearchReplaceEdits(src, block('def f():\n return 1', 'def f():\n return 42'));
|
|
assert.equal(r.content, 'def f():\n return 42\n\nprint(f())\n');
|
|
assert.equal(r.editCount, 1);
|
|
});
|
|
|
|
test('첫 번째 일치만 바꾼다 — 전역 치환이 아니다', () => {
|
|
const r = applySearchReplaceEdits('x\nx\nx\n', block('x', 'y'));
|
|
assert.equal(r.content, 'y\nx\nx\n');
|
|
assert.equal(r.editCount, 1);
|
|
});
|
|
});
|
|
|
|
describe('후행 공백 폴백 — 모델이 흔히 틀리는 지점', () => {
|
|
test('파일에 후행 공백이 있어도 매칭된다', () => {
|
|
const src = 'const a = 1; \nconst b = 2;\n';
|
|
const r = applySearchReplaceEdits(src, block('const a = 1;', 'const a = 99;'));
|
|
assert.equal(r.editCount, 1);
|
|
assert.ok(r.content.startsWith('const a = 99;'), r.content);
|
|
});
|
|
|
|
test('SEARCH 쪽에 후행 공백이 있어도 매칭된다', () => {
|
|
const r = applySearchReplaceEdits('const a = 1;\n', block('const a = 1; ', 'const a = 99;'));
|
|
assert.equal(r.editCount, 1);
|
|
assert.equal(r.content, 'const a = 99;\n');
|
|
});
|
|
|
|
test('매칭 구간 밖의 바이트는 보존된다', () => {
|
|
// SEARCH가 'target'뿐이면 파일에 있던 후행 공백은 매칭 구간 밖이므로 그대로 남는다.
|
|
// 앞뒤 줄의 공백도 손대지 않는다.
|
|
const src = 'head \ntarget \ntail \n';
|
|
const r = applySearchReplaceEdits(src, block('target', 'TARGET'));
|
|
assert.equal(r.editCount, 1);
|
|
assert.equal(r.content, 'head \nTARGET \ntail \n');
|
|
});
|
|
|
|
test('들여쓰기가 다르면 매칭을 거부한다 — fail closed', () => {
|
|
// 이전에는 4칸 SEARCH가 8칸 줄 안에서 부분 매칭돼 엉뚱한 줄을 고칠 수 있었다.
|
|
// 이제는 선행 공백이 있으면 줄 시작 정렬을 요구하므로 거부되고, 모델이 더 정확한
|
|
// 컨텍스트로 재시도하게 된다. 조용히 잘못 고치는 것보다 실패가 낫다.
|
|
const src = 'if x:\n deep()\n';
|
|
const r = applySearchReplaceEdits(src, block(' deep()', ' shallow()'));
|
|
assert.equal(r.editCount, 0);
|
|
assert.equal(r.content, src, '거부됐는데 파일이 변경됨');
|
|
assert.deepEqual(r.failedEdits, [' deep()']);
|
|
});
|
|
});
|
|
|
|
describe('줄 앵커 규칙 — SEARCH가 무엇을 주장하는지에 따라 달라진다', () => {
|
|
test('선행 공백 없는 한 줄은 부분 매칭 허용 — 들여쓰기를 주장하지 않았다', () => {
|
|
const r = applySearchReplaceEdits('def f():\n return 1\n', block('return 1', 'return 42'));
|
|
assert.equal(r.editCount, 1);
|
|
assert.equal(r.content, 'def f():\n return 42\n', '들여쓰기가 보존돼야 한다');
|
|
});
|
|
|
|
test('선행 공백이 있으면 줄 시작에 정렬돼야 한다', () => {
|
|
const r = applySearchReplaceEdits('x = 1\n y = 2\n', block(' y = 2', ' y = 3'));
|
|
assert.equal(r.editCount, 1, '정확히 정렬된 경우는 통과해야 한다');
|
|
assert.equal(r.content, 'x = 1\n y = 3\n');
|
|
});
|
|
|
|
test('여러 줄 SEARCH는 선행 공백이 없어도 줄 시작을 요구한다', () => {
|
|
// 줄 중간에서 시작하는 다중 줄 매칭은 구조를 깨뜨린다.
|
|
const src = 'prefix a = 1\nb = 2\n';
|
|
const r = applySearchReplaceEdits(src, block('a = 1\nb = 2', 'REPLACED'));
|
|
assert.equal(r.editCount, 0);
|
|
assert.equal(r.content, src);
|
|
});
|
|
|
|
test('앵커 실패 후 뒤쪽의 올바른 위치를 계속 찾는다', () => {
|
|
// 첫 등장이 줄 중간이어도 포기하지 않고, 제대로 정렬된 다음 등장을 찾아야 한다.
|
|
const src = 'wrapper target()\n target()\n';
|
|
const r = applySearchReplaceEdits(src, block(' target()', ' fixed()'));
|
|
assert.equal(r.editCount, 1);
|
|
assert.equal(r.content, 'wrapper target()\n fixed()\n', '줄 시작에 정렬된 두 번째 등장을 고쳐야 한다');
|
|
});
|
|
});
|
|
|
|
describe('실패 보고 — 적용하지 않은 편집을 성공으로 세지 않는다', () => {
|
|
test('찾지 못한 블록은 failedEdits에 남고 editCount에 포함되지 않는다', () => {
|
|
const r = applySearchReplaceEdits('a\n', block('없는텍스트', 'x'));
|
|
assert.equal(r.editCount, 0);
|
|
assert.deepEqual(r.failedEdits, ['없는텍스트']);
|
|
assert.equal(r.content, 'a\n', '실패했는데 파일이 바뀜');
|
|
});
|
|
|
|
test('성공과 실패가 섞이면 각각 정확히 집계된다', () => {
|
|
const r = applySearchReplaceEdits('a\nb\n', block('a', 'A') + '\n' + block('없음', 'x'));
|
|
assert.equal(r.editCount, 1);
|
|
assert.equal(r.failedEdits.length, 1);
|
|
assert.equal(r.content, 'A\nb\n');
|
|
});
|
|
|
|
test('실패 라벨은 첫 줄 80자로 자른다', () => {
|
|
const long = 'z'.repeat(200);
|
|
const r = applySearchReplaceEdits('a\n', block(long, 'x'));
|
|
assert.equal(r.failedEdits[0].length, 80);
|
|
});
|
|
});
|
|
|
|
describe('형식 오류 구분', () => {
|
|
test('SEARCH 마커가 아예 없으면 hadAnyBlock=false — 형식 안내를 보내야 하는 경우', () => {
|
|
const r = applySearchReplaceEdits('a\n', 'just some text, no markers');
|
|
assert.equal(r.hadAnyBlock, false);
|
|
assert.equal(r.editCount, 0);
|
|
});
|
|
|
|
test('마커는 있는데 매칭이 안 된 경우는 형식 오류가 아니다', () => {
|
|
// 이 둘을 섞으면 모델에게 엉뚱한 안내가 나간다.
|
|
const r = applySearchReplaceEdits('a\n', block('없음', 'x'));
|
|
assert.equal(r.hadAnyBlock, true);
|
|
assert.equal(r.editCount, 0);
|
|
});
|
|
|
|
test('빈 입력에도 안전하다', () => {
|
|
const r = applySearchReplaceEdits('', '');
|
|
assert.equal(r.content, '');
|
|
assert.equal(r.editCount, 0);
|
|
assert.equal(r.hadAnyBlock, false);
|
|
});
|
|
});
|
|
|
|
describe('호출 간 상태 오염 방지', () => {
|
|
test('연속 호출이 서로 영향을 주지 않는다 (정규식 lastIndex)', () => {
|
|
// /g 정규식을 모듈 스코프에서 재사용하면 두 번째 호출이 중간부터 시작해 조용히 편집을 건너뛴다.
|
|
const edits = block('a', 'A');
|
|
const first = applySearchReplaceEdits('a\n', edits);
|
|
const second = applySearchReplaceEdits('a\n', edits);
|
|
assert.equal(first.editCount, 1);
|
|
assert.equal(second.editCount, 1, '두 번째 호출에서 편집이 유실됨 — lastIndex 오염');
|
|
});
|
|
});
|