返回 AiToEarn
live-manual-edit-evidence.mjs
根目录 / project / aitoearn-web / .agents / skills / impeccable / scripts / live-manual-edit-evidence.mjs
1 #!/usr/bin/env node
2 /**
3 * Collect evidence for pending live copy edits.
4 *
5 * This module intentionally does not edit source files and does not choose a
6 * winner. It gathers staged browser edits, rendered context, framework source
7 * hints, and likely source candidates so the AI copy-edit batch runner can make
8 * source changes with full repo context.
9 */
10
11 import fs from 'node:fs';
12 import path from 'node:path';
13 import { isGeneratedFile } from './is-generated.mjs';
14 import { readBuffer, getBufferPath } from './live-manual-edits-buffer.mjs';
15
16 const EVIDENCE_VERSION = 1;
17 const TEXT_EXTENSIONS = new Set(['.html', '.jsx', '.tsx', '.vue', '.svelte', '.astro', '.js', '.mjs', '.ts']);
18 const SEARCH_DIRS = ['src', 'app', 'pages', 'components', 'public', 'views', 'templates', 'site', 'lib', 'data'];
19 const STRONG_LITERAL_MATCH_LIMIT = 8;
20 const WEAK_LITERAL_MATCH_LIMIT = 4;
21 const OBJECT_KEY_MATCH_LIMIT = 8;
22 const LOCATOR_MATCH_LIMIT = 4;
23 const CONTEXT_MATCH_LIMIT = 8;
24 const CONTEXT_MATCH_PER_HINT = 2;
25 const SKIP_DIRS = new Set([
26 'node_modules',
27 '.git',
28 '.impeccable',
29 '.astro',
30 '.next',
31 '.nuxt',
32 '.svelte-kit',
33 'dist',
34 'build',
35 'out',
36 'coverage',
37 ]);
38
39 export function buildManualEditEvidence({ cwd = process.cwd(), pageUrl = null } = {}) {
40 const buffer = readBuffer(cwd);
41 const entries = pageUrl
42 ? buffer.entries.filter((entry) => entry.pageUrl === pageUrl)
43 : buffer.entries;
44 const opCount = countOps(entries);
45
46 if (opCount === 0) {
47 return {
48 pageUrl,
49 count: 0,
50 entries: [],
51 ops: [],
52 candidates: [],
53 };
54 }
55
56 const searchFiles = collectSearchFiles(cwd);
57 const ops = flattenOps(entries);
58 const candidates = ops.map((op) => buildCandidatesForOp(op, cwd, searchFiles));
59 return {
60 version: EVIDENCE_VERSION,
61 pageUrl: pageUrl || null,
62 count: opCount,
63 entries,
64 ops,
65 context: {
66 cwd,
67 bufferPath: path.relative(cwd, getBufferPath(cwd)),
68 totalEntries: entries.length,
69 totalOps: opCount,
70 },
71 candidates,
72 };
73 }
74
75 function countOps(entries) {
76 let count = 0;
77 for (const entry of entries) count += Array.isArray(entry.ops) ? entry.ops.length : 0;
78 return count;
79 }
80
81 function flattenOps(entries) {
82 const out = [];
83 for (const entry of entries) {
84 const contextHintsByRef = buildContextHintsByRef(entry);
85 for (const op of entry.ops || []) {
86 out.push({
87 entryId: entry.id,
88 pageUrl: entry.pageUrl,
89 ref: op.ref,
90 contextRef: op.contextRef || null,
91 tag: op.tag,
92 elementId: op.elementId || null,
93 classes: Array.isArray(op.classes) ? op.classes : [],
94 originalText: op.originalText,
95 newText: op.newText,
96 deleted: op.deleted === true,
97 sourceHint: op.sourceHint || null,
98 leaf: op.leaf || null,
99 nearbyEditableTexts: Array.isArray(op.nearbyEditableTexts) ? op.nearbyEditableTexts : [],
100 container: op.container || null,
101 contextHints: contextHintsByRef.get(op.ref) || [],
102 });
103 }
104 }
105 return out;
106 }
107
108 function buildContextHintsByRef(entry) {
109 const map = new Map();
110 for (const op of entry.ops || []) {
111 const hints = new Set();
112 const add = (value) => {
113 const text = normalizeText(decodeBasicHtml(String(value || '')));
114 if (text.length < 3 || text.length > 160) return;
115 if (text === normalizeText(op.originalText) || text === normalizeText(op.newText)) return;
116 hints.add(text);
117 };
118
119 for (const item of op.nearbyEditableTexts || []) {
120 add(typeof item === 'string' ? item : item?.text);
121 }
122 const outer = typeof entry.element?.outerHTML === 'string' ? entry.element.outerHTML : '';
123 for (const match of outer.matchAll(/data-impeccable-original-text="([^"]*)"/g)) add(match[1]);
124 if (typeof entry.element?.textContent === 'string') {
125 for (const chunk of entry.element.textContent.split(/\s{2,}|\n|\t/)) add(chunk);
126 }
127 map.set(op.ref, [...hints].slice(0, 16));
128 }
129 return map;
130 }
131
132 function buildCandidatesForOp(op, cwd, searchFiles) {
133 const originalText = String(op.originalText || '');
134 const contextNeedles = op.contextHints || [];
135 return {
136 entryId: op.entryId,
137 ref: op.ref,
138 originalText,
139 sourceHint: analyzeSourceHint(op, cwd),
140 textMatches: originalText ? findLiteralMatches(searchFiles, originalText, { max: literalMatchLimit(originalText) }) : [],
141 objectKeyMatches: originalText ? findObjectKeyMatches(searchFiles, originalText, { max: OBJECT_KEY_MATCH_LIMIT }) : [],
142 locatorMatches: findLocatorMatches(searchFiles, op, { max: LOCATOR_MATCH_LIMIT }),
143 contextTextMatches: findContextMatches(searchFiles, contextNeedles, { maxPerHint: CONTEXT_MATCH_PER_HINT, max: CONTEXT_MATCH_LIMIT }),
144 };
145 }
146
147 function literalMatchLimit(text) {
148 return isWeakSourceNeedle(text) ? WEAK_LITERAL_MATCH_LIMIT : STRONG_LITERAL_MATCH_LIMIT;
149 }
150
151 function isWeakSourceNeedle(text) {
152 const normalized = normalizeText(text);
153 return normalized.length < 4 || /^[\d.,+\-%\s]+$/.test(normalized);
154 }
155
156 function analyzeSourceHint(op, cwd) {
157 const hint = normalizeSourceHint(op.sourceHint);
158 if (!hint.file) return null;
159 const file = path.resolve(cwd, hint.file);
160 const relativeFile = path.relative(cwd, file);
161 if (!isPathInsideOrEqual(cwd, file)) {
162 return { ...hint, status: 'outside_cwd', relativeFile: hint.file };
163 }
164 if (!fs.existsSync(file)) {
165 return { ...hint, status: 'file_missing', relativeFile };
166 }
167 if (isGeneratedFile(file, { cwd })) {
168 return { ...hint, status: 'generated', relativeFile };
169 }
170
171 const content = fs.readFileSync(file, 'utf-8');
172 const lines = content.split('\n');
173 const line = hint.line || 1;
174 const start = Math.max(0, line - 4);
175 const end = Math.min(lines.length, line + 3);
176 const windowText = lines.slice(start, end).join('\n');
177 const containsOriginalText = typeof op.originalText === 'string' && windowText.includes(op.originalText);
178 return {
179 ...hint,
180 status: containsOriginalText ? 'ok' : 'text_not_found_near_hint',
181 relativeFile,
182 excerpt: lines.slice(start, end).map((text, index) => ({
183 line: start + index + 1,
184 text: text.slice(0, 240),
185 })),
186 };
187 }
188
189 function normalizeSourceHint(hint) {
190 if (!hint || typeof hint !== 'object') return {};
191 let line = Number.isFinite(Number(hint.line)) ? Number(hint.line) : null;
192 let column = Number.isFinite(Number(hint.column)) ? Number(hint.column) : null;
193 if ((!line || !column) && typeof hint.loc === 'string') {
194 const match = hint.loc.match(/^(\d+)(?::(\d+))?/);
195 if (match) {
196 line = Number(match[1]);
197 if (match[2]) column = Number(match[2]);
198 }
199 }
200 return {
201 file: typeof hint.file === 'string' ? hint.file : '',
202 loc: typeof hint.loc === 'string' ? hint.loc : '',
203 line,
204 column,
205 };
206 }
207
208 function collectSearchFiles(cwd) {
209 const out = [];
210 const seenDirs = new Set();
211 const seenFiles = new Set();
212 for (const dir of SEARCH_DIRS) {
213 scanDir(path.join(cwd, dir), cwd, seenDirs, seenFiles, out, 0);
214 }
215 scanRootFiles(cwd, seenFiles, out);
216 return out;
217 }
218
219 function scanDir(dir, cwd, seenDirs, seenFiles, out, depth) {
220 if (depth > 7 || !fs.existsSync(dir)) return;
221 let realDir;
222 try { realDir = fs.realpathSync(dir); } catch { return; }
223 if (seenDirs.has(realDir)) return;
224 seenDirs.add(realDir);
225
226 let entries;
227 try { entries = fs.readdirSync(dir, { withFileTypes: true }); } catch { return; }
228 for (const entry of entries) {
229 const fullPath = path.join(dir, entry.name);
230 if (entry.isDirectory()) {
231 if (SKIP_DIRS.has(entry.name)) continue;
232 scanDir(fullPath, cwd, seenDirs, seenFiles, out, depth + 1);
233 continue;
234 }
235 if (!entry.isFile() || !TEXT_EXTENSIONS.has(path.extname(entry.name).toLowerCase())) continue;
236 maybeAddSearchFile(fullPath, cwd, seenFiles, out);
237 }
238 }
239
240 function scanRootFiles(cwd, seenFiles, out) {
241 let entries;
242 try { entries = fs.readdirSync(cwd, { withFileTypes: true }); } catch { return; }
243 for (const entry of entries) {
244 if (!entry.isFile() || !TEXT_EXTENSIONS.has(path.extname(entry.name).toLowerCase())) continue;
245 maybeAddSearchFile(path.join(cwd, entry.name), cwd, seenFiles, out);
246 }
247 }
248
249 function maybeAddSearchFile(file, cwd, seenFiles, out) {
250 let realFile;
251 try { realFile = fs.realpathSync(file); } catch { return; }
252 if (seenFiles.has(realFile)) return;
253 seenFiles.add(realFile);
254 if (isGeneratedFile(file, { cwd })) return;
255 let content;
256 try { content = fs.readFileSync(file, 'utf-8'); } catch { return; }
257 out.push({ file, relativeFile: path.relative(cwd, file), content, lines: content.split('\n') });
258 }
259
260 function findLiteralMatches(searchFiles, needle, { max }) {
261 return findMatches(searchFiles, needle, { kind: 'text', max });
262 }
263
264 function findObjectKeyMatches(searchFiles, text, { max }) {
265 const re = new RegExp('(["\\\'`])' + escapeRegExp(text) + '\\1(?=\\s*:)', 'g');
266 const out = [];
267 for (const file of searchFiles) {
268 for (const match of file.content.matchAll(re)) {
269 out.push(matchForIndex(file, match.index, 'object_key', text));
270 if (out.length >= max) return out;
271 }
272 }
273 return out;
274 }
275
276 function findLocatorMatches(searchFiles, op, { max }) {
277 const needles = [];
278 if (op.elementId) needles.push({ kind: 'id', needle: op.elementId });
279 for (const cls of op.classes || []) {
280 if (cls) needles.push({ kind: 'class', needle: cls });
281 }
282 if (op.tag) needles.push({ kind: 'tag', needle: '<' + op.tag });
283
284 const out = [];
285 const seen = new Set();
286 for (const { kind, needle } of needles) {
287 for (const match of findMatches(searchFiles, needle, { kind, max })) {
288 const key = match.file + ':' + match.line + ':' + kind + ':' + needle;
289 if (seen.has(key)) continue;
290 seen.add(key);
291 out.push({ ...match, needle });
292 if (out.length >= max) return out;
293 }
294 }
295 return out;
296 }
297
298 function findContextMatches(searchFiles, hints, { maxPerHint, max }) {
299 const out = [];
300 const seen = new Set();
301 for (const hint of hints || []) {
302 for (const match of findMatches(searchFiles, hint, { kind: 'context', max: maxPerHint })) {
303 const key = match.file + ':' + match.line + ':' + hint;
304 if (seen.has(key)) continue;
305 seen.add(key);
306 out.push({ ...match, needle: hint });
307 if (out.length >= max) return out;
308 }
309 }
310 return out;
311 }
312
313 function findMatches(searchFiles, needle, { kind, max }) {
314 const text = String(needle || '');
315 if (!text) return [];
316 const out = [];
317 for (const file of searchFiles) {
318 let index = 0;
319 while (out.length < max) {
320 index = file.content.indexOf(text, index);
321 if (index === -1) break;
322 out.push(matchForIndex(file, index, kind, text));
323 index += Math.max(1, text.length);
324 }
325 if (out.length >= max) break;
326 }
327 return out;
328 }
329
330 function matchForIndex(file, index, kind, needle) {
331 const line = file.content.slice(0, index).split('\n').length;
332 const lineText = file.lines[line - 1] || '';
333 return {
334 kind,
335 file: file.relativeFile,
336 line,
337 needle,
338 excerpt: lineText.trim().slice(0, 240),
339 };
340 }
341
342 function isPathInsideOrEqual(cwd, file) {
343 const rel = path.relative(path.resolve(cwd), path.resolve(file));
344 return rel === '' || (!rel.startsWith('..') && !path.isAbsolute(rel));
345 }
346
347 function normalizeText(value) {
348 return String(value || '').replace(/\s+/g, ' ').trim();
349 }
350
351 function decodeBasicHtml(value) {
352 return value
353 .replace(/&quot;/g, '"')
354 .replace(/&#39;/g, "'")
355 .replace(/&apos;/g, "'")
356 .replace(/&amp;/g, '&')
357 .replace(/&lt;/g, '<')
358 .replace(/&gt;/g, '>');
359 }
360
361 function escapeRegExp(value) {
362 return String(value).replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
363 }
364
364 lines Plain Text