| 1 | #!/usr/bin/env node |
| 2 | /** |
| 3 | * Collect evidence for pending live copy edits. |
| 4 | * |
| 5 | * This module intentionally does not edit source files and does not choose a |
| 6 | * winner. It gathers staged browser edits, rendered context, framework source |
| 7 | * hints, and likely source candidates so the AI copy-edit batch runner can make |
| 8 | * source changes with full repo context. |
| 9 | */ |
| 10 | |
| 11 | import fs from 'node:fs'; |
| 12 | import path from 'node:path'; |
| 13 | import { isGeneratedFile } from './is-generated.mjs'; |
| 14 | import { readBuffer, getBufferPath } from './live-manual-edits-buffer.mjs'; |
| 15 | |
| 16 | const EVIDENCE_VERSION = 1; |
| 17 | const TEXT_EXTENSIONS = new Set(['.html', '.jsx', '.tsx', '.vue', '.svelte', '.astro', '.js', '.mjs', '.ts']); |
| 18 | const SEARCH_DIRS = ['src', 'app', 'pages', 'components', 'public', 'views', 'templates', 'site', 'lib', 'data']; |
| 19 | const STRONG_LITERAL_MATCH_LIMIT = 8; |
| 20 | const WEAK_LITERAL_MATCH_LIMIT = 4; |
| 21 | const OBJECT_KEY_MATCH_LIMIT = 8; |
| 22 | const LOCATOR_MATCH_LIMIT = 4; |
| 23 | const CONTEXT_MATCH_LIMIT = 8; |
| 24 | const CONTEXT_MATCH_PER_HINT = 2; |
| 25 | const SKIP_DIRS = new Set([ |
| 26 | 'node_modules', |
| 27 | '.git', |
| 28 | '.impeccable', |
| 29 | '.astro', |
| 30 | '.next', |
| 31 | '.nuxt', |
| 32 | '.svelte-kit', |
| 33 | 'dist', |
| 34 | 'build', |
| 35 | 'out', |
| 36 | 'coverage', |
| 37 | ]); |
| 38 | |
| 39 | export function buildManualEditEvidence({ cwd = process.cwd(), pageUrl = null } = {}) { |
| 40 | const buffer = readBuffer(cwd); |
| 41 | const entries = pageUrl |
| 42 | ? buffer.entries.filter((entry) => entry.pageUrl === pageUrl) |
| 43 | : buffer.entries; |
| 44 | const opCount = countOps(entries); |
| 45 | |
| 46 | if (opCount === 0) { |
| 47 | return { |
| 48 | pageUrl, |
| 49 | count: 0, |
| 50 | entries: [], |
| 51 | ops: [], |
| 52 | candidates: [], |
| 53 | }; |
| 54 | } |
| 55 | |
| 56 | const searchFiles = collectSearchFiles(cwd); |
| 57 | const ops = flattenOps(entries); |
| 58 | const candidates = ops.map((op) => buildCandidatesForOp(op, cwd, searchFiles)); |
| 59 | return { |
| 60 | version: EVIDENCE_VERSION, |
| 61 | pageUrl: pageUrl || null, |
| 62 | count: opCount, |
| 63 | entries, |
| 64 | ops, |
| 65 | context: { |
| 66 | cwd, |
| 67 | bufferPath: path.relative(cwd, getBufferPath(cwd)), |
| 68 | totalEntries: entries.length, |
| 69 | totalOps: opCount, |
| 70 | }, |
| 71 | candidates, |
| 72 | }; |
| 73 | } |
| 74 | |
| 75 | function countOps(entries) { |
| 76 | let count = 0; |
| 77 | for (const entry of entries) count += Array.isArray(entry.ops) ? entry.ops.length : 0; |
| 78 | return count; |
| 79 | } |
| 80 | |
| 81 | function flattenOps(entries) { |
| 82 | const out = []; |
| 83 | for (const entry of entries) { |
| 84 | const contextHintsByRef = buildContextHintsByRef(entry); |
| 85 | for (const op of entry.ops || []) { |
| 86 | out.push({ |
| 87 | entryId: entry.id, |
| 88 | pageUrl: entry.pageUrl, |
| 89 | ref: op.ref, |
| 90 | contextRef: op.contextRef || null, |
| 91 | tag: op.tag, |
| 92 | elementId: op.elementId || null, |
| 93 | classes: Array.isArray(op.classes) ? op.classes : [], |
| 94 | originalText: op.originalText, |
| 95 | newText: op.newText, |
| 96 | deleted: op.deleted === true, |
| 97 | sourceHint: op.sourceHint || null, |
| 98 | leaf: op.leaf || null, |
| 99 | nearbyEditableTexts: Array.isArray(op.nearbyEditableTexts) ? op.nearbyEditableTexts : [], |
| 100 | container: op.container || null, |
| 101 | contextHints: contextHintsByRef.get(op.ref) || [], |
| 102 | }); |
| 103 | } |
| 104 | } |
| 105 | return out; |
| 106 | } |
| 107 | |
| 108 | function buildContextHintsByRef(entry) { |
| 109 | const map = new Map(); |
| 110 | for (const op of entry.ops || []) { |
| 111 | const hints = new Set(); |
| 112 | const add = (value) => { |
| 113 | const text = normalizeText(decodeBasicHtml(String(value || ''))); |
| 114 | if (text.length < 3 || text.length > 160) return; |
| 115 | if (text === normalizeText(op.originalText) || text === normalizeText(op.newText)) return; |
| 116 | hints.add(text); |
| 117 | }; |
| 118 | |
| 119 | for (const item of op.nearbyEditableTexts || []) { |
| 120 | add(typeof item === 'string' ? item : item?.text); |
| 121 | } |
| 122 | const outer = typeof entry.element?.outerHTML === 'string' ? entry.element.outerHTML : ''; |
| 123 | for (const match of outer.matchAll(/data-impeccable-original-text="([^"]*)"/g)) add(match[1]); |
| 124 | if (typeof entry.element?.textContent === 'string') { |
| 125 | for (const chunk of entry.element.textContent.split(/\s{2,}|\n|\t/)) add(chunk); |
| 126 | } |
| 127 | map.set(op.ref, [...hints].slice(0, 16)); |
| 128 | } |
| 129 | return map; |
| 130 | } |
| 131 | |
| 132 | function buildCandidatesForOp(op, cwd, searchFiles) { |
| 133 | const originalText = String(op.originalText || ''); |
| 134 | const contextNeedles = op.contextHints || []; |
| 135 | return { |
| 136 | entryId: op.entryId, |
| 137 | ref: op.ref, |
| 138 | originalText, |
| 139 | sourceHint: analyzeSourceHint(op, cwd), |
| 140 | textMatches: originalText ? findLiteralMatches(searchFiles, originalText, { max: literalMatchLimit(originalText) }) : [], |
| 141 | objectKeyMatches: originalText ? findObjectKeyMatches(searchFiles, originalText, { max: OBJECT_KEY_MATCH_LIMIT }) : [], |
| 142 | locatorMatches: findLocatorMatches(searchFiles, op, { max: LOCATOR_MATCH_LIMIT }), |
| 143 | contextTextMatches: findContextMatches(searchFiles, contextNeedles, { maxPerHint: CONTEXT_MATCH_PER_HINT, max: CONTEXT_MATCH_LIMIT }), |
| 144 | }; |
| 145 | } |
| 146 | |
| 147 | function literalMatchLimit(text) { |
| 148 | return isWeakSourceNeedle(text) ? WEAK_LITERAL_MATCH_LIMIT : STRONG_LITERAL_MATCH_LIMIT; |
| 149 | } |
| 150 | |
| 151 | function isWeakSourceNeedle(text) { |
| 152 | const normalized = normalizeText(text); |
| 153 | return normalized.length < 4 || /^[\d.,+\-%\s]+$/.test(normalized); |
| 154 | } |
| 155 | |
| 156 | function analyzeSourceHint(op, cwd) { |
| 157 | const hint = normalizeSourceHint(op.sourceHint); |
| 158 | if (!hint.file) return null; |
| 159 | const file = path.resolve(cwd, hint.file); |
| 160 | const relativeFile = path.relative(cwd, file); |
| 161 | if (!isPathInsideOrEqual(cwd, file)) { |
| 162 | return { ...hint, status: 'outside_cwd', relativeFile: hint.file }; |
| 163 | } |
| 164 | if (!fs.existsSync(file)) { |
| 165 | return { ...hint, status: 'file_missing', relativeFile }; |
| 166 | } |
| 167 | if (isGeneratedFile(file, { cwd })) { |
| 168 | return { ...hint, status: 'generated', relativeFile }; |
| 169 | } |
| 170 | |
| 171 | const content = fs.readFileSync(file, 'utf-8'); |
| 172 | const lines = content.split('\n'); |
| 173 | const line = hint.line || 1; |
| 174 | const start = Math.max(0, line - 4); |
| 175 | const end = Math.min(lines.length, line + 3); |
| 176 | const windowText = lines.slice(start, end).join('\n'); |
| 177 | const containsOriginalText = typeof op.originalText === 'string' && windowText.includes(op.originalText); |
| 178 | return { |
| 179 | ...hint, |
| 180 | status: containsOriginalText ? 'ok' : 'text_not_found_near_hint', |
| 181 | relativeFile, |
| 182 | excerpt: lines.slice(start, end).map((text, index) => ({ |
| 183 | line: start + index + 1, |
| 184 | text: text.slice(0, 240), |
| 185 | })), |
| 186 | }; |
| 187 | } |
| 188 | |
| 189 | function normalizeSourceHint(hint) { |
| 190 | if (!hint || typeof hint !== 'object') return {}; |
| 191 | let line = Number.isFinite(Number(hint.line)) ? Number(hint.line) : null; |
| 192 | let column = Number.isFinite(Number(hint.column)) ? Number(hint.column) : null; |
| 193 | if ((!line || !column) && typeof hint.loc === 'string') { |
| 194 | const match = hint.loc.match(/^(\d+)(?::(\d+))?/); |
| 195 | if (match) { |
| 196 | line = Number(match[1]); |
| 197 | if (match[2]) column = Number(match[2]); |
| 198 | } |
| 199 | } |
| 200 | return { |
| 201 | file: typeof hint.file === 'string' ? hint.file : '', |
| 202 | loc: typeof hint.loc === 'string' ? hint.loc : '', |
| 203 | line, |
| 204 | column, |
| 205 | }; |
| 206 | } |
| 207 | |
| 208 | function collectSearchFiles(cwd) { |
| 209 | const out = []; |
| 210 | const seenDirs = new Set(); |
| 211 | const seenFiles = new Set(); |
| 212 | for (const dir of SEARCH_DIRS) { |
| 213 | scanDir(path.join(cwd, dir), cwd, seenDirs, seenFiles, out, 0); |
| 214 | } |
| 215 | scanRootFiles(cwd, seenFiles, out); |
| 216 | return out; |
| 217 | } |
| 218 | |
| 219 | function scanDir(dir, cwd, seenDirs, seenFiles, out, depth) { |
| 220 | if (depth > 7 || !fs.existsSync(dir)) return; |
| 221 | let realDir; |
| 222 | try { realDir = fs.realpathSync(dir); } catch { return; } |
| 223 | if (seenDirs.has(realDir)) return; |
| 224 | seenDirs.add(realDir); |
| 225 | |
| 226 | let entries; |
| 227 | try { entries = fs.readdirSync(dir, { withFileTypes: true }); } catch { return; } |
| 228 | for (const entry of entries) { |
| 229 | const fullPath = path.join(dir, entry.name); |
| 230 | if (entry.isDirectory()) { |
| 231 | if (SKIP_DIRS.has(entry.name)) continue; |
| 232 | scanDir(fullPath, cwd, seenDirs, seenFiles, out, depth + 1); |
| 233 | continue; |
| 234 | } |
| 235 | if (!entry.isFile() || !TEXT_EXTENSIONS.has(path.extname(entry.name).toLowerCase())) continue; |
| 236 | maybeAddSearchFile(fullPath, cwd, seenFiles, out); |
| 237 | } |
| 238 | } |
| 239 | |
| 240 | function scanRootFiles(cwd, seenFiles, out) { |
| 241 | let entries; |
| 242 | try { entries = fs.readdirSync(cwd, { withFileTypes: true }); } catch { return; } |
| 243 | for (const entry of entries) { |
| 244 | if (!entry.isFile() || !TEXT_EXTENSIONS.has(path.extname(entry.name).toLowerCase())) continue; |
| 245 | maybeAddSearchFile(path.join(cwd, entry.name), cwd, seenFiles, out); |
| 246 | } |
| 247 | } |
| 248 | |
| 249 | function maybeAddSearchFile(file, cwd, seenFiles, out) { |
| 250 | let realFile; |
| 251 | try { realFile = fs.realpathSync(file); } catch { return; } |
| 252 | if (seenFiles.has(realFile)) return; |
| 253 | seenFiles.add(realFile); |
| 254 | if (isGeneratedFile(file, { cwd })) return; |
| 255 | let content; |
| 256 | try { content = fs.readFileSync(file, 'utf-8'); } catch { return; } |
| 257 | out.push({ file, relativeFile: path.relative(cwd, file), content, lines: content.split('\n') }); |
| 258 | } |
| 259 | |
| 260 | function findLiteralMatches(searchFiles, needle, { max }) { |
| 261 | return findMatches(searchFiles, needle, { kind: 'text', max }); |
| 262 | } |
| 263 | |
| 264 | function findObjectKeyMatches(searchFiles, text, { max }) { |
| 265 | const re = new RegExp('(["\\\'`])' + escapeRegExp(text) + '\\1(?=\\s*:)', 'g'); |
| 266 | const out = []; |
| 267 | for (const file of searchFiles) { |
| 268 | for (const match of file.content.matchAll(re)) { |
| 269 | out.push(matchForIndex(file, match.index, 'object_key', text)); |
| 270 | if (out.length >= max) return out; |
| 271 | } |
| 272 | } |
| 273 | return out; |
| 274 | } |
| 275 | |
| 276 | function findLocatorMatches(searchFiles, op, { max }) { |
| 277 | const needles = []; |
| 278 | if (op.elementId) needles.push({ kind: 'id', needle: op.elementId }); |
| 279 | for (const cls of op.classes || []) { |
| 280 | if (cls) needles.push({ kind: 'class', needle: cls }); |
| 281 | } |
| 282 | if (op.tag) needles.push({ kind: 'tag', needle: '<' + op.tag }); |
| 283 | |
| 284 | const out = []; |
| 285 | const seen = new Set(); |
| 286 | for (const { kind, needle } of needles) { |
| 287 | for (const match of findMatches(searchFiles, needle, { kind, max })) { |
| 288 | const key = match.file + ':' + match.line + ':' + kind + ':' + needle; |
| 289 | if (seen.has(key)) continue; |
| 290 | seen.add(key); |
| 291 | out.push({ ...match, needle }); |
| 292 | if (out.length >= max) return out; |
| 293 | } |
| 294 | } |
| 295 | return out; |
| 296 | } |
| 297 | |
| 298 | function findContextMatches(searchFiles, hints, { maxPerHint, max }) { |
| 299 | const out = []; |
| 300 | const seen = new Set(); |
| 301 | for (const hint of hints || []) { |
| 302 | for (const match of findMatches(searchFiles, hint, { kind: 'context', max: maxPerHint })) { |
| 303 | const key = match.file + ':' + match.line + ':' + hint; |
| 304 | if (seen.has(key)) continue; |
| 305 | seen.add(key); |
| 306 | out.push({ ...match, needle: hint }); |
| 307 | if (out.length >= max) return out; |
| 308 | } |
| 309 | } |
| 310 | return out; |
| 311 | } |
| 312 | |
| 313 | function findMatches(searchFiles, needle, { kind, max }) { |
| 314 | const text = String(needle || ''); |
| 315 | if (!text) return []; |
| 316 | const out = []; |
| 317 | for (const file of searchFiles) { |
| 318 | let index = 0; |
| 319 | while (out.length < max) { |
| 320 | index = file.content.indexOf(text, index); |
| 321 | if (index === -1) break; |
| 322 | out.push(matchForIndex(file, index, kind, text)); |
| 323 | index += Math.max(1, text.length); |
| 324 | } |
| 325 | if (out.length >= max) break; |
| 326 | } |
| 327 | return out; |
| 328 | } |
| 329 | |
| 330 | function matchForIndex(file, index, kind, needle) { |
| 331 | const line = file.content.slice(0, index).split('\n').length; |
| 332 | const lineText = file.lines[line - 1] || ''; |
| 333 | return { |
| 334 | kind, |
| 335 | file: file.relativeFile, |
| 336 | line, |
| 337 | needle, |
| 338 | excerpt: lineText.trim().slice(0, 240), |
| 339 | }; |
| 340 | } |
| 341 | |
| 342 | function isPathInsideOrEqual(cwd, file) { |
| 343 | const rel = path.relative(path.resolve(cwd), path.resolve(file)); |
| 344 | return rel === '' || (!rel.startsWith('..') && !path.isAbsolute(rel)); |
| 345 | } |
| 346 | |
| 347 | function normalizeText(value) { |
| 348 | return String(value || '').replace(/\s+/g, ' ').trim(); |
| 349 | } |
| 350 | |
| 351 | function decodeBasicHtml(value) { |
| 352 | return value |
| 353 | .replace(/"/g, '"') |
| 354 | .replace(/'/g, "'") |
| 355 | .replace(/'/g, "'") |
| 356 | .replace(/&/g, '&') |
| 357 | .replace(/</g, '<') |
| 358 | .replace(/>/g, '>'); |
| 359 | } |
| 360 | |
| 361 | function escapeRegExp(value) { |
| 362 | return String(value).replace(/[.*+?^${}()|[\]\\]/g, '\\$&'); |
| 363 | } |
| 364 |