| 1 | /** Pure transformations of Rust-captured data. No I/O, process, credential or Engine API. */ |
| 2 | import type { Json } from '../../protocol.ts' |
| 3 | import { transformWebSnapshot } from './web-adapters.ts' |
| 4 | import { transformGithubSnapshot } from './github-adapter.ts' |
| 5 | import { isReviewOperation, REVIEW_LIMIT, reviewEnvelopeBytes, transformReviewSnapshot } from './review-adapter.ts' |
| 6 | |
| 7 | interface Result { ok: true; result: { content: string; success: boolean; metadata: Json } } |
| 8 | type Row = Record<string, unknown> |
| 9 | const MAX_SNAPSHOT = 1024 * 1024 |
| 10 | |
| 11 | function row(value: unknown): Row { |
| 12 | if (!value || typeof value !== 'object' || Array.isArray(value)) throw new Error('invalid adapter snapshot') |
| 13 | return value as Row |
| 14 | } |
| 15 | function text(value: unknown): string { |
| 16 | if (typeof value !== 'string') throw new Error('invalid adapter text') |
| 17 | return value |
| 18 | } |
| 19 | function optionalText(value: unknown): string | undefined { |
| 20 | return value === null || value === undefined ? undefined : text(value) |
| 21 | } |
| 22 | function optionalNumber(value: unknown): number | undefined { |
| 23 | if (value === null || value === undefined) return undefined |
| 24 | if (typeof value !== 'number' || !Number.isFinite(value)) throw new Error('invalid captured numeric field') |
| 25 | return value |
| 26 | } |
| 27 | // Preserve Rust f64 values, negative zero, derived nonfinite values and exact i64 |
| 28 | // timestamps across JSON. Rust decodes these strings into its original DTO. |
| 29 | function float(value: number): string { |
| 30 | if (Object.is(value, -0)) return '-0' |
| 31 | if (value === Infinity) return 'inf' |
| 32 | if (value === -Infinity) return '-inf' |
| 33 | return String(value) |
| 34 | } |
| 35 | function timestamp(value: unknown): string | undefined { |
| 36 | const time = optionalText(value) |
| 37 | if (time === undefined) return undefined |
| 38 | if (!/^-?\d{1,19}$/.test(time)) throw new Error('invalid captured timestamp') |
| 39 | const integer = BigInt(time) |
| 40 | if (integer < -9223372036854775808n || integer > 9223372036854775807n) throw new Error('captured timestamp exceeds i64') |
| 41 | return time |
| 42 | } |
| 43 | function result(success: boolean, metadata: Json, content = ''): Result { |
| 44 | return { ok: true, result: { success, content, metadata } } |
| 45 | } |
| 46 | function failure(endpoint: string, kind: 'not_found' | 'upstream', detail: string): Result { |
| 47 | return result(false, { endpoint, kind, detail }) |
| 48 | } |
| 49 | |
| 50 | function asciiLower(value: string): string { return value.replace(/[A-Z]/g, letter => letter.toLowerCase()) } |
| 51 | |
| 52 | function finance(operation: string, input: Row): Result { |
| 53 | const request = row(input.request) |
| 54 | const requested = text(request.requested_ticker) |
| 55 | const symbol = text(request.resolved_symbol) |
| 56 | const parsed = row(input.parsed) |
| 57 | const chart = operation === 'finance_chart' |
| 58 | const source = chart ? 'yahoo_chart' : 'yahoo_quote' |
| 59 | let quote: Row |
| 60 | if (chart) { |
| 61 | const body = row(parsed.chart) |
| 62 | if (body.error !== null && body.error !== undefined) { |
| 63 | const error = row(body.error) |
| 64 | const description = optionalText(error.description) ?? 'chart endpoint returned an error' |
| 65 | const code = optionalText(error.code) |
| 66 | const notFound = (code === undefined ? undefined : asciiLower(code)) === 'not found' || asciiLower(description).includes('not found') || asciiLower(description).includes('symbol may be delisted') |
| 67 | return failure(source, notFound ? 'not_found' : 'upstream', description) |
| 68 | } |
| 69 | if (body.result !== null && body.result !== undefined && !Array.isArray(body.result)) throw new Error('invalid captured chart results') |
| 70 | const entries = body.result as unknown[] | null | undefined |
| 71 | if (!entries?.length) return failure(source, 'not_found', `no chart data for symbol '${symbol}'`) |
| 72 | quote = row(row(entries[0]).meta) |
| 73 | } else { |
| 74 | const body = row(parsed.quoteResponse) |
| 75 | if (!Array.isArray(body.result)) throw new Error('invalid captured quote results') |
| 76 | const candidate = body.result.map(row).find(value => asciiLower(text(value.symbol)) === asciiLower(symbol)) |
| 77 | if (!candidate) return failure(source, 'not_found', `no result for symbol '${symbol}'`) |
| 78 | quote = candidate |
| 79 | } |
| 80 | const price = optionalNumber(quote.regularMarketPrice) |
| 81 | if (price === undefined) return failure(source, 'upstream', 'response missing regularMarketPrice') |
| 82 | const previous = chart |
| 83 | ? optionalNumber(quote.chartPreviousClose) ?? optionalNumber(quote.previousClose) |
| 84 | : optionalNumber(quote.regularMarketPreviousClose) |
| 85 | const computedChange = previous === undefined ? undefined : price - previous |
| 86 | const computedPercent = previous === undefined || Math.abs(previous) < Number.EPSILON ? undefined : ((price - previous) / previous) * 100 |
| 87 | const change = chart ? computedChange : optionalNumber(quote.regularMarketChange) ?? computedChange |
| 88 | const percent = chart ? computedPercent : optionalNumber(quote.regularMarketChangePercent) ?? computedPercent |
| 89 | const name = optionalText(quote.longName) ?? optionalText(quote.shortName) |
| 90 | const currency = optionalText(quote.currency) |
| 91 | const state = chart ? undefined : optionalText(quote.marketState) |
| 92 | const type = optionalText(chart ? quote.instrumentType : quote.quoteType) |
| 93 | const exchange = optionalText(quote.fullExchangeName) ?? optionalText(chart ? quote.exchangeName : quote.exchange) |
| 94 | const time = timestamp(quote.regularMarketTime) |
| 95 | const value: Record<string, Json> = { |
| 96 | requested_ticker: requested, ticker: text(quote.symbol), price: float(price), source, fallback_used: chart, |
| 97 | ...(name === undefined ? {} : { name }), ...(currency === undefined ? {} : { currency }), |
| 98 | ...(change === undefined ? {} : { change: float(change) }), |
| 99 | ...(percent === undefined ? {} : { change_percent: float(percent) }), |
| 100 | ...(previous === undefined ? {} : { previous_close: float(previous) }), |
| 101 | ...(state === undefined ? {} : { market_state: state }), ...(type === undefined ? {} : { quote_type: type }), |
| 102 | ...(exchange === undefined ? {} : { exchange }), ...(time === undefined ? {} : { market_time: time }), |
| 103 | } |
| 104 | return result(true, value) |
| 105 | } |
| 106 | |
| 107 | function summary(value: unknown, format: 'json' | 'toml'): Json { |
| 108 | const descriptor = row(value) |
| 109 | const kind = text(descriptor.kind) |
| 110 | const accepted = format === 'json' |
| 111 | ? ['object', 'array', 'string', 'number', 'boolean', 'null'] |
| 112 | : ['table', 'array', 'string', 'integer', 'float', 'boolean', 'datetime'] |
| 113 | if (!accepted.includes(kind)) throw new Error('invalid captured parser kind') |
| 114 | if (kind === 'object' || kind === 'table') { |
| 115 | if (!Array.isArray(descriptor.keys) || descriptor.keys.some(key => typeof key !== 'string')) throw new Error('invalid captured parser keys') |
| 116 | // Core captured the parser's actual iteration order, including builds |
| 117 | // with serde_json preserve_order. Keep that exact preview order. |
| 118 | const keys = descriptor.keys as string[] |
| 119 | if (new Set(keys).size !== keys.length) throw new Error('duplicate captured parser keys') |
| 120 | return { top_level: kind, entries: keys.length, keys_preview: keys.slice(0, 10) } |
| 121 | } |
| 122 | if (kind === 'array') { |
| 123 | if (!Number.isSafeInteger(descriptor.entries) || (descriptor.entries as number) < 0) throw new Error('invalid captured array count') |
| 124 | return { top_level: kind, entries: descriptor.entries as number } |
| 125 | } |
| 126 | return { top_level: kind } |
| 127 | } |
| 128 | |
| 129 | function data(input: Row): Result { |
| 130 | const format = text(input.format) |
| 131 | if (!['auto', 'json', 'toml'].includes(format)) throw new Error('invalid captured data format') |
| 132 | const source = text(input.source) |
| 133 | const extension = optionalText(input.extension) |
| 134 | const requested = format === 'auto' && (extension === 'json' || extension === 'toml') ? extension : format |
| 135 | const json = row(input.json) |
| 136 | const toml = row(input.toml) |
| 137 | if (typeof json.ok !== 'boolean' || typeof toml.ok !== 'boolean') throw new Error('invalid captured parser result') |
| 138 | const use = (parser: Row, selected: 'json' | 'toml'): Result => { |
| 139 | if (parser.ok) return result(true, { valid: true, format: selected, source, summary: summary(parser.value, selected) }) |
| 140 | const error = text(parser.error) |
| 141 | return result(false, { valid: false, format: selected, source, error }, `Invalid ${selected.toUpperCase()}: ${error}`) |
| 142 | } |
| 143 | if (requested === 'json') return use(json, 'json') |
| 144 | if (requested === 'toml') return use(toml, 'toml') |
| 145 | if (json.ok) return use(json, 'json') |
| 146 | if (toml.ok) return use(toml, 'toml') |
| 147 | return result(false, { valid: false, format: 'auto', source, json_error: text(json.error), toml_error: text(toml.error) }, 'Validation failed in auto mode: content is neither valid JSON nor TOML.') |
| 148 | } |
| 149 | |
| 150 | // Unicode White_Space is Rust str::trim's contract. JS trim additionally removes |
| 151 | // BOM and omits NEL, which changes speech instructions on saved transcripts. |
| 152 | function rustTrim(value: string): string { return value.replace(/^[\u0009-\u000d\u0020\u0085\u00a0\u1680\u2000-\u200a\u2028\u2029\u202f\u205f\u3000]+|[\u0009-\u000d\u0020\u0085\u00a0\u1680\u2000-\u200a\u2028\u2029\u202f\u205f\u3000]+$/g, '') } |
| 153 | function bool(value: unknown): boolean { if (typeof value !== 'boolean') throw new Error('invalid captured boolean'); return value } |
| 154 | function speech(operation: string, input: Row): Result { |
| 155 | const surface = text(input.surface) |
| 156 | if (surface !== 'tool' && surface !== 'cli') throw new Error('invalid speech surface') |
| 157 | const tool = surface === 'tool' |
| 158 | const invalid = (error: string) => result(false, { error }) |
| 159 | if (operation === 'speech_format') { |
| 160 | const raw = text(input.format) |
| 161 | const format = asciiLower(rustTrim(raw)) |
| 162 | if (format === 'wav' || format === 'mp3' || format === 'pcm16' || format === 'pcm') return result(true, { format: format === 'pcm' ? 'pcm16' : format }) |
| 163 | return invalid(tool ? `unsupported speech format '${raw}' (allowed: wav, mp3, pcm16)` : `Unsupported speech format '${raw}' (allowed: wav, mp3, pcm16)`) |
| 164 | } |
| 165 | const modelHint = optionalText(input.model) |
| 166 | const hasClone = bool(input.has_clone_path) |
| 167 | const hasVoice = bool(input.has_voice) |
| 168 | const nonemptyVoice = bool(input.voice_nonempty) |
| 169 | const dataUri = bool(input.voice_is_data_uri) |
| 170 | if ((nonemptyVoice || dataUri) && !hasVoice) throw new Error('inconsistent captured voice presence') |
| 171 | const instructionInput = optionalText(input.instruction) |
| 172 | const promptInput = optionalText(input.voice_prompt) |
| 173 | if (hasClone && hasVoice) return invalid(tool ? 'use either clone_voice or voice for cloned voice data, not both' : 'Use either --clone-voice or --voice for cloned voice data, not both') |
| 174 | const model = modelHint ?? ((hasClone || dataUri) ? 'mimo-v2.5-tts-voiceclone' : promptInput !== undefined ? 'mimo-v2.5-tts-voicedesign' : 'mimo-v2.5-tts') |
| 175 | const lower = asciiLower(model) |
| 176 | if (!lower.includes('tts')) { |
| 177 | const examples = 'mimo-v2.5-tts, mimo-v2.5-tts-voicedesign, mimo-v2.5-tts-voiceclone, mimo-v2-tts' |
| 178 | return invalid(tool ? `speech tool requires a TTS model (examples: ${examples}), got '${model}'` : `speech requires a TTS model (examples: ${examples}); got ${model}`) |
| 179 | } |
| 180 | const parts = [promptInput, instructionInput].map(value => value === undefined ? undefined : rustTrim(value)).filter((value): value is string => value !== undefined && value !== '') |
| 181 | const instruction = parts.length ? parts.join('\n\n') : null |
| 182 | if (lower.includes('voicedesign') && instruction === null) return invalid(tool ? 'mimo-v2.5-tts-voicedesign requires voice_prompt or instruction' : 'mimo-v2.5-tts-voicedesign requires --voice-prompt or --instruction to describe the voice') |
| 183 | let voice: string |
| 184 | if (hasClone) voice = 'clone' |
| 185 | else if (lower.includes('voicedesign')) voice = 'omit' |
| 186 | else if (nonemptyVoice) voice = 'raw' |
| 187 | else if (lower.includes('voiceclone')) return invalid(tool ? 'mimo-v2.5-tts-voiceclone requires clone_voice <mp3|wav> or voice <data-uri>' : 'mimo-v2.5-tts-voiceclone requires --clone-voice <mp3|wav> or --voice <data-uri>') |
| 188 | else voice = 'default' |
| 189 | return result(true, { model, instruction, voice }) |
| 190 | } |
| 191 | |
| 192 | function pdfProjection(input: Row): Result { |
| 193 | const state = text(input.state) |
| 194 | const decision = (code: string, message?: string) => result(true, { kind: 'pdf_decision', code, ...(message === undefined ? {} : { message }) }) |
| 195 | if (state !== 'complete') { |
| 196 | if (!['binary_unavailable', 'cancelled', 'timed_out', 'execution'].includes(state)) throw new Error('invalid PDF process state') |
| 197 | return decision(state, text(input.message)) |
| 198 | } |
| 199 | const success = bool(input.success) |
| 200 | const stdoutOverflow = bool(input.stdout_truncated) |
| 201 | const stderrOverflow = bool(input.stderr_truncated) |
| 202 | const stderr = text(input.stderr) |
| 203 | const code = input.exit_code |
| 204 | if (code !== null && (!Number.isInteger(code) || (code as number) < -2147483648 || (code as number) > 2147483647)) throw new Error('invalid PDF exit code') |
| 205 | if (stdoutOverflow) return decision('execution', 'pdftotext output exceeded the 16777216 byte safety limit') |
| 206 | if (!success) return decision('execution', `pdftotext failed (exit ${code === null ? 'None' : `Some(${code})`}): ${stderr || 'no diagnostic output'}${stderrOverflow ? ' [truncated]' : ''}`) |
| 207 | return decision('success') |
| 208 | } |
| 209 | |
| 210 | function ocrProjection(input: Row): Result { |
| 211 | const decision = (code: string, trim_end = false, message?: string) => result(true, { kind: 'ocr_decision', code, trim_end, ...(message === undefined ? {} : { message }) }) |
| 212 | const state = text(input.state), status = text(input.status) |
| 213 | if (state === 'native') { |
| 214 | if (Object.keys(input).some(key => !['kind','state','status','can_fallback','next_ticket'].includes(key)) || !['success','error','unavailable'].includes(status)) throw new Error('invalid OCR Native projection') |
| 215 | const fallback = bool(input.can_fallback) |
| 216 | if (status === 'success') { |
| 217 | if (fallback || input.next_ticket !== undefined) throw new Error('successful OCR Native step cannot launch fallback') |
| 218 | return decision('native_success') |
| 219 | } |
| 220 | if (fallback) { |
| 221 | if (typeof input.next_ticket !== 'string' || input.next_ticket.length < 1 || input.next_ticket.length > 256) throw new Error('OCR fallback has no exact continuation grant') |
| 222 | return decision('fallback') |
| 223 | } |
| 224 | if (input.next_ticket !== undefined) throw new Error('OCR fallback is not admitted') |
| 225 | if (status === 'error') return decision('native_error') |
| 226 | return decision('no_backend', false, 'image_ocr: no local OCR backend is available. On macOS, update to a version with the Vision framework; on Linux/Windows install tesseract and restart codewhale.') |
| 227 | } |
| 228 | if (state !== 'tesseract') throw new Error('invalid OCR process stage') |
| 229 | if (status === 'fault') { |
| 230 | if (Object.keys(input).some(key => !['kind','state','status'].includes(key))) throw new Error('invalid OCR fault projection') |
| 231 | return decision('fault') |
| 232 | } |
| 233 | if (status !== 'complete' || Object.keys(input).some(key => !['kind','state','status','success','exit_code'].includes(key))) throw new Error('invalid OCR Tesseract projection') |
| 234 | const success = bool(input.success), code = input.exit_code |
| 235 | if (code !== null && (!Number.isInteger(code) || (code as number) < -2147483648 || (code as number) > 2147483647)) throw new Error('invalid OCR exit code') |
| 236 | return success ? decision('tesseract_success', true) : decision('execution', false, `tesseract failed (exit ${code === null ? 'None' : `Some(${code})`}): `) |
| 237 | } |
| 238 | |
| 239 | export function transformStockSnapshot(value: unknown): Result { |
| 240 | const snapshot = row(value) |
| 241 | const review = snapshot.kind === 'stock_adapter' && isReviewOperation(snapshot.operation) |
| 242 | if (review ? reviewEnvelopeBytes(snapshot) > REVIEW_LIMIT : Buffer.byteLength(JSON.stringify(value)) > MAX_SNAPSHOT) throw new Error(review ? 'serialized review snapshot/envelope exceeds 16 MiB' : 'adapter snapshot exceeds 1 MiB') |
| 243 | if (snapshot.kind === 'ocr_process') return ocrProjection(snapshot) |
| 244 | if (snapshot.kind === 'pdf_process') return pdfProjection(snapshot) |
| 245 | if (snapshot.kind !== 'stock_adapter') throw new Error('invalid adapter projection') |
| 246 | const input = row(snapshot.input) |
| 247 | if (review) return transformReviewSnapshot(snapshot.operation as string, input) |
| 248 | switch (snapshot.operation) { |
| 249 | case 'web_filters': case 'web_request': case 'web_provider': case 'web_entries': case 'web_finalize': case 'web_extract': case 'web_images': return transformWebSnapshot(snapshot.operation, input) |
| 250 | case 'github_result': return transformGithubSnapshot(input) |
| 251 | case 'finance_quote': case 'finance_chart': return finance(snapshot.operation, input) |
| 252 | case 'validate_data': return data(input) |
| 253 | case 'speech_options': case 'speech_format': return speech(snapshot.operation, input) |
| 254 | default: throw new Error('unadmitted adapter operation') |
| 255 | } |
| 256 | } |
| 257 |