| 1 | /** Pure proposals from Core-captured web facts. No HTTP, file, credential, clock or Engine API. */ |
| 2 | import type { Json } from '../../protocol.ts' |
| 3 | type Row = Record<string, unknown> |
| 4 | type Entry = { title: string, url: string, snippet?: string } |
| 5 | const own = (value: Row, key: string) => Object.prototype.hasOwnProperty.call(value, key) |
| 6 | function row(value: unknown): Row { if (!value || typeof value !== 'object' || Array.isArray(value)) throw new Error('invalid captured web record'); return value as Row } |
| 7 | function text(value: unknown): string { if (typeof value !== 'string') throw new Error('invalid captured web text'); return value } |
| 8 | function string(value: unknown): string | undefined { return typeof value === 'string' ? value : undefined } |
| 9 | function array(value: unknown): unknown[] { return Array.isArray(value) ? value : [] } |
| 10 | function count(value: unknown, max = 10): number { if (!Number.isSafeInteger(value) || (value as number) < 0 || (value as number) > max) throw new Error('invalid captured web count'); return value as number } |
| 11 | function rustTrim(value: string): string { return value.replace(/^[\u0009-\u000d\u0020\u0085\u00a0\u1680\u2000-\u200a\u2028\u2029\u202f\u205f\u3000]+|[\u0009-\u000d\u0020\u0085\u00a0\u1680\u2000-\u200a\u2028\u2029\u202f\u205f\u3000]+$/g, '') } |
| 12 | function firstPresent(value: Row, keys: string[]): unknown { for (const key of keys) if (own(value,key)) return value[key]; return undefined } |
| 13 | function firstText(value: Row, keys: string[]): string | undefined { for (const key of keys) { const found = string(value[key]); if (found !== undefined) { const trimmed = rustTrim(found); if (trimmed) return trimmed } } return undefined } |
| 14 | function integer(facts: Row, path: string): string | undefined { const value = facts[path]; if (value === undefined) return undefined; const number = row(value); if (number.i64 === null || number.i64 === undefined) return undefined; const result = text(number.i64); if (!/^-?\d{1,19}$/.test(result) || BigInt(result) < -9223372036854775808n || BigInt(result)>9223372036854775807n) throw new Error('invalid captured web integer'); return result } |
| 15 | function number(facts: Row, path: string): number { const value = facts[path]; if (value === undefined) return 0; const result = Number(text(row(value).score)); if (!Number.isFinite(result)) throw new Error('invalid captured web score'); return result } |
| 16 | function result(metadata: Json) { return { ok: true as const, result: { success: true, content: '', metadata } } } |
| 17 | |
| 18 | function filters(input: Row): Json { |
| 19 | const recency = input.recency_days === null || input.recency_days === undefined ? undefined : count(input.recency_days,3650) |
| 20 | const locale = input.locale === null || input.locale === undefined ? undefined : rustTrim(text(input.locale)) |
| 21 | const parts = locale?.split(/[-_]/) |
| 22 | const language = parts?.[0] !== undefined && /^[A-Za-z]{2,3}$/.test(parts[0]) ? parts[0].toLowerCase() : null |
| 23 | const region = parts?.[1] !== undefined && /^[A-Za-z]{2}$/.test(parts[1]) ? parts[1].toUpperCase() : null |
| 24 | const window = recency === undefined ? null : recency <= 1 ? 'day' : recency <= 7 ? 'week' : recency <= 31 ? 'month' : 'year' |
| 25 | return { kind:'web_filters', window, language, region } |
| 26 | } |
| 27 | |
| 28 | function validJson(value: unknown): boolean { |
| 29 | const pending:{value:unknown,depth:number}[]=[{value,depth:0}] |
| 30 | while(pending.length) { |
| 31 | const item=pending.pop()! |
| 32 | if(typeof item.value==='number'&&!Number.isFinite(item.value))return false |
| 33 | if(typeof item.value==='string'&&[...item.value].some(ch=>ch.length===1&&ch.charCodeAt(0)>=0xd800&&ch.charCodeAt(0)<=0xdfff))return false |
| 34 | if(item.value&&typeof item.value==='object') { |
| 35 | if(item.depth>=128)return false |
| 36 | for(const child of Object.values(item.value))pending.push({value:child,depth:item.depth+1}) |
| 37 | } |
| 38 | } |
| 39 | return true |
| 40 | } |
| 41 | |
| 42 | function provider(input: Row): Json { |
| 43 | const backend = text(input.backend), limit = count(input.max_results), parsed = row(input.parsed), facts = row(input.number_facts) |
| 44 | let entries: Entry[] = [], error: string | null = null |
| 45 | let items: unknown[] = [], titleKeys = ['title'], urlKeys = ['url'], snippetKeys = ['content','snippet'], trim = true, capSnippet = false, scorePath: string | undefined |
| 46 | switch (backend) { |
| 47 | case 'tavily': items = array(parsed.results); break |
| 48 | case 'firecrawl': { |
| 49 | const data = parsed.data |
| 50 | items = array(data && typeof data === 'object' && !Array.isArray(data) && own(row(data),'web') ? row(data).web : data) |
| 51 | snippetKeys=['description','markdown','content']; capSnippet=true |
| 52 | if (parsed.success === false) error=`Firecrawl search failed: ${firstText(parsed,['error','message']) ?? 'unknown API error'}` |
| 53 | break |
| 54 | } |
| 55 | case 'metaso': { |
| 56 | items=array(parsed.webpages); urlKeys=['link']; snippetKeys=['snippet','summary'] |
| 57 | const code=integer(facts,'/code') |
| 58 | if (code !== undefined && code !== '0') error = code === '3003' ? 'Metaso: daily search limit reached — set METASO_API_KEY or get one at https://metaso.cn/search-api/playground' : code === '2005' ? 'Metaso API key rejected — check METASO_API_KEY or set `[search] api_key` in config.toml' : `Metaso API error (code ${code}: ${string(parsed.message) ?? 'unknown error'})` |
| 59 | break |
| 60 | } |
| 61 | case 'bocha': { |
| 62 | const data = parsed.data && typeof parsed.data === 'object' && !Array.isArray(parsed.data) ? row(parsed.data) : undefined |
| 63 | let pages: unknown |
| 64 | if (data) { |
| 65 | const web = data.webPages && typeof data.webPages === 'object' && !Array.isArray(data.webPages) ? row(data.webPages) : undefined |
| 66 | pages = web && own(web,'value') ? web.value : data.pages |
| 67 | } |
| 68 | if (pages === undefined) pages=parsed.pages |
| 69 | items=array(pages); titleKeys=['name','title']; urlKeys=['url','link']; snippetKeys=['summary','snippet','description'] |
| 70 | const code=integer(facts,'/code') |
| 71 | if (code !== undefined && code !== '0' && code !== '200') error=`Bocha search API error (code ${code}: ${string(firstPresent(parsed,['msg','message'])) ?? 'unknown error'})` |
| 72 | break |
| 73 | } |
| 74 | case 'baidu': { |
| 75 | items=array(parsed.references); titleKeys=['title','name']; urlKeys=['url','link']; snippetKeys=['content','snippet','summary'] |
| 76 | const key=own(parsed,'error_code')?'error_code':'code', code=integer(facts,`/${key}`) |
| 77 | if (code !== undefined && code !== '0') error=`Baidu search API error (code ${code}: ${string(firstPresent(parsed,['error_msg','message'])) ?? 'unknown error'})` |
| 78 | break |
| 79 | } |
| 80 | case 'searxng': items=array(parsed.results); scorePath='/results'; break |
| 81 | case 'sofya': items=array(parsed.results); snippetKeys=['content','description']; trim=false; break |
| 82 | case 'serply': items=array(parsed.results); urlKeys=['link']; snippetKeys=['description','snippet']; trim=false; break |
| 83 | case 'volcengine': { |
| 84 | if (own(parsed,'error')) { |
| 85 | const value = parsed.error && typeof parsed.error==='object' && !Array.isArray(parsed.error) ? row(parsed.error) : {} |
| 86 | error=`Volcengine API error (code ${string(value.code) ?? 'unknown'}: ${string(value.message) ?? 'no details'})` |
| 87 | } |
| 88 | let response: string | undefined |
| 89 | const message=array(parsed.output).slice().reverse().find(value=>value && typeof value==='object' && !Array.isArray(value) && row(value).type==='message') |
| 90 | if (message) { const content=array(row(message).content).find(value=>value && typeof value==='object' && !Array.isArray(value) && typeof row(value).text==='string'); if(content) response=text(row(content).text) } |
| 91 | if (response === undefined && error === null) error='Volcengine response contains no output text' |
| 92 | if (response !== undefined) { |
| 93 | let body=response |
| 94 | const fence=response.indexOf('```json') |
| 95 | if(fence>=0) {const rest=response.slice(fence+7),end=rest.indexOf('```');if(end>=0)body=rustTrim(rest.slice(0,end));else {const first=response.indexOf('{'),last=response.lastIndexOf('}');if(first>=0&&last>=first)body=response.slice(first,last+1)}} |
| 96 | else {const first=response.indexOf('{'),last=response.lastIndexOf('}');if(first>=0&&last>=first)body=response.slice(first,last+1)} |
| 97 | try {const value=JSON.parse(body);if(!validJson(value))throw new Error('model JSON failed Core-compatible scalar/depth guard');if(value&&typeof value==='object'&&!Array.isArray(value))items=array(row(value).results)} catch {items=[]} |
| 98 | } |
| 99 | snippetKeys=['snippet']; break |
| 100 | } |
| 101 | default: throw new Error('unadmitted web provider') |
| 102 | } |
| 103 | const scored: {entry:Entry,score:number}[]=[] |
| 104 | for (const [index,value] of items.entries()) { |
| 105 | if(!value || typeof value!=='object' || Array.isArray(value))continue |
| 106 | const item=row(value),rawTitle=string(firstPresent(item,titleKeys)),rawUrl=string(firstPresent(item,urlKeys)) |
| 107 | if(rawTitle===undefined||rawUrl===undefined)continue |
| 108 | const title=trim?rustTrim(rawTitle):rawTitle,url=trim?rustTrim(rawUrl):rawUrl |
| 109 | if(trim&&(!title||!url))continue |
| 110 | // Some legacy providers select the first present snippet alias even if it |
| 111 | // is null/wrong/empty; others choose the first nonempty string. |
| 112 | const presentSnippet=['bocha','baidu','volcengine'].includes(backend) |
| 113 | const selected=presentSnippet?string(firstPresent(item,snippetKeys)):undefined |
| 114 | const snippet=presentSnippet?(selected===undefined?undefined:rustTrim(selected)||undefined):firstText(item,snippetKeys) |
| 115 | const entry={title,url,...(snippet===undefined?{}:{snippet:capSnippet?[...snippet].slice(0,1000).join(''):snippet})} |
| 116 | scored.push({entry,score:scorePath===undefined?0:number(facts,`${scorePath}/${index}/score`)}) |
| 117 | if(scorePath===undefined&&scored.length===limit)break |
| 118 | } |
| 119 | if(scorePath!==undefined)scored.sort((a,b)=>a.score === b.score ? Object.is(a.score,-0) === Object.is(b.score,-0) ? 0 : Object.is(a.score,-0) ? 1 : -1 : a.score > b.score ? -1 : 1) |
| 120 | entries=scored.slice(0,limit).map(value=>value.entry) |
| 121 | return {kind:'web_provider',entries,error} |
| 122 | } |
| 123 | |
| 124 | function extraction(input: Row): Json { |
| 125 | if(!Array.isArray(input.candidates) || input.candidates.length>3)throw new Error('invalid captured web regions') |
| 126 | const facts: {id:number,chars:number,words:number}[]=[] |
| 127 | let previous=-1 |
| 128 | for(const value of input.candidates) { |
| 129 | const candidate=row(value),id=count(candidate.id,2) |
| 130 | if(Object.keys(candidate).some(key=>!['id','non_whitespace','words'].includes(key)) || id<=previous)throw new Error('invalid ordered captured web region');previous=id |
| 131 | facts.push({id,chars:count(candidate.non_whitespace,100_000_000),words:count(candidate.words,100_000_000)}) |
| 132 | } |
| 133 | return {kind:'web_extract',candidate:facts.find(value=>value.chars>=32&&value.words>=5)?.id ?? null} |
| 134 | } |
| 135 | |
| 136 | function images(input: Row): Json { |
| 137 | const limit=count(input.max_results),parsed=row(input.parsed) |
| 138 | if(!Array.isArray(parsed.results))throw new Error('invalid captured image results') |
| 139 | const entries:Json[]=[] |
| 140 | for(const value of parsed.results) { |
| 141 | const entry=row(value),image=text(entry.image) |
| 142 | for(const key of ['thumbnail','title','url','source'])if(entry[key]!==null&&entry[key]!==undefined)text(entry[key]) |
| 143 | for(const key of ['width','height'])if(entry[key]!==null&&entry[key]!==undefined)count(entry[key],4294967295) |
| 144 | if(!rustTrim(image))continue |
| 145 | entries.push({image,...Object.fromEntries(['thumbnail','title','url','source','width','height'].filter(key=>entry[key]!==null&&entry[key]!==undefined).map(key=>[key,entry[key] as Json]))}) |
| 146 | // Domain filtering is mandatory Core URL policy and happens before the |
| 147 | // final count cap, so all bounded candidates are preserved here. |
| 148 | } |
| 149 | return {kind:'web_images',entries,max_results:limit} |
| 150 | } |
| 151 | |
| 152 | // Core chooses the endpoint, method, model and credential injection. This |
| 153 | // adapter owns only the keyless query payload or query-pair proposal. |
| 154 | function request(input: Row): Json { |
| 155 | const backend=text(input.backend), query=text(input.query), max=count(input.max_results) |
| 156 | const f=row(input.filters), window=f.window===null?undefined:text(f.window), region=f.region===null?undefined:text(f.region), language=f.language===null?undefined:text(f.language) |
| 157 | const locale=input.locale===null||input.locale===undefined?undefined:text(input.locale) |
| 158 | let payload:Json=null, pairs:Json=[] |
| 159 | switch(backend) { |
| 160 | case 'firecrawl': payload={query,limit:max,sources:[{type:'web'}],...(window===undefined?{}:{tbs:`qdr:${window[0]}`}),...(region===undefined?{}:{country:region})};break |
| 161 | case 'tavily':payload={query,search_depth:'basic',max_results:max,...(window===undefined?{}:{time_range:window})};break |
| 162 | case 'bocha':payload={query,freshness:'noLimit',count:max};break |
| 163 | case 'metaso':payload={q:query,scope:'webpage',size:Math.max(1,Math.min(100,max))};break |
| 164 | case 'sofya':payload={query,max_results:max};break |
| 165 | case 'baidu':payload={messages:[{role:'user',content:query}],search_source:'baidu_search_v2',resource_type_filter:[{type:'web',top_k:max}]};break |
| 166 | case 'volcengine':payload={input:[{role:'user',content:[{type:'input_text',text:`Search the web for: ${query}\n\nCRITICAL: Respond ONLY with a valid JSON object. No markdown, no explanation.\nSchema: {"results":[{"title":"...","url":"https://...","snippet":"..."}]}\n- results: 1-${max} most relevant pages\n- title: page title (required)\n- url: full URL starting with https:// (required)\n- snippet: 1-2 sentence factual summary (required)\n- If zero results: {"results":[]}\n- Your entire response must be valid, parseable JSON.`}]}]};break |
| 167 | case 'serply':pairs=[['q',query],['num',String(max)],...(language===undefined?[]:[['hl',language]]),...(region===undefined?[]:[['gl',region.toLowerCase()]])];break |
| 168 | case 'searxng':pairs=[['q',query],['format','json'],...(window===undefined?[]:[['time_range',window]]),...(locale===undefined?[]:[['language',locale]])];break |
| 169 | case 'bing':case 'duckduckgo':pairs=[['q',query]];break |
| 170 | default:throw new Error('unadmitted web request provider') |
| 171 | } |
| 172 | return {kind:'web_request',payload,pairs} |
| 173 | } |
| 174 | |
| 175 | // These are already-parsed native/scraper candidates. URLs are opaque Core |
| 176 | // handles; no credentials, configured endpoint, source ref or domain crosses. |
| 177 | function entries(input: Row): Json { |
| 178 | if(!Array.isArray(input.entries))throw new Error('invalid captured web candidates') |
| 179 | const values:Json[]=input.entries.map(value=>{ |
| 180 | const item=row(value) |
| 181 | return {title:text(item.title),url:text(item.url),...(item.snippet===null||item.snippet===undefined?{}:{snippet:text(item.snippet)}),...(item.published===null||item.published===undefined?{}:{published:text(item.published)})} |
| 182 | }) |
| 183 | return {kind:'web_entries',entries:values} |
| 184 | } |
| 185 | |
| 186 | function finalization(input: Row): Json { |
| 187 | const requested=row(input.requested),capabilities=row(input.capabilities),counted=count(input.count) |
| 188 | if(!Array.isArray(input.degraded))throw new Error('invalid captured web receipt') |
| 189 | const degraded:Json[]=input.degraded.map(value=>row(value) as Json) |
| 190 | const ignored=(knob:string)=>degraded.some(value=>row(value).kind==='knob_ignored'&&row(value).knob===knob) |
| 191 | const honored={max_results:capabilities.max_results==='supported',recency:false,domains:requested.domains===true,locale:false} |
| 192 | for(const knob of ['recency','locale'] as const) { |
| 193 | if(knob==='locale') { |
| 194 | if(!Array.isArray(input.domain_extra))throw new Error('invalid captured domain receipt') |
| 195 | degraded.push(...input.domain_extra.map(value=>row(value) as Json)) |
| 196 | } |
| 197 | if(requested[knob]===true&&!ignored(knob)) { |
| 198 | if(capabilities[knob]==='supported')honored[knob]=true |
| 199 | else degraded.push({kind:'knob_ignored',knob}) |
| 200 | } |
| 201 | } |
| 202 | const hasNote=input.has_note===true |
| 203 | const prefix=counted===0 ? hasNote?'No results found. ':'No results found' : `Found ${counted} result(s)${hasNote?'. ':''}` |
| 204 | const suffix=degraded.some(value=>row(value).kind==='answer_cut_by_provider')?'\n[the provider stopped the search answer at its output limit; the answer is incomplete]':'' |
| 205 | return {kind:'web_finalize',honored,degraded,prefix,suffix} |
| 206 | } |
| 207 | |
| 208 | export function transformWebSnapshot(operation: string, value: unknown) { |
| 209 | const input=row(value) |
| 210 | switch(operation) { |
| 211 | case 'web_request':return result(request(input)) |
| 212 | case 'web_entries':return result(entries(input)) |
| 213 | case 'web_finalize':return result(finalization(input)) |
| 214 | case 'web_filters':return result(filters(input)) |
| 215 | case 'web_provider':return result(provider(input)) |
| 216 | case 'web_extract':return result(extraction(input)) |
| 217 | case 'web_images':return result(images(input)) |
| 218 | default:throw new Error('unadmitted web adapter operation') |
| 219 | } |
| 220 | } |
| 221 |