返回 DeepSeek-Reasonix
task.go
根目录 / internal / agent / task.go
1 package agent
2
3 import (
4 "context"
5 "encoding/json"
6 "fmt"
7 "io"
8 "path/filepath"
9 "runtime/debug"
10 "slices"
11 "strconv"
12 "strings"
13
14 "reasonix/internal/ablation"
15 "reasonix/internal/checkpoint"
16 "reasonix/internal/event"
17 "reasonix/internal/evidence"
18 "reasonix/internal/imageinput"
19 "reasonix/internal/jobs"
20 "reasonix/internal/memory"
21 "reasonix/internal/permission"
22 "reasonix/internal/planmode"
23 "reasonix/internal/provider"
24 "reasonix/internal/sandbox"
25 "reasonix/internal/sessiontemp"
26 "reasonix/internal/tool"
27 "reasonix/internal/workspacelease"
28 )
29
30 // DefaultTaskSystemPrompt steers a sub-agent toward focused, terse delivery —
31 // it doesn't see the parent's conversation so it must self-contain.
32 const DefaultTaskSystemPrompt = `You are a sub-agent invoked by a parent coding agent to carry out one focused task.
33 Use the provided tools to investigate or act. For MCP, use the stable use_capability
34 proxy (list → inspect → call); do not expect direct mcp__* tool schemas. Return a
35 single final answer that is concise and self-contained — the parent will see only
36 that answer, not your tool calls or reasoning. If you need to ask for clarification,
37 fail with a precise question instead of guessing.`
38
39 // DefaultReadOnlyTaskSystemPrompt steers read-only sub-agents toward isolated
40 // research. They never receive writer tools, persisted transcript controls, or
41 // background process controls, so their final answer is the only handoff.
42 const DefaultReadOnlyTaskSystemPrompt = `You are a read-only research sub-agent invoked by a parent coding agent.
43 Use only the provided read-only tools to inspect code, docs, history, and safe shell output.
44 For MCP, use use_capability only for authorized tools that declare readOnly and are
45 not destructive; never treat missing readOnlyHint as permission to call. Do not
46 attempt to write files, install capabilities, mutate memory, control long-lived
47 processes, or delegate to writer-capable agents. If a read-only delegation tool is
48 available and genuinely useful, you may use it within the configured depth limit.
49 Return a concise, self-contained final answer with the evidence the parent needs.`
50
51 const subagentStartContext = `<subagent-context event="SubagentStart">
52 Before acting, check the available skills and tools. If a relevant skill is available, invoke it before continuing. Delegate to another sub-agent only when the task genuinely benefits from isolated context and the delegation tool is available.
53 </subagent-context>`
54
55 // read_skill is deliberately not listed: it renders playbook text inline and
56 // cannot recurse, so depth-capped sub-agents keep it and can still read
57 // playbooks even when they can no longer delegate.
58 var subagentRecursiveTools = []string{
59 "task",
60 "read_only_task",
61 "run_skill",
62 "read_only_skill",
63 "explore",
64 "research",
65 "review",
66 "security_review",
67 }
68
69 var subagentAlwaysHiddenTools = []string{
70 "parallel_tasks",
71 "fleet",
72 "read_subagent_result",
73 "set_session_title",
74 "install_skill",
75 "install_source",
76 // Kept in the parent registry only as a clear retirement tombstone for
77 // replayed/model-stale calls. New child contexts must never advertise it.
78 "complete_step",
79 }
80
81 var subagentJobTools = []string{
82 "job_output",
83 "job_kill",
84 "wait",
85 "bash_output",
86 "kill_shell",
87 }
88
89 var readOnlySubagentWorkflowTools = []string{
90 "connect_tool_source",
91 }
92
93 const subagentToolBoundarySummary = "Recursive agent/skill tools are exposed only while max_subagent_depth leaves another delegation layer; background job tools (job_output/job_kill and the legacy wait/bash_output/kill_shell aliases) are excluded; the platform shell is exposed as foreground-only inside subagents."
94
95 // maxConcurrentBackgroundTasks is the legacy writer-background fallback used
96 // only when a TaskTool has no session scheduler (tests). Production boots
97 // inject MaxParallelWriters via SubagentScheduler.
98 const maxConcurrentBackgroundTasks = DefaultMaxParallelWriters
99
100 // AlwaysHiddenSubagentTools returns the tool names excluded from every
101 // subagent's registry regardless of an explicit allowlist or delegation
102 // depth (unlike subagentRecursiveTools, which depends on remaining depth).
103 // That covers both subagentAlwaysHiddenTools and subagentJobTools —
104 // SubagentToolRegistryForDepth and its read-only variant strip the job tools
105 // unconditionally too. Host UIs offering a tool picker for a subagent
106 // profile's allowed-tools should exclude these from the offered choices —
107 // selecting them would be silently ignored at runtime.
108 func AlwaysHiddenSubagentTools() []string {
109 names := append([]string(nil), subagentAlwaysHiddenTools...)
110 return append(names, subagentJobTools...)
111 }
112
113 // SubagentMetaTools returns the tool names that spawned agents should not inherit
114 // from the parent registry unless a future call site deliberately opts into a
115 // different boundary. They can spawn or author more agent work, so excluding them
116 // preserves one layer of delegation without adding a spawn-count cap.
117 // read_skill stays listed here so the guardian and planner surfaces, which
118 // exclude these names, keep their provider-visible tool sets byte-identical —
119 // only the sub-agent depth cap deliberately stopped stripping it.
120 func SubagentMetaTools() []string {
121 out := append([]string(nil), subagentRecursiveTools...)
122 out = append(out, "read_skill")
123 out = append(out, subagentAlwaysHiddenTools...)
124 return out
125 }
126
127 // SubagentToolRegistry returns the tool set exposed inside spawned sub-agents:
128 // the requested whitelist (or every parent tool), minus meta tools that would
129 // spawn more agent work and job tools whose runtime manager is not injected into
130 // sub-agents. When bash is present, it is wrapped to advertise and allow only
131 // foreground execution.
132 func SubagentToolRegistry(parent *tool.Registry, names []string) *tool.Registry {
133 return SubagentToolRegistryForDepth(parent, names, 1, 1)
134 }
135
136 // SubagentToolRegistryForDepth returns the writer-capable tool set for a spawned
137 // subagent at childDepth. Recursive delegation tools are available only when the
138 // child still has room to spawn one more subagent.
139 //
140 // Direct mcp__* schemas are never exposed: MCP goes only through the fixed
141 // use_capability proxy so connect/disconnect/tool-list churn cannot change the
142 // child provider-visible tool prefix. With no explicit allowlist the child gets
143 // the full proxy (installed/authorized MCP, including tools without
144 // readOnlyHint). An explicit allowlist converts mcp__* / mcp-tool: names into a
145 // capability-id allowlist on a restricted proxy.
146 func SubagentToolRegistryForDepth(parent *tool.Registry, names []string, childDepth, maxDepth int) *tool.Registry {
147 return SubagentToolRegistryForDepthWithRuntime(parent, names, childDepth, maxDepth, nil)
148 }
149
150 // SubagentToolRegistryForDepthWithRuntime is SubagentToolRegistryForDepth with
151 // an optional session MCP runtime used when the parent registry has no
152 // use_capability (for example Economy or legacy callers) but sub-agents still
153 // need the proxy.
154 func SubagentToolRegistryForDepthWithRuntime(parent *tool.Registry, names []string, childDepth, maxDepth int, runtime *MCPCapabilityRuntime) *tool.Registry {
155 exclude := append([]string(nil), subagentAlwaysHiddenTools...)
156 if childDepth >= NormalizeMaxSubagentDepth(maxDepth) {
157 exclude = append(exclude, subagentRecursiveTools...)
158 }
159 exclude = append(exclude, subagentJobTools...)
160 sub := FilterRegistry(parent, names, exclude...)
161 stripDirectMCPTools(sub)
162 AttachCompleteSubtaskTool(sub)
163 attachSubagentCapabilityProxy(parent, sub, names, runtime)
164 shellName := "bash"
165 if _, ok := sub.Get("pwsh"); ok {
166 shellName = "pwsh"
167 sub.RemovePrefix("bash")
168 }
169 if shell, ok := sub.Get(shellName); ok {
170 sub.Add(foregroundOnlyBash{inner: shell})
171 }
172 return sub
173 }
174
175 type foregroundOnlyBash struct {
176 inner tool.Tool
177 }
178
179 func (b foregroundOnlyBash) Name() string { return b.inner.Name() }
180
181 func (b foregroundOnlyBash) Description() string {
182 desc := strings.TrimSpace(b.inner.Description())
183 if desc == "" {
184 desc = "Execute a command in the shell and return combined stdout/stderr."
185 }
186 desc = strings.Replace(desc, "Execute a command in the shell", "Execute a foreground command in the shell", 1)
187 return desc + " Background execution is unavailable inside subagents."
188 }
189
190 func (b foregroundOnlyBash) Schema() json.RawMessage {
191 if b.Name() == "pwsh" {
192 return json.RawMessage(`{"type":"object","properties":{"command":{"type":"string","description":"PowerShell command to execute in the foreground"},"description":{"type":"string","description":"Clear 5-10 word active-voice description shown in the UI"},"timeout_ms":{"type":"integer","minimum":1}},"required":["command","description"]}`)
193 }
194 return json.RawMessage(`{"type":"object","properties":{"command":{"type":"string","description":"Shell command to execute in the foreground"}},"required":["command"]}`)
195 }
196
197 func (b foregroundOnlyBash) Execute(ctx context.Context, args json.RawMessage) (string, error) {
198 var p struct {
199 RunInBackground bool `json:"run_in_background"`
200 }
201 if err := json.Unmarshal(args, &p); err != nil {
202 return "", fmt.Errorf("invalid args: %w", err)
203 }
204 if p.RunInBackground {
205 return "", tool.Blocked(fmt.Sprintf("blocked: background %s is unavailable in subagents; run a foreground command or ask the parent agent to start a background job", b.Name()))
206 }
207 return b.inner.Execute(ctx, args)
208 }
209
210 func (b foregroundOnlyBash) ReadOnly() bool { return b.inner.ReadOnly() }
211
212 type readOnlyBash struct {
213 inner tool.Tool
214 }
215
216 func (b readOnlyBash) Name() string { return b.inner.Name() }
217
218 func (b readOnlyBash) Description() string {
219 desc := strings.TrimSpace(b.inner.Description())
220 if desc == "" {
221 desc = "Execute a command in the shell and return combined stdout/stderr."
222 }
223 desc = strings.Replace(desc, "Execute a command in the shell", "Execute a foreground read-only command in the shell", 1)
224 return desc + " Only permission-classified read-only commands are allowed; shell operators, background execution, process preservation, and write-capable arguments are blocked."
225 }
226
227 func (b readOnlyBash) Schema() json.RawMessage {
228 if b.Name() == "pwsh" {
229 return json.RawMessage(`{"type":"object","properties":{"command":{"type":"string","description":"Read-only PowerShell command to execute in the foreground"},"description":{"type":"string","description":"Clear 5-10 word active-voice description shown in the UI"},"timeout_ms":{"type":"integer","minimum":1}},"required":["command","description"]}`)
230 }
231 return json.RawMessage(`{"type":"object","properties":{"command":{"type":"string","description":"Read-only shell command to execute in the foreground. Must match the permission-layer read-only command policy."}},"required":["command"]}`)
232 }
233
234 func (b readOnlyBash) Execute(ctx context.Context, args json.RawMessage) (string, error) {
235 if !permission.BashCommandIsReadOnly(args) {
236 return "", tool.Blocked("blocked: read-only subagents can run only permission-classified foreground read-only commands")
237 }
238 return b.inner.Execute(ctx, args)
239 }
240
241 func (readOnlyBash) ReadOnly() bool { return true }
242
243 // TaskTool spawns a sub-agent in its own session for a focused sub-task. The
244 // sub-agent runs with a filtered tool whitelist and the same step budget shape
245 // as the parent (see Execute); its tool calls are forwarded to the parent's
246 // event stream nested under this call, while only its final assistant message is
247 // returned to the parent model. Use cases: keep noisy tool sequences (multi-file
248 // exploration, repeated grep / read_file) out of the parent's context budget, or
249 // parallel research across independent areas (the parallel-dispatch path picks
250 // these up only when readOnly, which task is not).
251 type TaskTool struct {
252 imageInput *imageinput.Config
253 prov provider.Provider
254 pricing *provider.Pricing
255 quoteContext *event.QuoteContext
256 parentReg *tool.Registry
257 maxSteps int
258 contextWindow int
259 compactRatio float64
260 recentKeep int
261 temperature float64
262 archiveDir string
263 keepPolicy KeepPolicy
264 sysPrompt string
265 gate Gate
266 subagentModel, subagentEffort string
267 resolveProvider func(modelRef, effort string) (provider.Provider, *provider.Pricing, int, error)
268 transcripts *SubagentStore
269 workspaceRoot string
270 baseModel string
271 baseEffort string
272 identityProfile func(modelRef, effort string) (string, string)
273 maxSubagentDepth int
274 ablation ablation.Set
275 workspaceLease *workspacelease.Owner
276 // scheduler is the session-scoped concurrency + write-claim controller.
277 // nil falls back to the legacy jobs.ReserveStart cap for background tasks.
278 scheduler *SubagentScheduler
279 // profileLookup resolves profile= names from the live Skill store without
280 // embedding the name list in the tool schema (cache stability).
281 profileLookup ProfileLookup
282 // profileConfigModel/Effort look up persistent per-profile overrides
283 // (agent.subagent_models / subagent_efforts).
284 profileConfigModel func(profile string) string
285 profileConfigEffort func(profile string) string
286 // bashSandboxEnforced reports whether OS sandbox can honour write roots
287 // for bash inside path-bound writer sub-agents.
288 bashSandboxEnforced func() bool
289 // mutationObserver is shared with spawned sub-agents for checkpoint capture.
290 mutationObserver *checkpoint.MutationObserver
291 writeRoots *sandbox.WritableRootSet
292 imageResolver ImageRequestResolver
293 hooksForSession func(string) ToolHooks
294 // capabilityRuntime is the session-shared MCP Host/specs substrate. Each
295 // sub-agent gets its own use_capability frontend so ledger state stays
296 // isolated while connections reuse the parent Host.
297 capabilityRuntime *MCPCapabilityRuntime
298 }
299
300 // NewTaskTool wires a task tool to the parent agent's environment so its
301 // sub-agents can use the same provider and tools. sysPrompt is the system
302 // prompt every sub-agent starts with; pass "" for DefaultTaskSystemPrompt. gate
303 // is the permission gate sub-agents inherit — pass the headless variant so
304 // deny rules still bite while autonomous sub-agents are never blocked on an
305 // interactive prompt (there is no UI to answer one).
306 //
307 // Compatibility wrapper: new call sites should prefer NewTaskToolWithOptions.
308 // The positional form is kept for at least one full iteration cycle.
309 func NewTaskTool(prov provider.Provider, pricing *provider.Pricing, parentReg *tool.Registry,
310 maxSteps, contextWindow, recentKeep int, softCompactRatio, toolResultSnipRatio, compactRatio, compactForceRatio, temperature float64, archiveDir, sysPrompt string, gate Gate,
311 keepPolicy KeepPolicy, subagentModel, subagentEffort string, resolveProvider func(string, string) (provider.Provider, *provider.Pricing, int, error)) *TaskTool {
312 return NewTaskToolWithOptions(TaskToolOptions{
313 Provider: prov,
314 Pricing: pricing,
315 ParentRegistry: parentReg,
316 MaxSteps: maxSteps,
317 ContextWindow: contextWindow,
318 RecentKeep: recentKeep,
319 SoftCompactRatio: softCompactRatio,
320 ToolResultSnipRatio: toolResultSnipRatio,
321 CompactRatio: compactRatio,
322 CompactForceRatio: compactForceRatio,
323 Temperature: temperature,
324 ArchiveDir: archiveDir,
325 SysPrompt: sysPrompt,
326 Gate: gate,
327 KeepPolicy: keepPolicy,
328 SubagentModel: subagentModel,
329 SubagentEffort: subagentEffort,
330 ResolveProvider: resolveProvider,
331 })
332 }
333
334 // WithTranscripts enables persisted sub-agent transcript continuation for this
335 // task tool. The base model/effort are the parent provider identity used when no
336 // subagent override is configured.
337 func (t *TaskTool) WithTranscripts(store *SubagentStore, workspaceRoot, baseModel, baseEffort string) *TaskTool {
338 t.transcripts = store
339 t.workspaceRoot = strings.TrimSpace(workspaceRoot)
340 t.baseModel = strings.TrimSpace(baseModel)
341 t.baseEffort = strings.TrimSpace(baseEffort)
342 return t
343 }
344
345 func (t *TaskTool) WithTranscriptIdentityResolver(resolve func(modelRef, effort string) (string, string)) *TaskTool {
346 t.identityProfile = resolve
347 return t
348 }
349
350 func (t *TaskTool) WithMaxSubagentDepth(depth int) *TaskTool {
351 t.maxSubagentDepth = NormalizeMaxSubagentDepth(depth)
352 return t
353 }
354
355 // WithAblation propagates the parent's benchmark arm so a sub-agent runs with
356 // the same subsystems switched off.
357 func (t *TaskTool) WithAblation(set ablation.Set) *TaskTool {
358 t.ablation = set
359 return t
360 }
361
362 // WithWorkspaceLease shares the parent's workspace-wide delivery write lease
363 // with every spawned sub-agent. A shared owner is required: independent owners
364 // in one session would deadlock when a child tries to write while its parent
365 // already retains the lease.
366 func (t *TaskTool) WithWorkspaceLease(owner *workspacelease.Owner) *TaskTool {
367 t.workspaceLease = owner
368 return t
369 }
370
371 // WithScheduler attaches the session-scoped concurrency and write-claim
372 // controller used by task, fleet, parallel_tasks, and profile skill runners.
373 func (t *TaskTool) WithScheduler(s *SubagentScheduler) *TaskTool {
374 t.scheduler = s
375 return t
376 }
377
378 // Scheduler returns the attached session scheduler (may be nil in unit tests).
379 func (t *TaskTool) Scheduler() *SubagentScheduler {
380 if t == nil {
381 return nil
382 }
383 return t.scheduler
384 }
385
386 // WithProfileLookup enables task/fleet profile= resolution from the Skill store.
387 func (t *TaskTool) WithProfileLookup(lookup ProfileLookup) *TaskTool {
388 t.profileLookup = lookup
389 return t
390 }
391
392 // WithProfileConfigResolvers supplies persistent per-profile model/effort
393 // overrides (agent.subagent_models / subagent_efforts).
394 func (t *TaskTool) WithProfileConfigResolvers(model, effort func(profile string) string) *TaskTool {
395 t.profileConfigModel = model
396 t.profileConfigEffort = effort
397 return t
398 }
399
400 // WithBashSandboxEnforced tells path-bound writer runs whether bash can keep
401 // the same write roots under the OS sandbox.
402 func (t *TaskTool) WithBashSandboxEnforced(fn func() bool) *TaskTool {
403 t.bashSandboxEnforced = fn
404 return t
405 }
406
407 // WithCapabilityRuntime attaches the session-shared MCP runtime so ordinary and
408 // read-only sub-agents receive a stable use_capability frontend without
409 // inheriting dynamic mcp__* schemas.
410 func (t *TaskTool) WithCapabilityRuntime(rt *MCPCapabilityRuntime) *TaskTool {
411 if t != nil {
412 t.capabilityRuntime = rt
413 }
414 return t
415 }
416
417 func (t *TaskTool) Name() string { return tool.HostTask }
418
419 func (t *TaskTool) Description() string {
420 return "Spawn a sub-agent for a focused sub-task. Optional profile selects a runAs=subagent Skill whose body becomes the full system prompt (no implicit concise default). Optional write_paths declare non-overlapping write targets so background writers may run in parallel; omitting write_paths on a writer claims the whole workspace and serializes writers. The sub-agent runs in its own session with a filtered tool list (defaults to every parent tool, then applies the subagent boundary: " + subagentToolBoundarySummary + "). Only its final answer is returned."
421 }
422
423 func (t *TaskTool) Schema() json.RawMessage {
424 return json.RawMessage(`{
425 "type":"object",
426 "properties":{
427 "prompt":{"type":"string","description":"What the sub-agent should accomplish. Be specific about the deliverable — the sub-agent does not see this conversation."},
428 "description":{"type":"string","description":"Short label for the sub-task (3-7 words). Surfaced in the dispatch line so the user sees what's running."},
429 "profile":{"type":"string","description":"Optional runAs=subagent profile name. Resolved at runtime from the Skill store; explicit names may invoke invocation=manual profiles. The profile body becomes the full system prompt."},
430 "write_paths":{"type":"array","items":{"type":"string"},"description":"Optional workspace-relative or absolute file/directory paths this writer may modify. Globs and workspace escapes are rejected. Writers without write_paths claim the whole workspace (serializing against every other writer claim). Non-overlapping paths allow parallel writers up to max_parallel_writers. In fleet, multiple whole-workspace claims fail preflight before any task starts."},
431 "tools":{"type":"array","items":{"type":"string"},"description":"Optional tool whitelist. When profile sets allowed-tools, this list is intersected (call args cannot expand profile permissions). ` + subagentToolBoundarySummary + `"},
432 "max_steps":{"type":"integer","description":"Optional cap on tool-call rounds. Defaults to half the parent's cap (min 5).","minimum":1},
433 "run_in_background":{"type":"boolean","description":"Run the sub-agent asynchronously: returns a job id immediately and keeps working across turns. Collect its final answer with job_output, and you'll be notified when it finishes. Use for long, independent sub-tasks you don't need to block on right now."},
434 "model":{"type":"string","description":"Optional model override for the sub-agent (a configured provider/model name). Precedence: persistent profile config, this argument, profile frontmatter, global subagent default, parent model."},
435 "effort":{"type":"string","description":"Optional reasoning effort for the sub-agent (e.g. high, max). Same precedence as model."},
436 "continue_from":{"type":"string","description":"Continue a prior compatible subagent transcript in the current conversation context. Pass only the 'sa_...' value from the prior result's 'Subagent reference: ...' line. If the ref belongs to an ancestor conversation, the framework continues a current-conversation copy."}
437 },
438 "required":["prompt"]
439 }`)
440 }
441
442 // ReadOnly is false: a sub-agent can invoke any whitelisted tool, including
443 // writers. Conservative classification keeps the parallel-dispatch path from
444 // running two sub-agents at once and letting their writes race.
445 func (t *TaskTool) ReadOnly() bool { return false }
446
447 // ResolveProfile extracts model/effort from task args (and optional profile
448 // overrides) for dispatch-line display. Runtime execution re-resolves with the
449 // full precedence chain.
450 func (t *TaskTool) ResolveProfile(args json.RawMessage) *event.Profile {
451 var p struct {
452 Model string `json:"model"`
453 Effort string `json:"effort"`
454 Profile string `json:"profile"`
455 }
456 if err := json.Unmarshal(args, &p); err != nil {
457 return nil
458 }
459 profileModel, profileEffort := "", ""
460 configModel, configEffort := "", ""
461 if name := strings.TrimSpace(p.Profile); name != "" {
462 if def, err := ResolveProfileDefinition(t.profileLookup, name); err == nil {
463 profileModel, profileEffort = def.Model, def.Effort
464 }
465 if t.profileConfigModel != nil {
466 configModel = t.profileConfigModel(name)
467 }
468 if t.profileConfigEffort != nil {
469 configEffort = t.profileConfigEffort(name)
470 }
471 }
472 model, effort := ResolveModelEffort(
473 configModel, configEffort,
474 p.Model, p.Effort,
475 profileModel, profileEffort,
476 t.subagentModel, t.subagentEffort,
477 )
478 if model == "" && effort == "" {
479 return nil
480 }
481 return &event.Profile{Model: model, Effort: effort}
482 }
483
484 // ReadOnlyTaskTool runs an isolated sub-agent with a strictly read-only tool
485 // registry. It intentionally omits background execution and transcript
486 // continuation/fork controls so the call has no durable host side effects.
487 type ReadOnlyTaskTool struct {
488 task *TaskTool
489 }
490
491 func NewReadOnlyTaskTool(task *TaskTool) *ReadOnlyTaskTool {
492 return &ReadOnlyTaskTool{task: task}
493 }
494
495 func (*ReadOnlyTaskTool) Name() string { return tool.HostReadOnlyTask }
496
497 func (*ReadOnlyTaskTool) Description() string {
498 return "Spawn a read-only research sub-agent for a focused investigation. The sub-agent runs in an isolated, ephemeral session with read-only tools only; bash is wrapped to allow only permission-classified foreground read-only commands. It cannot write files, install capabilities, mutate memory, run background jobs, continue/fork transcripts, or delegate to writer-capable agents. Read-only nested delegation may be available until max_subagent_depth is reached. Only its final answer is returned."
499 }
500
501 func (*ReadOnlyTaskTool) Schema() json.RawMessage {
502 return json.RawMessage(`{
503 "type":"object",
504 "properties":{
505 "prompt":{"type":"string","description":"What the read-only sub-agent should investigate. Be specific about the evidence or summary to return — the sub-agent does not see this conversation."},
506 "description":{"type":"string","description":"Short label for the read-only sub-task (3-7 words). Surfaced in the dispatch line so the user sees what's running."},
507 "tools":{"type":"array","items":{"type":"string"},"description":"Optional read-only tool whitelist. Writer, installer, memory mutation, background job, and delegation tools are never exposed."},
508 "max_steps":{"type":"integer","description":"Optional cap on tool-call rounds. Defaults to half the parent's cap (min 5).","minimum":1},
509 "model":{"type":"string","description":"Optional model override for the sub-agent (a configured provider/model name)."},
510 "effort":{"type":"string","description":"Optional reasoning effort for the sub-agent (e.g. high, max)."}
511 },
512 "required":["prompt"]
513 }`)
514 }
515
516 func (*ReadOnlyTaskTool) ReadOnly() bool { return true }
517
518 // PlanModeSafe reports true: read_only_task spawns a strictly read-only research
519 // sub-agent (no writers, installers, memory mutation, background jobs, or
520 // delegation), so it is safe to run while planning.
521 func (*ReadOnlyTaskTool) PlanModeSafe() bool { return true }
522
523 func (r *ReadOnlyTaskTool) ResolveProfile(args json.RawMessage) *event.Profile {
524 if r == nil || r.task == nil {
525 return nil
526 }
527 return r.task.ResolveProfile(args)
528 }
529
530 func (r *ReadOnlyTaskTool) Execute(ctx context.Context, args json.RawMessage) (string, error) {
531 if r == nil || r.task == nil {
532 return "", fmt.Errorf("read_only_task is not configured")
533 }
534 var p struct {
535 Prompt string `json:"prompt"`
536 Description string `json:"description"`
537 Tools []string `json:"tools"`
538 MaxSteps int `json:"max_steps"`
539 Model string `json:"model"`
540 Effort string `json:"effort"`
541 }
542 if err := json.Unmarshal(args, &p); err != nil {
543 return "", fmt.Errorf("invalid args: %w", err)
544 }
545 // Every entry point compiles to a spec and runs through RunProfileSpec, so a
546 // boundary added there cannot be missed by one caller. read_only_task keeps
547 // its own promise of no durable side effects through Ephemeral.
548 spec, err := r.task.buildTaskSpec(ctx, p.Prompt, p.Description, "", nil, p.Tools, p.MaxSteps, p.Model, p.Effort, "", "", false, true)
549 if err != nil {
550 return "", err
551 }
552 spec.Worker.SystemPrompt = DefaultReadOnlyTaskSystemPrompt
553 spec.Context.Ephemeral = true
554 return r.task.RunProfileSpec(ctx, spec)
555 }
556
557 func (t *TaskTool) effectiveProfile(model, effort string) (string, string) {
558 model = strings.TrimSpace(model)
559 effort = strings.TrimSpace(effort)
560 if model == "" {
561 model = strings.TrimSpace(t.subagentModel)
562 }
563 if effort == "" {
564 effort = strings.TrimSpace(t.subagentEffort)
565 }
566 return model, effort
567 }
568
569 func (t *TaskTool) Execute(ctx context.Context, args json.RawMessage) (string, error) {
570 var p struct {
571 Prompt string `json:"prompt"`
572 Description string `json:"description"`
573 Profile string `json:"profile"`
574 WritePaths []string `json:"write_paths"`
575 Tools []string `json:"tools"`
576 MaxSteps int `json:"max_steps"`
577 RunInBackground bool `json:"run_in_background"`
578 Model string `json:"model"`
579 Effort string `json:"effort"`
580 ContinueFrom string `json:"continue_from"`
581 ForkFrom string `json:"fork_from"`
582 }
583 if err := json.Unmarshal(args, &p); err != nil {
584 return "", fmt.Errorf("invalid args: %w", err)
585 }
586 if strings.TrimSpace(p.Prompt) == "" {
587 return "", fmt.Errorf("prompt is required")
588 }
589
590 spec, err := t.buildTaskSpec(ctx, p.Prompt, p.Description, p.Profile, p.WritePaths, p.Tools, p.MaxSteps, p.Model, p.Effort, p.ContinueFrom, p.ForkFrom, p.RunInBackground, false)
591 if err != nil {
592 return "", err
593 }
594 return t.RunProfileSpec(ctx, spec)
595 }
596
597 // buildTaskSpec resolves profile, tools, model/effort, and write claims for a
598 // single task/fleet item. forceReadOnly forces the read-only registry.
599 func (t *TaskTool) buildTaskSpec(ctx context.Context, prompt, description, profile string, writePaths, tools []string, maxSteps int, model, effort, continueFrom, forkFrom string, background, forceReadOnly bool) (ProfileExecSpec, error) {
600 spec := ProfileExecSpec{
601 Task: TaskSpec{Objective: prompt, Description: description},
602 Worker: WorkerSpec{Kind: "task", Name: "task", SystemPrompt: t.sysPrompt},
603 Grant: CapabilityGrant{CallTools: tools},
604 Context: ContextRequest{ContinueFrom: strings.TrimSpace(continueFrom), ForkFrom: strings.TrimSpace(forkFrom)},
605 Sched: SchedulerPolicy{MaxSteps: maxSteps, RunInBackground: background, Nested: SubagentDepth(ctx) > 0},
606 }
607 profile = strings.TrimSpace(profile)
608 readOnly := forceReadOnly
609 var profileTools []string
610 var profileModel, profileEffort string
611 if profile != "" {
612 def, err := ResolveProfileDefinition(t.profileLookup, profile)
613 if err != nil {
614 return ProfileExecSpec{}, err
615 }
616 spec.Worker.Profile = def.Name
617 spec.Worker.Name = def.Name
618 spec.Worker.Kind = "skill"
619 spec.Worker.SystemPrompt = def.Body
620 spec.Worker.UseProfilePrompt = true
621 profileTools = def.AllowedTools
622 profileModel, profileEffort = def.Model, def.Effort
623 if def.ReadOnly {
624 readOnly = true
625 }
626 }
627 spec.Grant.ReadOnly = readOnly
628 spec.Grant.ProfileTools = profileTools
629
630 configModel, configEffort := "", ""
631 if profile != "" {
632 if t.profileConfigModel != nil {
633 configModel = t.profileConfigModel(profile)
634 }
635 if t.profileConfigEffort != nil {
636 configEffort = t.profileConfigEffort(profile)
637 }
638 }
639 spec.Worker.Model, spec.Worker.Effort = ResolveModelEffort(
640 configModel, configEffort,
641 model, effort,
642 profileModel, profileEffort,
643 t.subagentModel, t.subagentEffort,
644 )
645
646 if !readOnly {
647 // Every writer carries a claim. Omitting write_paths conservatively claims
648 // the whole workspace, including foreground task calls, so they cannot
649 // bypass an already-running background/fleet writer claim. Direct legacy
650 // TaskTool constructions without a workspace/scheduler keep their old
651 // no-claim behavior; production boot always configures both.
652 requireClaim := t.scheduler != nil || strings.TrimSpace(t.workspaceRoot) != "" || background || len(writePaths) > 0
653 claims, err := t.resolveWriterClaims(writePaths, requireClaim)
654 if err != nil {
655 return ProfileExecSpec{}, err
656 }
657 spec.Grant.WritePaths = claims
658 if requireClaim && claims.Empty() {
659 return ProfileExecSpec{}, fmt.Errorf("writer claim resolved empty")
660 }
661 } else if len(writePaths) > 0 {
662 return ProfileExecSpec{}, fmt.Errorf("write_paths is not valid for read-only tasks")
663 }
664 return spec, nil
665 }
666
667 func (t *TaskTool) resolveWriterClaims(writePaths []string, requireClaim bool) (WritePathSet, error) {
668 if len(writePaths) > 0 {
669 return NormalizeWritePaths(t.workspaceRoot, writePaths)
670 }
671 if !requireClaim {
672 return WritePathSet{}, nil
673 }
674 return WholeWorkspaceWriteClaim(t.workspaceRoot)
675 }
676
677 // RunProfileSpec executes a unified profile/task specification. Shared by task,
678 // fleet items, and boot-wired skill runners so prompt, tools, claims, and
679 // scheduling cannot drift across entry points.
680 func (t *TaskTool) RunProfileSpec(ctx context.Context, spec ProfileExecSpec) (result string, err error) {
681 if t == nil {
682 return "", fmt.Errorf("task tool is not configured")
683 }
684 // Per-child progress tracker: converts the child's reasoning/text/notice/
685 // retrying into reserved ToolProgress previews and guarantees exactly one
686 // terminal status (completed/cancelled/failed). The background job owns
687 // finish after handoff; every other exit finishes here, including
688 // validation errors and panics.
689 trk := newSubagentProgressTracker(ctx, subSink(ctx))
690 backgroundHandoff := false
691 defer func() {
692 if backgroundHandoff {
693 return
694 }
695 if p := recover(); p != nil {
696 trk.finish(nil, fmt.Errorf("panic: %v", p))
697 panic(p)
698 }
699 trk.finish(ctx.Err(), err)
700 }()
701 if !spec.Sched.RunInBackground {
702 trk.running()
703 }
704 if strings.TrimSpace(spec.Task.Objective) == "" {
705 return "", fmt.Errorf("prompt is required")
706 }
707 if strings.TrimSpace(spec.Worker.SystemPrompt) == "" {
708 if spec.Worker.UseProfilePrompt {
709 return "", fmt.Errorf("profile system prompt is empty")
710 }
711 spec.Worker.SystemPrompt = t.sysPrompt
712 }
713
714 ctx, maxSteps := t.childMaxStepsForSpec(ctx, &spec)
715 childDepth, err := t.nextSubagentDepth(ctx)
716 if err != nil {
717 return "", err
718 }
719
720 toolNames, err := IntersectToolLists(t.parentReg, spec.Grant.ProfileTools, spec.Grant.CallTools)
721 if err != nil {
722 return "", err
723 }
724 subReg, childWriteRoots, err := t.buildSubagentRegistry(spec, toolNames, childDepth)
725 if err != nil {
726 return "", err
727 }
728
729 modelRef, effortRef := spec.Worker.Model, spec.Worker.Effort
730 usageModelRef := t.usageModelRef(modelRef, effortRef)
731 parentID, parentSink, _, _ := CallContext(ctx)
732 run, err := t.prepareTranscriptRunWithPrompt(ctx, subReg, modelRef, effortRef, spec.Context.parentSession(ctx), parentID, spec.Context.ContinueFrom, spec.Context.ForkFrom, spec.Worker.SystemPrompt, spec.Worker.Kind, spec.Worker.Name)
733 if err != nil {
734 return "", err
735 }
736 prov, pricing, ctxWin, err := t.resolveSubSessionRuntime(modelRef, effortRef)
737 if err != nil {
738 return t.failBeforeSubagentRelease(run, fmt.Errorf("sub-agent profile: %w", err))
739 }
740 lifecyclePhase := "child_created"
741 if strings.TrimSpace(spec.Context.ContinueFrom) != "" || strings.TrimSpace(spec.Context.ForkFrom) != "" {
742 lifecyclePhase = "child_resume"
743 }
744 emitSubagentLifecycle(parentSink, lifecyclePhase, parentID, spec.Worker.Name, usageModelRef, effortRef, run, nil)
745
746 isWriter := !spec.Grant.ReadOnly
747 acquireReq := AcquireRequest{
748 Writer: isWriter,
749 WritePaths: spec.Grant.WritePaths,
750 Nested: spec.Sched.Nested,
751 Label: firstNonEmpty(spec.Task.Description, spec.Worker.Name, "task"),
752 }
753 // Defensive fallback for callers that manually construct a background spec
754 // instead of going through buildTaskSpec.
755 if isWriter && spec.Grant.WritePaths.Empty() && spec.Sched.RunInBackground {
756 whole, werr := WholeWorkspaceWriteClaim(t.workspaceRoot)
757 if werr != nil {
758 return t.failBeforeSubagentRelease(run, werr)
759 }
760 acquireReq.WritePaths = whole
761 spec.Grant.WritePaths = whole
762 }
763
764 recoveryTaskID := subagentRecoveryTaskID(ctx, run.Ref)
765 backgroundWriter := (spec.Sched.RunInBackground || spec.Sched.BackgroundWriter) && !spec.Grant.ReadOnly
766 var mutationObserver *checkpoint.MutationObserver
767 if t.mutationObserver != nil {
768 turn := t.mutationObserver.OwnershipTurn()
769 mutationObserver = t.mutationObserver.CloneForSubagent(recoveryTaskID, turn, backgroundWriter)
770 }
771 runSession := func(runCtx context.Context, sink event.Sink, writerAlreadyRegistered bool) (string, error) {
772 if mutationObserver != nil && backgroundWriter && !writerAlreadyRegistered {
773 turn := mutationObserver.OwnershipTurn()
774 if err := mutationObserver.RegisterWriter(recoveryTaskID, "background_subagent", turn); err != nil {
775 return "", err
776 }
777 defer mutationObserver.UnregisterWriter(recoveryTaskID)
778 }
779 if spec.Grant.ReadOnly {
780 return t.runReadOnlySubSession(runCtx, composeChildTaskPrompt(spec), subReg, sink, maxSteps, prov, pricing, ctxWin, run.Session, childDepth, recoveryTaskID, usageModelRef, mutationObserver)
781 }
782 return t.runSubSession(WithSubagentWriteClaim(runCtx, spec.Grant.WritePaths), composeChildTaskPrompt(spec), subReg, sink, maxSteps, prov, pricing, ctxWin, run.Session, childDepth, recoveryTaskID, usageModelRef, mutationObserver, childWriteRoots)
783 }
784
785 if spec.Sched.RunInBackground {
786 result, runErr, handedOff := t.runBackgroundProfileSpec(ctx, spec, run, trk, parentID, parentSink, usageModelRef, effortRef, runSession, acquireReq, mutationObserver, backgroundWriter, recoveryTaskID)
787 backgroundHandoff = handedOff
788 return result, runErr
789 }
790
791 // Foreground: acquire a slot (queue if needed), then run synchronously.
792 releaseSlot, claimID, err := t.acquireSlot(ctx, acquireReq)
793 if err != nil {
794 return t.failedSubagentResult(run, err)
795 }
796 defer releaseSlot()
797 defer run.Release()
798 ctx = WithSubagentClaimID(ctx, claimID)
799 emitSubagentLifecycle(parentSink, "child_running", parentID, spec.Worker.Name, usageModelRef, effortRef, run, nil)
800 answer, err := runSession(ctx, trk.wrap(), false)
801 if err != nil {
802 result, runErr := t.resolveAmbiguousSubagentFailure(ctx, run, spec.Task.Objective, usageModelRef, parentSink, err)
803 phase, outcome := terminalSubagentLifecycle(runErr)
804 emitSubagentLifecycle(parentSink, phase, parentID, spec.Worker.Name, usageModelRef, effortRef, run, outcome)
805 return result, runErr
806 }
807 if t.transcripts != nil && run.Ref != "" {
808 if err := t.transcripts.SaveCompleted(run); err != nil {
809 result, runErr := t.failedSubagentResult(run, err)
810 phase, outcome := terminalSubagentLifecycle(runErr)
811 emitSubagentLifecycle(parentSink, phase, parentID, spec.Worker.Name, usageModelRef, effortRef, run, outcome)
812 return result, runErr
813 }
814 emitSubagentLifecycle(parentSink, "child_completed", parentID, spec.Worker.Name, usageModelRef, effortRef, run, &SubagentOutcome{Status: SubagentOutcomeCompleted, FinalAnswer: answer})
815 return FormatSubagentRunResult(answer, run, false), nil
816 }
817 return GuardSubagentHostDecisionText(answer), nil
818 }
819
820 func (t *TaskTool) runBackgroundProfileSpec(ctx context.Context, spec ProfileExecSpec, run *SubagentRun, trk *subagentProgressTracker, parentID string, parentSink event.Sink, usageModelRef, effortRef string,
821 runSession func(context.Context, event.Sink, bool) (string, error), acquireReq AcquireRequest, mutationObserver *checkpoint.MutationObserver, backgroundWriter bool, recoveryTaskID string,
822 ) (string, error, bool) {
823 jm, ok := jobs.FromContext(ctx)
824 if !ok {
825 result, err := t.failBeforeSubagentRelease(run, fmt.Errorf("background execution is not available in this context"))
826 return result, err, false
827 }
828 var releaseStart func()
829 if t.scheduler == nil {
830 var running int
831 var okReserve bool
832 releaseStart, running, okReserve = jm.ReserveStartForSession(jobs.SessionFromContext(ctx), "task", maxConcurrentBackgroundTasks)
833 if !okReserve {
834 result, err := t.failBeforeSubagentRelease(run, fmt.Errorf("%d background tasks are already running for this session (limit %d); collect their results with job_output — or run this sub-task in the foreground — before starting more", running, maxConcurrentBackgroundTasks))
835 return result, err, false
836 }
837 defer releaseStart()
838 } else {
839 releaseStart = func() {}
840 }
841 label := firstNonEmpty(spec.Task.Description, spec.Worker.Name, "task")
842 if t.transcripts != nil && run != nil && run.Ref != "" {
843 if err := t.transcripts.MarkRunning(run); err != nil {
844 releaseStart()
845 result, saveErr := t.failBeforeSubagentRelease(run, err)
846 return result, saveErr, false
847 }
848 }
849 writerRegistered := false
850 if mutationObserver != nil && backgroundWriter {
851 turn := mutationObserver.OwnershipTurn()
852 if err := mutationObserver.RegisterWriter(recoveryTaskID, "background_subagent", turn); err != nil {
853 releaseStart()
854 result, saveErr := t.failBeforeSubagentRelease(run, err)
855 return result, saveErr, false
856 }
857 writerRegistered = true
858 }
859 parentSession := ParentSession(ctx)
860 backgroundEvidence := evidence.NewLedger()
861 slotReq := acquireReq
862 trk.queued()
863 job, startErr := jm.TryStartForSession(jobs.SessionFromContext(ctx), "task", label, func(jobCtx context.Context, _ io.Writer) (result string, err error) {
864 if writerRegistered {
865 defer mutationObserver.UnregisterWriter(recoveryTaskID)
866 }
867 jobCtx = WithParentSession(jobCtx, parentSession)
868 jobCtx = withInheritedHostConstraints(ctx, jobCtx)
869 jobCtx = evidence.WithLedger(jobCtx, backgroundEvidence)
870 defer run.Release()
871 defer publishBackgroundEvidence(jobCtx, backgroundEvidence, t.workspaceRoot)
872 defer func() {
873 if r := recover(); r != nil {
874 panicErr := fmt.Errorf("internal error: panic: %v\n%s", r, debug.Stack())
875 result, err = t.failedSubagentResult(run, panicErr)
876 }
877 phase, outcome := terminalSubagentLifecycle(err)
878 emitSubagentLifecycle(parentSink, phase, parentID, spec.Worker.Name, usageModelRef, effortRef, run, outcome)
879 trk.finish(jobCtx.Err(), err)
880 }()
881 releaseSlot, claimID, slotErr := t.acquireSlot(jobCtx, slotReq)
882 if slotErr != nil {
883 return t.failedSubagentResult(run, slotErr)
884 }
885 defer releaseSlot()
886 jobCtx = WithSubagentClaimID(jobCtx, claimID)
887 trk.running()
888 emitSubagentLifecycle(parentSink, "child_running", parentID, spec.Worker.Name, usageModelRef, effortRef, run, nil)
889 answer, err := runSession(jobCtx, trk.wrap(), writerRegistered)
890 if err != nil {
891 return t.resolveAmbiguousSubagentFailure(jobCtx, run, spec.Task.Objective, usageModelRef, parentSink, err)
892 }
893 if err := t.transcripts.SaveCompleted(run); err != nil {
894 return t.failedSubagentResult(run, err)
895 }
896 return FormatSubagentRunResult(answer, run, false), nil
897 })
898 releaseStart()
899 if startErr != nil {
900 if writerRegistered {
901 mutationObserver.UnregisterWriter(recoveryTaskID)
902 }
903 result, saveErr := t.failBeforeSubagentRelease(run, startErr)
904 return result, saveErr, false
905 }
906 queuedNote := ""
907 if t.scheduler != nil {
908 queuedNote = " It may wait in the session queue until a concurrency/write slot is free."
909 }
910 if run != nil && run.Ref != "" {
911 return fmt.Sprintf("Started background task %q (%s).%s\n%s\nIt runs across turns; collect its final answer with job_output, and you'll be notified when it finishes.", job.ID, label, queuedNote, FormatSubagentReference(run)), nil, true
912 }
913 return fmt.Sprintf("Started background task %q (%s).%s It runs across turns; collect its final answer with job_output, and you'll be notified when it finishes.", job.ID, label, queuedNote), nil, true
914 }
915
916 func (t *TaskTool) acquireSlot(ctx context.Context, req AcquireRequest) (func(), int64, error) {
917 noop := func() {}
918 if t.scheduler == nil {
919 return noop, 0, nil
920 }
921 return t.scheduler.AcquireWithID(ctx, req)
922 }
923
924 func (t *TaskTool) bashCanEnforceWriteRoots() bool {
925 if t != nil && t.bashSandboxEnforced != nil {
926 return t.bashSandboxEnforced()
927 }
928 return false
929 }
930
931 func (t *TaskTool) prepareTranscriptRunWithPrompt(ctx context.Context, subReg *tool.Registry, modelRef, effortRef, parentSession, parentID, continueFrom, legacyForkFrom, systemPrompt, kind, name string) (*SubagentRun, error) {
932 continueFrom = strings.TrimSpace(continueFrom)
933 legacyForkFrom = strings.TrimSpace(legacyForkFrom)
934 parentSession = strings.TrimSpace(parentSession)
935 if continueFrom != "" && legacyForkFrom != "" {
936 return nil, fmt.Errorf("continue_from and fork_from are mutually exclusive; pass only continue_from")
937 }
938 if t.transcripts == nil {
939 return nil, fmt.Errorf("subagent transcript store is required")
940 }
941 if systemPrompt == "" {
942 systemPrompt = t.sysPrompt
943 }
944 if kind == "" {
945 kind = "task"
946 }
947 if name == "" {
948 name = "task"
949 }
950 if parentSession == "" {
951 if continueFrom != "" || legacyForkFrom != "" {
952 return nil, fmt.Errorf("subagent continuation requires a persisted session; none is active in this run")
953 }
954 return EphemeralSubagentRun(systemPrompt), nil
955 }
956 identityModel, identityEffort := t.effectiveIdentity(modelRef, effortRef)
957 spec := SubagentSpec{
958 Kind: kind,
959 Name: name,
960 WorkspaceRoot: t.workspaceRoot,
961 ParentSession: parentSession,
962 ParentToolCallID: parentID,
963 SystemPrompt: systemPrompt,
964 Registry: subReg,
965 ToolContext: childToolIdentityContext(ctx),
966 Model: identityModel,
967 Effort: identityEffort,
968 ResumedFrom: firstNonEmpty(continueFrom, legacyForkFrom),
969 }
970 if continueFrom != "" {
971 return t.transcripts.PrepareContinue(continueFrom, spec)
972 }
973 if legacyForkFrom != "" {
974 return t.transcripts.PrepareLegacyForkFrom(legacyForkFrom, spec)
975 }
976 return t.transcripts.PrepareFresh(spec)
977 }
978
979 func childToolIdentityContext(ctx context.Context) context.Context {
980 ctx = tool.WithoutGoalLifecycle(ctx)
981 ctx = memory.WithoutQueue(ctx)
982 ctx = jobs.WithoutManager(ctx)
983 return planmode.WithActive(ctx, PlanModeFromContext(ctx))
984 }
985
986 func (t *TaskTool) effectiveIdentity(modelRef, effort string) (string, string) {
987 if t.identityProfile != nil {
988 model, eff := t.identityProfile(modelRef, effort)
989 return strings.TrimSpace(model), strings.TrimSpace(eff)
990 }
991 return t.effectiveModelIdentity(modelRef), t.effectiveEffortIdentity(effort)
992 }
993
994 // usageModelRef returns the canonical provider/model identity of the runtime
995 // selected for a child. The resolver expands aliases and supplies the parent
996 // model when no child override is configured.
997 func (t *TaskTool) usageModelRef(modelRef, effort string) string {
998 model, _ := t.effectiveIdentity(modelRef, effort)
999 if model != "" {
1000 return model
1001 }
1002 return firstNonEmpty(modelRef, t.baseModel, t.subagentModel)
1003 }
1004
1005 func (t *TaskTool) effectiveModelIdentity(modelRef string) string {
1006 if strings.TrimSpace(modelRef) != "" {
1007 return strings.TrimSpace(modelRef)
1008 }
1009 return strings.TrimSpace(t.baseModel)
1010 }
1011
1012 func (t *TaskTool) effectiveEffortIdentity(effort string) string {
1013 if strings.TrimSpace(effort) != "" {
1014 return strings.TrimSpace(effort)
1015 }
1016 return strings.TrimSpace(t.baseEffort)
1017 }
1018
1019 // buildSubReg returns the sub-agent's tool set: the named whitelist (minus
1020 // unavailable sub-agent tools), or every parent tool except those tools.
1021 func (t *TaskTool) buildSubReg(names []string, childDepth int) *tool.Registry {
1022 return SubagentToolRegistryForDepthWithRuntime(t.parentReg, names, childDepth, t.maxDepth(), t.capabilityRuntime)
1023 }
1024
1025 func (t *TaskTool) maxDepth() int {
1026 if t == nil {
1027 return DefaultMaxSubagentDepth
1028 }
1029 if t.maxSubagentDepth == 0 {
1030 return DefaultMaxSubagentDepth
1031 }
1032 return NormalizeMaxSubagentDepth(t.maxSubagentDepth)
1033 }
1034
1035 func (t *TaskTool) nextSubagentDepth(ctx context.Context) (int, error) {
1036 current := SubagentDepth(ctx)
1037 next := current + 1
1038 maxDepth := t.maxDepth()
1039 if next > maxDepth {
1040 return 0, fmt.Errorf("subagent delegation depth limit reached (max_subagent_depth=%d)", maxDepth)
1041 }
1042 return next, nil
1043 }
1044
1045 // FilterRegistry builds a sub-registry from parent: the named whitelist (empty =
1046 // every parent tool), minus any excluded names. Used to scope what a spawned
1047 // sub-agent — a `task` sub-agent or a subagent skill — may call, e.g. excluding
1048 // `task` to bar recursive nesting, or restricting to a skill's allowed-tools.
1049 // Direct MCP tools may be copied here; callers that need a stable MCP surface
1050 // should strip them and attach use_capability via attachSubagentCapabilityProxy.
1051 func FilterRegistry(parent *tool.Registry, names []string, exclude ...string) *tool.Registry {
1052 sub := tool.NewRegistry()
1053 if parent == nil {
1054 return sub
1055 }
1056 ex := make(map[string]bool, len(exclude))
1057 for _, e := range exclude {
1058 ex[e] = true
1059 }
1060 customAllowlist := len(names) > 0
1061 src := normalizeSubagentShellNames(parent, names)
1062 if !customAllowlist {
1063 src = parent.Names()
1064 } else {
1065 src = expandToolPatterns(parent, src)
1066 }
1067 for _, name := range src {
1068 if ex[name] || retiredTool(name) {
1069 continue
1070 }
1071 // MCP never enters through the generic filter when named as capability
1072 // ids; model-visible mcp__* may still be listed for conversion later.
1073 if strings.HasPrefix(name, "mcp-tool:") || strings.HasPrefix(name, "mcp-server:") {
1074 continue
1075 }
1076 tl, ok := parent.Get(name)
1077 if !ok {
1078 continue
1079 }
1080 sub.Add(tl)
1081 }
1082 return sub
1083 }
1084
1085 // stripDirectMCPTools removes provider-visible mcp__* tools so sub-agents use
1086 // only the stable use_capability proxy for MCP.
1087 func stripDirectMCPTools(reg *tool.Registry) {
1088 if reg == nil {
1089 return
1090 }
1091 for _, name := range append([]string(nil), reg.Names()...) {
1092 if strings.HasPrefix(name, tool.MCPNamePrefix) {
1093 reg.RemovePrefix(name)
1094 }
1095 }
1096 }
1097
1098 // restrictedCapabilityProxy preserves a subagent allowed-tools boundary when
1099 // MCP is available only through use_capability. The pseudo mcp-tool: and
1100 // mcp-server: entries never become provider tools; they select one proxy schema
1101 // whose resolver rejects every capability outside the exact allowlist.
1102 //
1103 // Provider-visible name/description/schema stay identical to the unrestricted
1104 // proxy so allowlist expansion never changes the child cache prefix. Allowlist
1105 // enforcement is host-local (check + filtered list results).
1106 type restrictedCapabilityProxy struct {
1107 tool.Tool
1108 resolver tool.CallResolver
1109 allowed map[string]bool
1110 // servers is the set of MCP server names implied by allowed IDs; list
1111 // results are filtered to this set so profile isolation covers discovery.
1112 servers map[string]bool
1113 }
1114
1115 func (t *restrictedCapabilityProxy) ClassifyCall(args json.RawMessage) tool.CallClass {
1116 if t == nil || t.check(args) != nil {
1117 return tool.CallClass{}
1118 }
1119 classifier, ok := t.Tool.(tool.BatchClassifier)
1120 if !ok {
1121 return tool.CallClass{}
1122 }
1123 return classifier.ClassifyCall(args)
1124 }
1125
1126 // Description is fixed: never embed dynamic capability IDs (they change with
1127 // MCP install/tool-list and would break the stable provider tool prefix).
1128 func (t *restrictedCapabilityProxy) Description() string {
1129 return t.Tool.Description()
1130 }
1131
1132 func (t *restrictedCapabilityProxy) check(args json.RawMessage) error {
1133 var p struct {
1134 Action string `json:"action"`
1135 CapabilityID string `json:"capability_id"`
1136 }
1137 if err := json.Unmarshal(args, &p); err != nil {
1138 return fmt.Errorf("invalid args: %w", err)
1139 }
1140 action := strings.ToLower(strings.TrimSpace(p.Action))
1141 if action == "list" || action == "search" {
1142 return nil
1143 }
1144 id := strings.TrimSpace(p.CapabilityID)
1145 if id == "" {
1146 return fmt.Errorf("capability_id is required")
1147 }
1148 if id == sessionToolResultCapabilityID {
1149 return nil
1150 }
1151 if !t.allowed[id] {
1152 return fmt.Errorf("capability %q is outside this subagent's allowed-tools", id)
1153 }
1154 return nil
1155 }
1156
1157 func (t *restrictedCapabilityProxy) ResolveCall(ctx context.Context, args json.RawMessage) (tool.ResolvedCall, error) {
1158 if err := t.check(args); err != nil {
1159 return tool.ResolvedCall{}, err
1160 }
1161 rc, err := t.resolver.ResolveCall(ctx, args)
1162 if err != nil {
1163 return rc, err
1164 }
1165 var p struct {
1166 Action string `json:"action"`
1167 }
1168 _ = json.Unmarshal(args, &p)
1169 action := strings.ToLower(strings.TrimSpace(p.Action))
1170 if rc.SkipExecute {
1171 switch action {
1172 case "list":
1173 rc.Result = filterCapabilityListResult(rc.Result, t.servers)
1174 case "search":
1175 rc.Result = filterCapabilitySearchResult(rc.Result, t.allowed)
1176 }
1177 }
1178 return rc, nil
1179 }
1180
1181 func (t *restrictedCapabilityProxy) Execute(ctx context.Context, args json.RawMessage) (string, error) {
1182 if err := t.check(args); err != nil {
1183 return "", err
1184 }
1185 out, err := t.Tool.Execute(ctx, args)
1186 if err != nil {
1187 return out, err
1188 }
1189 var p struct {
1190 Action string `json:"action"`
1191 }
1192 _ = json.Unmarshal(args, &p)
1193 switch strings.ToLower(strings.TrimSpace(p.Action)) {
1194 case "list":
1195 return filterCapabilityListResult(out, t.servers), nil
1196 case "search":
1197 return filterCapabilitySearchResult(out, t.allowed), nil
1198 }
1199 return out, nil
1200 }
1201
1202 // validMCPServerCapabilityID accepts mcp-server:<non-empty-name> only.
1203 func validMCPServerCapabilityID(id string) (server string, ok bool) {
1204 return tool.ParseMCPServerReference(id)
1205 }
1206
1207 // validMCPToolCapabilityID accepts mcp-tool:<server>/<tool> with both parts non-empty.
1208 func validMCPToolCapabilityID(id string) (server, raw string, ok bool) {
1209 return tool.ParseMCPToolReference(id)
1210 }
1211
1212 func serversFromCapabilityAllowlist(allowed map[string]bool) map[string]bool {
1213 servers := map[string]bool{}
1214 for id := range allowed {
1215 id = strings.TrimSpace(id)
1216 if server, ok := validMCPServerCapabilityID(id); ok {
1217 servers[server] = true
1218 continue
1219 }
1220 if server, _, ok := validMCPToolCapabilityID(id); ok {
1221 servers[server] = true
1222 }
1223 }
1224 return servers
1225 }
1226
1227 // attachSubagentCapabilityProxy installs a per-agent use_capability frontend.
1228 // Any parent-copied proxy is replaced so children never share Executor ledger
1229 // state. No allowlist → full proxy. Explicit allowlist with MCP names →
1230 // restricted proxy. Explicit "use_capability" → full proxy. Explicit allowlist
1231 // without MCP entries → no proxy.
1232 func attachSubagentCapabilityProxy(parent, sub *tool.Registry, names []string, runtime *MCPCapabilityRuntime) {
1233 if sub == nil {
1234 return
1235 }
1236 // Drop any provider-copied use_capability so we always install an isolated
1237 // frontend (shared Host/runtime, independent ledger/audit).
1238 if _, ok := sub.Get("use_capability"); ok {
1239 sub.RemovePrefix("use_capability")
1240 }
1241 frontend := newSubagentCapabilityFrontend(parent, runtime)
1242 if frontend == nil {
1243 return
1244 }
1245 if len(names) == 0 || allowlistRequestsUnrestrictedProxy(names) {
1246 sub.Add(frontend)
1247 return
1248 }
1249 allowed := mcpCapabilityAllowlist(parent, names)
1250 if len(allowed) == 0 {
1251 // Custom allowlist with no valid MCP entries: do not expose the proxy.
1252 return
1253 }
1254 servers := serversFromCapabilityAllowlist(allowed)
1255 if len(servers) == 0 {
1256 // Incomplete capability IDs produced an empty server set: fail closed
1257 // rather than installing a restricted proxy that would list everything.
1258 return
1259 }
1260 resolver, ok := frontend.(tool.CallResolver)
1261 if !ok {
1262 return
1263 }
1264 sub.Add(&restrictedCapabilityProxy{
1265 Tool: frontend,
1266 resolver: resolver,
1267 allowed: allowed,
1268 servers: servers,
1269 })
1270 }
1271
1272 func newSubagentCapabilityFrontend(parent *tool.Registry, runtime *MCPCapabilityRuntime) tool.Tool {
1273 if runtime != nil {
1274 return runtime.NewFrontend(nil, nil)
1275 }
1276 if parent == nil {
1277 return nil
1278 }
1279 inner, ok := parent.Get("use_capability")
1280 if !ok {
1281 return nil
1282 }
1283 return cloneCapabilityFrontend(inner)
1284 }
1285
1286 // mcpCapabilityAllowlist converts profile/call tool names into capability IDs
1287 // for the restricted use_capability proxy. Accepts complete mcp-tool:<s>/<t>,
1288 // mcp-server:<s>, model-visible mcp__* names, and wildcards expanded against
1289 // the parent. Incomplete prefixes such as "mcp-server:" or "mcp-tool:foo" are
1290 // rejected so they cannot install a restricted proxy with an empty server set.
1291 func mcpCapabilityAllowlist(parent *tool.Registry, names []string) map[string]bool {
1292 if len(names) == 0 {
1293 return nil
1294 }
1295 expanded := names
1296 if parent != nil {
1297 expanded = expandToolPatterns(parent, names)
1298 }
1299 allowed := map[string]bool{}
1300 for _, name := range expanded {
1301 name = strings.TrimSpace(name)
1302 switch {
1303 case name == "use_capability":
1304 // Explicit proxy grant is handled as a full frontend by the caller
1305 // when this is the only MCP-related entry; leave empty here so a
1306 // bare use_capability allowlist entry still installs unrestricted.
1307 continue
1308 case strings.HasPrefix(name, "mcp-server:"):
1309 if server, ok := validMCPServerCapabilityID(name); ok {
1310 allowed["mcp-server:"+server] = true
1311 }
1312 case strings.HasPrefix(name, "mcp-tool:"):
1313 if server, raw, ok := validMCPToolCapabilityID(name); ok {
1314 allowed["mcp-tool:"+server+"/"+raw] = true
1315 }
1316 default:
1317 if parent != nil {
1318 if tl, ok := parent.Get(name); ok {
1319 if m, ok := tl.(tool.MCPMetadata); ok {
1320 server := strings.TrimSpace(m.MCPServerName())
1321 raw := strings.TrimSpace(m.MCPRawToolName())
1322 if server != "" && raw != "" {
1323 allowed["mcp-tool:"+server+"/"+raw] = true
1324 continue
1325 }
1326 }
1327 }
1328 }
1329 if server, raw, ok := tool.SplitMCPName(name); ok {
1330 allowed["mcp-tool:"+server+"/"+raw] = true
1331 }
1332 }
1333 }
1334 return allowed
1335 }
1336
1337 func allowlistRequestsUnrestrictedProxy(names []string) bool {
1338 for _, name := range names {
1339 if strings.TrimSpace(name) == "use_capability" {
1340 return true
1341 }
1342 }
1343 return false
1344 }
1345
1346 // ReadOnlySubagentToolRegistry returns the tool set exposed to read-only
1347 // sub-agents: read-only research tools plus a bash wrapper that enforces the
1348 // permission-layer read-only command policy at execution time. Workflow/meta tools are
1349 // excluded even when their Tool.ReadOnly contract is true.
1350 func ReadOnlySubagentToolRegistry(parent *tool.Registry, names []string) *tool.Registry {
1351 return ReadOnlySubagentToolRegistryForDepth(parent, names, 1, 1)
1352 }
1353
1354 // ReadOnlySubagentToolRegistryForDepth returns the tool set exposed to read-only
1355 // subagents. It permits only read-only delegation tools while another depth
1356 // layer is available. Direct mcp__* schemas are never exposed; MCP goes only
1357 // through use_capability. Dynamic execution still requires authorized server +
1358 // readOnlyHint + non-destructive (enforced by ReadOnlyExecution), so strict
1359 // agents share the stable proxy schema and connection reuse without permission
1360 // relaxation.
1361 //
1362 // Custom profile/call allowlists remain authoritative and convert MCP names
1363 // into a capability-id allowlist on a restricted proxy.
1364 func ReadOnlySubagentToolRegistryForDepth(parent *tool.Registry, names []string, childDepth, maxDepth int) *tool.Registry {
1365 return ReadOnlySubagentToolRegistryForDepthWithRuntime(parent, names, childDepth, maxDepth, nil)
1366 }
1367
1368 // ReadOnlySubagentToolRegistryForDepthWithRuntime is the read-only registry
1369 // builder with an optional session MCP runtime for proxy injection.
1370 func ReadOnlySubagentToolRegistryForDepthWithRuntime(parent *tool.Registry, names []string, childDepth, maxDepth int, runtime *MCPCapabilityRuntime) *tool.Registry {
1371 exclude := append([]string(nil), subagentAlwaysHiddenTools...)
1372 if childDepth >= NormalizeMaxSubagentDepth(maxDepth) {
1373 exclude = append(exclude, subagentRecursiveTools...)
1374 } else {
1375 exclude = append(exclude, "task", "run_skill", "explore", "research", "review", "security_review")
1376 }
1377 exclude = append(exclude, subagentJobTools...)
1378 exclude = append(exclude, plannerNonResearchTools...)
1379 exclude = append(exclude, readOnlySubagentWorkflowTools...)
1380 ex := make(map[string]bool, len(exclude))
1381 for _, e := range exclude {
1382 ex[e] = true
1383 }
1384 sub := tool.NewRegistry()
1385 if parent == nil {
1386 return sub
1387 }
1388 src := normalizeSubagentShellNames(parent, names)
1389 if len(src) == 0 {
1390 src = parent.Names()
1391 } else {
1392 src = expandToolPatterns(parent, src)
1393 }
1394 for _, name := range src {
1395 if ex[name] || retiredTool(name) {
1396 continue
1397 }
1398 if strings.HasPrefix(name, "mcp-tool:") || strings.HasPrefix(name, "mcp-server:") {
1399 continue
1400 }
1401 tl, ok := parent.Get(name)
1402 if !ok {
1403 continue
1404 }
1405 _, parentHasPwsh := parent.Get("pwsh")
1406 if name == "bash" && parentHasPwsh {
1407 continue
1408 }
1409 if name == "bash" || name == "pwsh" {
1410 sub.Add(readOnlyBash{inner: tl})
1411 continue
1412 }
1413 // Direct MCP never enters the strict registry — use_capability only.
1414 if isInstalledMCPTool(tl) || strings.HasPrefix(name, tool.MCPNamePrefix) {
1415 continue
1416 }
1417 if !tl.ReadOnly() {
1418 continue
1419 }
1420 sub.Add(tl)
1421 }
1422 attachSubagentCapabilityProxy(parent, sub, names, runtime)
1423 return sub
1424 }
1425
1426 // expandToolPatterns resolves explicit wildcard allowlist entries from imported
1427 // agent profiles against the current registry. Expansion is deterministic and
1428 // session-local, so optional MCP tools only enter a child after connection.
1429 func expandToolPatterns(parent *tool.Registry, names []string) []string {
1430 if parent == nil {
1431 return nil
1432 }
1433 available := parent.Names()
1434 seen := map[string]bool{}
1435 out := make([]string, 0, len(names))
1436 for _, name := range names {
1437 if !strings.ContainsAny(name, "*?[") {
1438 if !seen[name] {
1439 seen[name] = true
1440 out = append(out, name)
1441 }
1442 continue
1443 }
1444 for _, candidate := range available {
1445 matched, err := filepath.Match(name, candidate)
1446 if err == nil && matched && !seen[candidate] {
1447 seen[candidate] = true
1448 out = append(out, candidate)
1449 }
1450 }
1451 }
1452 return out
1453 }
1454
1455 // FilterReadOnlyRegistry builds a sub-registry containing only tools whose
1456 // ReadOnly contract is true, minus explicit exclusions. MCP tools must
1457 // additionally come from an authorized server and must not carry
1458 // destructiveHint.
1459 func FilterReadOnlyRegistry(parent *tool.Registry, exclude ...string) *tool.Registry {
1460 ex := make(map[string]bool, len(exclude))
1461 for _, e := range exclude {
1462 ex[e] = true
1463 }
1464 sub := tool.NewRegistry()
1465 if parent == nil {
1466 return sub
1467 }
1468 for _, name := range parent.Names() {
1469 if ex[name] || retiredTool(name) {
1470 continue
1471 }
1472 tl, ok := parent.Get(name)
1473 if !ok || !tl.ReadOnly() {
1474 continue
1475 }
1476 if isInstalledMCPTool(tl) && (!mcpServerAuthorized(tl) || mcpDestructiveHint(tl)) {
1477 continue
1478 }
1479 sub.Add(tl)
1480 }
1481 return sub
1482 }
1483
1484 func (t *TaskTool) resolveSubSessionRuntime(modelRef, effort string) (provider.Provider, *provider.Pricing, int, error) {
1485 prov, pricing, ctxWin := t.prov, t.pricing, t.contextWindow
1486 if t.resolveProvider != nil && (modelRef != "" || effort != "") {
1487 p, pr, cw, err := t.resolveProvider(modelRef, effort)
1488 if err != nil {
1489 return nil, nil, 0, err
1490 }
1491 prov, pricing, ctxWin = p, pr, cw
1492 }
1493 return prov, pricing, ctxWin, nil
1494 }
1495
1496 func (t *TaskTool) runSubSession(ctx context.Context, prompt string, subReg *tool.Registry, sink event.Sink, maxSteps int, prov provider.Provider, pricing *provider.Pricing, ctxWin int, sess *Session, childDepth int, recoveryTaskID, modelRef string, mutationObserver *checkpoint.MutationObserver, writeRoots *sandbox.WritableRootSet) (string, error) {
1497 opts := t.subagentOptions(ctx, maxSteps, pricing, ctxWin, childDepth, recoveryTaskID, mutationObserver)
1498 if writeRoots != nil {
1499 opts.WriteRoots = writeRoots
1500 }
1501 opts.ModelRef = modelRef
1502 // Capture the pristine task before host framing is prepended: delivery
1503 // intent classification must judge the task, not the wrapper.
1504 opts.ClassifierTaskText = prompt
1505 prompt = t.withWorkspaceContext(prompt) + "\n\n" + completeSubtaskContract
1506 // The child provider owns the final vision decision. Text-only providers
1507 // retain the attachment metadata but omit image parts during serialization.
1508 ctx = withSubagentTurnImages(ctx)
1509 return RunSubAgentWithSession(ctx, prov, subReg, sess, prompt, opts, sink)
1510 }
1511
1512 func (t *TaskTool) runReadOnlySubSession(ctx context.Context, prompt string, subReg *tool.Registry, sink event.Sink, maxSteps int, prov provider.Provider, pricing *provider.Pricing, ctxWin int, sess *Session, childDepth int, recoveryTaskID, modelRef string, mutationObserver *checkpoint.MutationObserver) (string, error) {
1513 opts := t.subagentOptions(ctx, maxSteps, pricing, ctxWin, childDepth, recoveryTaskID, mutationObserver)
1514 opts.ModelRef = modelRef
1515 // Capture the pristine task before host framing is prepended: delivery
1516 // intent classification must judge the task, not the wrapper.
1517 opts.ClassifierTaskText = prompt
1518 prompt = t.withWorkspaceContext(prompt)
1519 ctx = withSubagentTurnImages(ctx)
1520 return RunReadOnlySubAgentWithSession(ctx, prov, subReg, sess, prompt, opts, sink)
1521 }
1522
1523 func subagentRecoveryTaskID(ctx context.Context, ref string) string {
1524 if ref = strings.TrimSpace(ref); ref != "" {
1525 return "subagent:" + ref
1526 }
1527 if callID, _, _, ok := CallContext(ctx); ok && strings.TrimSpace(callID) != "" {
1528 return "subagent:" + strings.TrimSpace(callID)
1529 }
1530 return "subagent"
1531 }
1532
1533 func (t *TaskTool) WithWriteRoots(set *sandbox.WritableRootSet) *TaskTool {
1534 if t == nil {
1535 return nil
1536 }
1537 t.writeRoots = set
1538 return t
1539 }
1540
1541 func (t *TaskTool) WithRecoveryGate(g RecoveryGate) *TaskTool {
1542 // Retired source-compatible option. Sub-agents inherit execution facts but
1543 // never an Auto Guard admission policy.
1544 return t
1545 }
1546
1547 // WithMutationObserver shares the host mutation observer with spawned sub-agents.
1548 // Foreground children inherit the parent ownership turn; background children
1549 // keep the turn that spawned them (set via OwnershipTurn at Begin).
1550 func (t *TaskTool) WithMutationObserver(obs *checkpoint.MutationObserver) *TaskTool {
1551 if t == nil {
1552 return nil
1553 }
1554 t.mutationObserver = obs
1555 return t
1556 }
1557
1558 func (t *TaskTool) withWorkspaceContext(prompt string) string {
1559 if t == nil {
1560 return prompt
1561 }
1562 ctx := subagentWorkspaceContext(t.workspaceRoot)
1563 if ctx == "" {
1564 return prompt
1565 }
1566 return ctx + "\n\n" + prompt
1567 }
1568
1569 func subagentWorkspaceContext(root string) string {
1570 root = strings.TrimSpace(root)
1571 if root == "" {
1572 return ""
1573 }
1574 // Wording note: avoid incidental action verbs ("resolve", "fix", …) in this
1575 // host framing — it is prepended to every sub-agent prompt and must never
1576 // read as task intent (see classifierTaskText, which also strips it).
1577 return `<workspace-context event="SubagentWorkspace">
1578 Current workspace: ` + strconv.Quote(root) + `
1579 File tools interpret relative paths against this workspace. For project inspection, prefer "." or relative paths unless the user explicitly named another absolute path.
1580 </workspace-context>`
1581 }
1582
1583 func FormatSubagentReference(run *SubagentRun) string {
1584 if run == nil || run.Ref == "" {
1585 return ""
1586 }
1587 var b strings.Builder
1588 fmt.Fprintf(&b, "Subagent reference: %s\n", run.Ref)
1589 if strings.TrimSpace(run.ForkedFrom) != "" {
1590 fmt.Fprintf(&b, "Forked from: %s\n", strings.TrimSpace(run.ForkedFrom))
1591 b.WriteString("The requested ref resolves to an ancestor conversation transcript, so the framework continues a copy owned by the current conversation. To continue this copied subagent transcript in a later call, pass ")
1592 b.WriteString(run.Ref)
1593 b.WriteString(" as `continue_from`. Start a fresh subagent when the next task is independent.")
1594 return b.String()
1595 }
1596 b.WriteString("To continue this same subagent transcript in a later call, pass this ref as `continue_from`. Start a fresh subagent when the next task is independent.")
1597 return b.String()
1598 }
1599
1600 // GuardSubagentHostDecisionText appends a fixed boundary warning only when a
1601 // child agent result appears to discuss host approval or user-owned decisions.
1602 // The implementation lives in internal/tool so the skill tools share the exact
1603 // same phrase list and notice.
1604 func GuardSubagentHostDecisionText(answer string) string {
1605 return tool.GuardSubagentHostDecisionText(answer)
1606 }
1607
1608 // RunSubAgentWithSession continues an existing sub-agent session with prompt and
1609 // returns the latest final assistant answer. Fresh sub-agents pass a newly-created
1610 // session; continued sub-agents pass a loaded transcript session.
1611 //
1612 // Each call installs an independent session-private temporary directory Manager
1613 // so parent, sibling, and nested sub-agents never share temporary files.
1614 // continue_from restores conversation history only; each run gets a fresh temp dir.
1615 func RunSubAgentWithSession(ctx context.Context, prov provider.Provider, reg *tool.Registry, sess *Session, prompt string, opts Options, sink event.Sink) (string, error) {
1616 if sess == nil {
1617 return "", fmt.Errorf("sub-agent session is nil")
1618 }
1619 ctx = WithoutTurnContextBundle(ctx)
1620 // Isolate temporary files for this run before any tool execution.
1621 ctx = tool.WithoutGoalLifecycle(ctx)
1622 if opts.MemoryQueue != nil {
1623 ctx = memory.WithQueue(ctx, opts.MemoryQueue)
1624 } else {
1625 ctx = memory.WithoutQueue(ctx)
1626 }
1627 if opts.Jobs == nil {
1628 ctx = jobs.WithoutManager(ctx)
1629 }
1630 ctx, releaseTemp := withSubagentSessionTemp(ctx)
1631 defer releaseTemp()
1632 opts.SessionTemp = sessiontemp.FromContext(ctx)
1633 if opts.SubagentDepth > 0 {
1634 ctx = WithSubagentDepth(ctx, opts.SubagentDepth)
1635 }
1636 // Callers that wrap the prompt themselves (runSubSession) set
1637 // ClassifierTaskText before wrapping; for everyone else the prompt is
1638 // still pristine here, so capture it before host framing is prepended.
1639 if strings.TrimSpace(opts.ClassifierTaskText) == "" {
1640 opts.ClassifierTaskText = prompt
1641 }
1642 planWorkflow := PlanModeFromContext(ctx)
1643 if opts.SubagentDepth > 0 && isFreshSubagentSession(sess) {
1644 prompt = subagentStartContext + "\n\n" + prompt
1645 }
1646 if planWorkflow && !strings.Contains(prompt, planmode.Marker) {
1647 prompt = planmode.Marker + "\n\n" + prompt
1648 }
1649 opts.RequireReviewReportKind = ""
1650 // Nested reasoning stays isolated; the parent consumes only final Content.
1651 // Require it so a reasoning-only stop cannot fall back to older tool text.
1652 opts.RequireVisibleFinal = true
1653 sub := New(prov, reg, sess, opts, sink)
1654 sub.SetPlanMode(planWorkflow)
1655 if err := sub.Run(ctx, prompt); err != nil {
1656 // Preserve actual partial child execution even when the child fails.
1657 mergeChildEvidence(ctx, sub)
1658 return "", fmt.Errorf("sub-agent: %w", err)
1659 }
1660 mergeChildEvidence(ctx, sub)
1661 if answer := latestAssistantAnswer(sess); answer != "" {
1662 return composeSubagentAnswer(ctx, answer, sub, SubagentWriteClaim(ctx), opts.ClassifierTaskText), nil
1663 }
1664 return "", fmt.Errorf("sub-agent finished without producing a final answer")
1665 }
1666
1667 // readOnlyAgentConstruction is the single pairing every strictly read-only
1668 // loop shares: the permanent ReadOnlyExecution flag plus the final registry
1669 // filter. Batch children (RunReadOnlySubAgentWithSession) and legacy call sites
1670 // that still use NewReadOnlyAgent build through it, so a missed call site
1671 // cannot set only half the boundary. The interactive two-model planner uses
1672 // NewPlannerAgent instead (PlannerMCPExecution).
1673 func readOnlyAgentConstruction(reg *tool.Registry, opts Options) (*tool.Registry, Options) {
1674 opts.ReadOnlyExecution = true
1675 opts.PlannerMCPExecution = false
1676 return strictReadOnlyExecutionRegistry(reg), opts
1677 }
1678
1679 // NewReadOnlyAgent constructs a long-lived, strictly read-only agent through
1680 // the shared construction boundary. Prefer NewPlannerAgent for the two-model
1681 // planner so authorized non-destructive MCP can run via use_capability.
1682 func NewReadOnlyAgent(prov provider.Provider, reg *tool.Registry, sess *Session, opts Options, sink event.Sink) *Agent {
1683 reg, opts = readOnlyAgentConstruction(reg, opts)
1684 return New(prov, reg, sess, opts, sink)
1685 }
1686
1687 // NewPlannerAgent constructs the interactive two-model planner: permanent
1688 // ReadOnlyExecution still blocks bash, file writers, and ordinary non-MCP
1689 // writers, while PlannerMCPExecution allows authorized, non-destructive MCP
1690 // through the stable use_capability proxy without requiring readOnlyHint.
1691 func NewPlannerAgent(prov provider.Provider, reg *tool.Registry, sess *Session, opts Options, sink event.Sink) *Agent {
1692 opts.ReadOnlyExecution = true
1693 opts.PlannerMCPExecution = true
1694 // The coordinator needs visible plan text to hand off to the executor;
1695 // reasoning shown in a frontend is not a substitute for that contract.
1696 opts.RequireVisibleFinal = true
1697 // Keep construction-time filter for ordinary tools; use_capability stays
1698 // because it is ReadOnly. Direct mcp__* tools are already excluded by
1699 // PlannerToolRegistry. Dynamic MCP targets are re-checked after resolve.
1700 reg = plannerExecutionRegistry(reg)
1701 return New(prov, reg, sess, opts, sink)
1702 }
1703
1704 // plannerExecutionRegistry is the construction-time filter for NewPlannerAgent.
1705 // It removes ordinary writers and destructive direct MCP tools while keeping
1706 // use_capability and built-in research tools. Host-starting deferred MCP
1707 // targets are allowed at execution time under PlannerMCPExecution.
1708 func plannerExecutionRegistry(reg *tool.Registry) *tool.Registry {
1709 filtered := tool.NewRegistry()
1710 if reg == nil {
1711 return filtered
1712 }
1713 for _, name := range reg.Names() {
1714 target, ok := reg.Get(name)
1715 if !ok {
1716 continue
1717 }
1718 if name == "use_capability" {
1719 filtered.Add(target)
1720 continue
1721 }
1722 if strings.HasPrefix(name, tool.MCPNamePrefix) {
1723 // Defense in depth: planner never exposes direct MCP schemas.
1724 continue
1725 }
1726 if !target.ReadOnly() || mcpDestructiveHint(target) {
1727 continue
1728 }
1729 if h, ok := target.(tool.ReadOnlyExecutionHostMutation); ok && h.ReadOnlyExecutionHostMutation() {
1730 // Ordinary host mutations stay out; MCP startup is only via proxy.
1731 continue
1732 }
1733 filtered.Add(target)
1734 }
1735 return filtered
1736 }
1737
1738 // RunReadOnlySubAgentWithSession is the construction boundary for every
1739 // strictly read-only child loop. Registry filtering limits the visible surface;
1740 // this permanent execution flag also re-checks targets resolved dynamically by
1741 // proxy tools such as use_capability. It never enables PlannerMCPExecution.
1742 func RunReadOnlySubAgentWithSession(ctx context.Context, prov provider.Provider, reg *tool.Registry, sess *Session, prompt string, opts Options, sink event.Sink) (string, error) {
1743 reg, opts = readOnlyAgentConstruction(reg, opts)
1744 return RunSubAgentWithSession(ctx, prov, reg, sess, prompt, opts, sink)
1745 }
1746
1747 // strictReadOnlyExecutionRegistry is the final construction-time filter shared
1748 // by every strict child. Callers still apply role-specific filtering (review,
1749 // planner, profile allowlists), while this layer guarantees that a missed call
1750 // site cannot expose writers, destructive MCP tools, readers from unauthorized
1751 // servers, or an unauthorized host-starting target to the model.
1752 func strictReadOnlyExecutionRegistry(reg *tool.Registry) *tool.Registry {
1753 filtered := tool.NewRegistry()
1754 if reg == nil {
1755 return filtered
1756 }
1757 for _, name := range reg.Names() {
1758 target, ok := reg.Get(name)
1759 if retiredTool(name) || !ok || !target.ReadOnly() || mcpDestructiveHint(target) {
1760 continue
1761 }
1762 if isInstalledMCPTool(target) && !mcpServerAuthorized(target) {
1763 continue
1764 }
1765 if mutation, ok := target.(tool.ReadOnlyExecutionHostMutation); ok && mutation.ReadOnlyExecutionHostMutation() && !readOnlyExecutionAllowsMCPStartup(target) {
1766 continue
1767 }
1768 filtered.Add(target)
1769 }
1770 return filtered
1771 }
1772
1773 // latestAssistantAnswer walks the session backwards for the last assistant
1774 // message with content — that's the sub-agent's final answer. Intermediate
1775 // assistant messages with tool_calls but no text don't count.
1776 func latestAssistantAnswer(sess *Session) string {
1777 if sess == nil {
1778 return ""
1779 }
1780 for _, v := range slices.Backward(sess.Messages) {
1781 m := v
1782 if m.Role == provider.RoleAssistant && strings.TrimSpace(m.Content) != "" {
1783 return m.Content
1784 }
1785 }
1786 return ""
1787 }
1788
1789 // mergeChildEvidence folds a sub-agent's real receipts into the parent ledger
1790 // carried on ctx. Meta tools themselves are never mutations.
1791 func mergeChildEvidence(ctx context.Context, sub *Agent) {
1792 if sub == nil {
1793 return
1794 }
1795 parent, ok := evidence.FromContext(ctx)
1796 if !ok || parent == nil {
1797 return
1798 }
1799 parent.MergeChild(sub.EvidenceSummary())
1800 }
1801
1802 // EvidenceSummary exports this agent's turn-scoped receipts for parent merge.
1803 func (a *Agent) EvidenceSummary() evidence.ChildEvidenceSummary {
1804 if a == nil || a.task.ledger == nil {
1805 return evidence.ChildEvidenceSummary{}
1806 }
1807 return a.task.ledger.Summary()
1808 }
1809
1810 func isFreshSubagentSession(sess *Session) bool {
1811 if sess == nil {
1812 return false
1813 }
1814 snap := sess.Snapshot()
1815 return len(snap) == 1 && snap[0].Role == provider.RoleSystem
1816 }
1817
1818 // NestedSink returns a sink that forwards a sub-agent's tool activity to the
1819 // parent stream, nested under the tool call carried by ctx, so a frontend shows
1820 // it beneath that call (the same nesting `task` uses). Falls back to the given
1821 // sink when ctx carries no call context. Used by subagent skills.
1822 func NestedSink(ctx context.Context, fallback event.Sink) event.Sink {
1823 parentID, parent, _, ok := CallContext(ctx)
1824 if !ok || parent == nil {
1825 return fallback
1826 }
1827 return subSinkFor(parentID, parent)
1828 }
1829
1830 // subSink forwards a sub-agent's tool dispatch/result/progress events and
1831 // billable usage to the parent's event stream. Only tool activity is nested
1832 // visually; the sub-agent's text/reasoning stays isolated (progress previews
1833 // travel as reserved ToolProgress channels, not as parent Text/Reasoning) and
1834 // only its final answer is returned.
1835 //
1836 // The sub-agent's own turn/text/reasoning events are dropped — forwarding them
1837 // would make the parent transcript noisy and could imply they belong to the
1838 // parent model context, which they do not.
1839 //
1840 // Usage events are observability only, so forwarding them preserves billing
1841 // totals without polluting the parent provider-visible prefix.
1842 //
1843 // Tool events are tagged with the parent task call's ID so a frontend nests them
1844 // under it. The forwarded call IDs are namespaced with the parent ID so a
1845 // sub-agent call can never collide with a parent call in the frontend's
1846 // dispatch→result matching. ToolProgress covers both the sub-agent's real tool
1847 // output and nested sub-agent progress previews, which ride the same sink so
1848 // their IDs match the cards they belong to. Falls back to Discard when there's
1849 // no parent stream (the headless run loop, or a direct Execute in tests).
1850 func subSink(ctx context.Context) event.Sink {
1851 parentID, parent, _, ok := CallContext(ctx)
1852 if !ok || parent == nil {
1853 return event.Discard
1854 }
1855 return subSinkFor(parentID, parent)
1856 }
1857
1858 // subSinkFor builds the nesting sink from an already-captured parent ID + stream,
1859 // for the background path where the job runs under a context that no longer
1860 // carries the call context. Falls back to Discard when there's no parent stream.
1861 func subSinkFor(parentID string, parent event.Sink) event.Sink {
1862 if parent == nil {
1863 return event.Discard
1864 }
1865 return nestedSink{AuditForwarder: event.AuditForwarder{Inner: parent}, parentID: parentID, parent: parent}
1866 }
1867
1867 lines GO