| 1 | # Agent Fleet dogfood smoke spec (#3166) |
| 2 | # |
| 3 | # This spec exercises the fleet end-to-end: create a run with two local |
| 4 | # workers, run a workspace-check task and a protocol-review task, verify the |
| 5 | # ledger records receipts, and confirm the status surfaces work. Each worker is |
| 6 | # a headless `codewhale exec` run (see docs/AGENT_RUNTIME.md). |
| 7 | # |
| 8 | # Automated CI-safe smoke (no external services, no model calls): |
| 9 | # cargo test -p codewhale-tui --bins fleet::executor |
| 10 | # It drives several concurrent exec-style workers (with one injected failure) |
| 11 | # through the real host adapter and asserts terminal pass/fail outcomes. |
| 12 | # |
| 13 | # Manual run (drives real `codewhale exec` workers; needs provider creds): |
| 14 | # codewhale fleet run docs/examples/fleet-dogfood.toml --max-workers 2 --once |
| 15 | # |
| 16 | # Then check: |
| 17 | # codewhale fleet status |
| 18 | # codewhale fleet inspect <worker-id-from-status> |
| 19 | # codewhale fleet logs <worker-id-from-status> |
| 20 | # |
| 21 | # NOTE: this manual run path now drives real `codewhale exec` workers through |
| 22 | # the FleetExecutor. Use `--once` when you only want to enqueue/lease once and |
| 23 | # inspect state manually instead of keeping the manager loop attached. |
| 24 | |
| 25 | name = "dogfood smoke" |
| 26 | labels = { milestone = "v0.8.60", class = "smoke" } |
| 27 | |
| 28 | security_policy = { default_trust_level = "local", allowed_secrets = [], require_identity_verification = false } |
| 29 | |
| 30 | [[tasks]] |
| 31 | id = "cargo-check" |
| 32 | name = "Workspace check" |
| 33 | description = "Run `cargo check --workspace` and report any compilation errors." |
| 34 | objective = "Verify the workspace compiles cleanly with zero errors." |
| 35 | instructions = "Run `cargo check --workspace` in the repo root. If it compiles cleanly, report success. If there are errors, list each file:line and the error message. Do NOT attempt to fix anything — just report what you found." |
| 36 | worker = { role = "release-checker", tool_profile = "read-only", tools = ["cargo"], capabilities = ["rust"] } |
| 37 | workspace = { required_files = ["Cargo.toml"], writable_paths = [".codewhale/fleet"], environment = { required = ["PATH"] } } |
| 38 | input_files = ["Cargo.toml"] |
| 39 | context = ["You are running in a fleet smoke test. Be concise. Only report the pass/fail and any specific errors."] |
| 40 | budget = { max_tokens = 8000, max_tool_calls = 12, max_seconds = 300 } |
| 41 | expected_artifacts = ["log", "report", "receipt"] |
| 42 | scorer = { kind = "exit_code" } |
| 43 | retry_policy = { max_attempts = 2, initial_backoff_seconds = 5, max_backoff_seconds = 30 } |
| 44 | timeout_seconds = 300 |
| 45 | tags = ["smoke", "check"] |
| 46 | |
| 47 | [[tasks]] |
| 48 | id = "protocol-review" |
| 49 | name = "Protocol review" |
| 50 | description = "Review fleet protocol types for security and correctness." |
| 51 | objective = "Inspect crates/protocol/src/fleet.rs and crates/secrets/src/lib.rs. Report any missing serde defaults, unsafe wire changes, or security-sensitive fields lacking SecretRef." |
| 52 | instructions = "Read crates/protocol/src/fleet.rs and crates/secrets/src/lib.rs. Check for: (1) new fields without serde(default) or skip_serializing_if, (2) raw secrets in struct fields instead of FleetSecretRef, (3) missing Clone/Debug/PartialEq derives on new types. Write a concise report with file:line references for each finding. If there are no findings, report 'all clear'." |
| 53 | worker = { role = "reviewer", tool_profile = "read-only", tools = ["rg", "git", "cargo"], capabilities = ["rust"] } |
| 54 | workspace = { required_files = ["crates/protocol/src/fleet.rs", "crates/secrets/src/lib.rs"], writable_paths = [".codewhale/fleet"], environment = { required = ["PATH"] } } |
| 55 | input_files = ["crates/protocol/src/fleet.rs", "crates/secrets/src/lib.rs"] |
| 56 | context = ["You are a fleet protocol reviewer. Be thorough but concise. Reference specific lines."] |
| 57 | budget = { max_tokens = 10000, max_tool_calls = 16, max_seconds = 600 } |
| 58 | expected_artifacts = ["log", "report", "receipt"] |
| 59 | scorer = { kind = "code_whale_verifier_prompt", prompt = "Verify the review includes at least one concrete file:line finding or explicitly says 'all clear'." } |
| 60 | retry_policy = { max_attempts = 1, initial_backoff_seconds = 10 } |
| 61 | timeout_seconds = 600 |
| 62 | tags = ["smoke", "review", "protocol"] |
| 63 |