From 3ae1102d13993ff9085ac0eb7d6c29fe20e3466c Mon Sep 17 00:00:00 2001 From: Mohamed Boudra Date: Sat, 14 Mar 2026 22:04:09 +0700 Subject: [PATCH] feat: add agent ID/name resolution and provider/model syntax support --- packages/cli/src/cli.ts | 2 +- packages/cli/src/commands/agent/archive.ts | 16 +- packages/cli/src/commands/agent/index.ts | 4 +- packages/cli/src/commands/agent/inspect.ts | 6 + packages/cli/src/commands/agent/run.ts | 61 +++- packages/cli/tests/05-agent-run.test.ts | 22 ++ .../tests/21-run-output-schema-helper.test.ts | 29 ++ .../cli/tests/29-agent-id-resolution.test.ts | 44 +++ skills/paseo-loop/SKILL.md | 281 ++++++++------- skills/paseo-loop/bin/loop.sh | 325 +++++++++++------- 10 files changed, 533 insertions(+), 257 deletions(-) create mode 100644 packages/cli/tests/29-agent-id-resolution.test.ts diff --git a/packages/cli/src/cli.ts b/packages/cli/src/cli.ts index 03e5a90d7..c02f8c8f5 100644 --- a/packages/cli/src/cli.ts +++ b/packages/cli/src/cli.ts @@ -75,7 +75,7 @@ export function createCli(): Command { .argument('', 'The task/prompt for the agent') .option('-d, --detach', 'Run in background (detached)') .option('--name ', 'Assign a name/title to the agent') - .option('--provider ', 'Agent provider: claude | codex | opencode', 'claude') + .option('--provider ', 'Agent provider, or provider/model (e.g. codex or codex/gpt-5.4)', 'claude') .option('--model ', 'Model to use (e.g., claude-sonnet-4-20250514, claude-3-5-haiku-20241022)') .option('--thinking ', 'Thinking option ID to use for this run') .option('--mode ', 'Provider-specific mode (e.g., plan, default, bypass)') diff --git a/packages/cli/src/commands/agent/archive.ts b/packages/cli/src/commands/agent/archive.ts index d842d128b..517e775ea 100644 --- a/packages/cli/src/commands/agent/archive.ts +++ b/packages/cli/src/commands/agent/archive.ts @@ -1,5 +1,5 @@ import type { Command } from 'commander' -import { connectToDaemon, getDaemonHost } from '../../utils/client.js' +import { connectToDaemon, getDaemonHost, resolveAgentId } from '../../utils/client.js' import type { CommandOptions, SingleResult, OutputSchema, CommandError } from '../../output/index.js' /** Result type for agent archive command */ @@ -38,7 +38,7 @@ export async function runArchiveCommand( const error: CommandError = { code: 'MISSING_AGENT_ID', message: 'Agent ID is required', - details: 'Usage: paseo agent archive ', + details: 'Usage: paseo agent archive ', } throw error } @@ -57,8 +57,10 @@ export async function runArchiveCommand( } try { - const fetchResult = await client.fetchAgent(agentIdArg) - if (!fetchResult) { + const agentsPayload = await client.fetchAgents({ filter: { includeArchived: true } }) + const agents = agentsPayload.entries.map((entry) => entry.agent) + const agentId = resolveAgentId(agentIdArg, agents) + if (!agentId) { const error: CommandError = { code: 'AGENT_NOT_FOUND', message: `Agent not found: ${agentIdArg}`, @@ -66,8 +68,10 @@ export async function runArchiveCommand( } throw error } - const agent = fetchResult.agent - const agentId = agent.id + const agent = agents.find((entry) => entry.id === agentId) + if (!agent) { + throw new Error(`Resolved agent missing from fetched agents: ${agentId}`) + } // Check if agent is already archived if (agent.archivedAt) { diff --git a/packages/cli/src/commands/agent/index.ts b/packages/cli/src/commands/agent/index.ts index 60901576f..244e3cb0d 100644 --- a/packages/cli/src/commands/agent/index.ts +++ b/packages/cli/src/commands/agent/index.ts @@ -39,7 +39,7 @@ export function createAgentCommand(): Command { .argument('', 'The task/prompt for the agent') .option('-d, --detach', 'Run in background (detached)') .option('--name ', 'Assign a name/title to the agent') - .option('--provider ', 'Agent provider: claude | codex | opencode', 'claude') + .option('--provider ', 'Agent provider, or provider/model (e.g. codex or codex/gpt-5.4)', 'claude') .option('--model ', 'Model to use (e.g., claude-sonnet-4-20250514, claude-3-5-haiku-20241022)') .option('--thinking ', 'Thinking option ID to use for this run') .option('--mode ', 'Provider-specific mode (e.g., plan, default, bypass)') @@ -121,7 +121,7 @@ export function createAgentCommand(): Command { agent .command('archive') .description('Archive an agent (soft-delete)') - .argument('', 'Agent ID (or prefix)') + .argument('', 'Agent ID, prefix, or name') .option('--force', 'Force archive running agent (interrupts active run first)') ).action(withOutput(runArchiveCommand)) diff --git a/packages/cli/src/commands/agent/inspect.ts b/packages/cli/src/commands/agent/inspect.ts index fc76552ab..4cb761aae 100644 --- a/packages/cli/src/commands/agent/inspect.ts +++ b/packages/cli/src/commands/agent/inspect.ts @@ -11,6 +11,8 @@ interface AgentInspect { Model: string Thinking: string Status: string + Archived: boolean + ArchivedAt: string | null Mode: string Cwd: string CreatedAt: string @@ -127,6 +129,8 @@ function toInspectData(snapshot: AgentSnapshotPayload): AgentInspect { Model: resolveModel(snapshot) ?? '-', Thinking: snapshot.effectiveThinkingOptionId ?? 'auto', Status: snapshot.status, + Archived: snapshot.archivedAt != null, + ArchivedAt: snapshot.archivedAt ?? null, Mode: snapshot.currentModeId ?? 'default', Cwd: snapshot.cwd, CreatedAt: snapshot.createdAt, @@ -154,6 +158,8 @@ function toInspectRows(agent: AgentInspect): InspectRow[] { { key: 'Model', value: agent.Model }, { key: 'Thinking', value: agent.Thinking }, { key: 'Status', value: agent.Status }, + { key: 'Archived', value: String(agent.Archived) }, + { key: 'ArchivedAt', value: agent.ArchivedAt ?? 'null' }, { key: 'Mode', value: agent.Mode }, { key: 'Cwd', value: shortenPath(agent.Cwd) }, { key: 'CreatedAt', value: agent.CreatedAt }, diff --git a/packages/cli/src/commands/agent/run.ts b/packages/cli/src/commands/agent/run.ts index 85df55174..b67ee0a64 100644 --- a/packages/cli/src/commands/agent/run.ts +++ b/packages/cli/src/commands/agent/run.ts @@ -46,6 +46,11 @@ export interface AgentRunOptions extends CommandOptions { outputSchema?: string } +interface ResolvedProviderModel { + provider: string + model: string | undefined +} + function toRunResult( agent: AgentSnapshotPayload, statusOverride?: AgentRunResult['status'] @@ -165,6 +170,52 @@ function structuredRunSchema(output: Record): OutputSchema): ResolvedProviderModel { + const providerInput = options.provider?.trim() || 'claude' + const modelInput = options.model?.trim() + + if (options.model !== undefined && !modelInput) { + const error: CommandError = { + code: 'INVALID_MODEL', + message: '--model cannot be empty', + } + throw error + } + + const slashIndex = providerInput.indexOf('/') + if (slashIndex === -1) { + return { + provider: providerInput, + model: modelInput, + } + } + + const provider = providerInput.slice(0, slashIndex).trim() + const modelFromProvider = providerInput.slice(slashIndex + 1).trim() + if (!provider || !modelFromProvider) { + const error: CommandError = { + code: 'INVALID_PROVIDER', + message: 'Invalid --provider value', + details: 'Use --provider or --provider /', + } + throw error + } + + if (modelInput && modelInput !== modelFromProvider) { + const error: CommandError = { + code: 'CONFLICTING_MODEL_OPTIONS', + message: 'Conflicting model values provided', + details: `--provider specifies model ${modelFromProvider}, but --model specifies ${modelInput}`, + } + throw error + } + + return { + provider, + model: modelInput ?? modelFromProvider, + } +} + export async function runRunCommand( prompt: string, options: AgentRunOptions, @@ -203,6 +254,8 @@ export async function runRunCommand( throw error } + const resolvedProviderModel = resolveProviderAndModel(options) + let client try { client = await connectToDaemon({ host: options.host as string | undefined }) @@ -289,11 +342,11 @@ export async function runRunCommand( const callStructuredTurn = async (structuredPrompt: string): Promise => { if (!structuredAgent) { structuredAgent = await client.createAgent({ - provider: (options.provider as 'claude' | 'codex' | 'opencode') ?? 'claude', + provider: resolvedProviderModel.provider as 'claude' | 'codex' | 'opencode', cwd, title: options.name, modeId: options.mode, - model: options.model, + model: resolvedProviderModel.model, thinkingOptionId, initialPrompt: structuredPrompt, images, @@ -387,11 +440,11 @@ export async function runRunCommand( // Create the agent const agent = await client.createAgent({ - provider: (options.provider as 'claude' | 'codex' | 'opencode') ?? 'claude', + provider: resolvedProviderModel.provider as 'claude' | 'codex' | 'opencode', cwd, title: options.name, modeId: options.mode, - model: options.model, + model: resolvedProviderModel.model, thinkingOptionId, initialPrompt: prompt, images, diff --git a/packages/cli/tests/05-agent-run.test.ts b/packages/cli/tests/05-agent-run.test.ts index 99ac41503..56c73e6d1 100644 --- a/packages/cli/tests/05-agent-run.test.ts +++ b/packages/cli/tests/05-agent-run.test.ts @@ -130,6 +130,17 @@ try { console.log('✓ run --provider flag is accepted\n') } + // Test 6b: run --provider provider/model syntax is accepted + { + console.log('Test 6b: run --provider provider/model syntax is accepted') + const result = + await $`PASEO_HOST=localhost:${port} PASEO_HOME=${paseoHome} npx paseo run --provider codex/gpt-5.4 "test prompt"`.nothrow() + const output = result.stdout + result.stderr + assert(!output.includes('unknown option'), 'should accept provider/model syntax') + assert(!output.includes('error: option'), 'should not have option parsing error') + console.log('✓ run --provider provider/model syntax is accepted\n') + } + // Test 7: run --mode flag is accepted { console.log('Test 7: run --mode flag is accepted') @@ -199,6 +210,17 @@ try { console.log('✓ Combined flags work together\n') } + // Test 12b: conflicting provider/model syntax is rejected before connect + { + console.log('Test 12b: conflicting provider/model syntax is rejected') + const result = + await $`PASEO_HOST=localhost:${port} PASEO_HOME=${paseoHome} npx paseo run --provider codex/gpt-5.4 --model gpt-5.5 "test prompt"`.nothrow() + assert.notStrictEqual(result.exitCode, 0, 'should fail for conflicting model inputs') + const output = result.stdout + result.stderr + assert(output.includes('Conflicting model values provided'), 'should explain conflicting model inputs') + console.log('✓ conflicting provider/model syntax is rejected\n') + } + // Test 13: paseo --help shows run command { console.log('Test 13: paseo --help shows run command') diff --git a/packages/cli/tests/21-run-output-schema-helper.test.ts b/packages/cli/tests/21-run-output-schema-helper.test.ts index edbcf3a6b..f9e4dbd8c 100644 --- a/packages/cli/tests/21-run-output-schema-helper.test.ts +++ b/packages/cli/tests/21-run-output-schema-helper.test.ts @@ -2,6 +2,7 @@ import assert from 'node:assert' import { + resolveProviderAndModel, resolveStructuredResponseMessage, type StructuredResponseTimelineClient, } from '../src/commands/agent/run.ts' @@ -94,4 +95,32 @@ console.log('=== Run Output Schema Helper Tests ===\n') console.log('✓ returns null when timeline fetch throws') } +// Test 5: Provider/model slash syntax resolves both values. +{ + const result = resolveProviderAndModel({ provider: 'codex/gpt-5.4' }) + assert.deepStrictEqual(result, { + provider: 'codex', + model: 'gpt-5.4', + }) + console.log('✓ resolves provider/model slash syntax') +} + +// Test 6: Explicit matching --model coexists with slash syntax. +{ + const result = resolveProviderAndModel({ provider: 'codex/gpt-5.4', model: 'gpt-5.4' }) + assert.deepStrictEqual(result, { + provider: 'codex', + model: 'gpt-5.4', + }) + console.log('✓ accepts matching explicit model with slash syntax') +} + +// Test 7: Conflicting --model is rejected. +{ + assert.throws(() => resolveProviderAndModel({ provider: 'codex/gpt-5.4', model: 'gpt-5.5' }), { + message: 'Conflicting model values provided', + }) + console.log('✓ rejects conflicting explicit model with slash syntax') +} + console.log('\n=== All helper tests passed ===') diff --git a/packages/cli/tests/29-agent-id-resolution.test.ts b/packages/cli/tests/29-agent-id-resolution.test.ts new file mode 100644 index 000000000..ed3130c58 --- /dev/null +++ b/packages/cli/tests/29-agent-id-resolution.test.ts @@ -0,0 +1,44 @@ +#!/usr/bin/env npx tsx + +import assert from 'node:assert' +import { resolveAgentId } from '../src/utils/client.js' + +console.log('=== Agent ID Resolution Tests ===\n') + +const agents = [ + { id: 'abc1234-full-agent-id', title: 'Build docs' }, + { id: 'def5678-other-agent-id', title: 'Fix archive flow' }, + { id: '987zyxw-third-agent-id', title: 'Review PR' }, +] + +{ + console.log('Test 1: exact ID match resolves') + assert.strictEqual(resolveAgentId('abc1234-full-agent-id', agents), 'abc1234-full-agent-id') + console.log('✓ exact ID match resolves\n') +} + +{ + console.log('Test 2: ID prefix resolves') + assert.strictEqual(resolveAgentId('def5678', agents), 'def5678-other-agent-id') + console.log('✓ ID prefix resolves\n') +} + +{ + console.log('Test 3: exact name match resolves case-insensitively') + assert.strictEqual(resolveAgentId('fix archive flow', agents), 'def5678-other-agent-id') + console.log('✓ exact name match resolves case-insensitively\n') +} + +{ + console.log('Test 4: partial name match resolves when unique') + assert.strictEqual(resolveAgentId('review', agents), '987zyxw-third-agent-id') + console.log('✓ partial name match resolves when unique\n') +} + +{ + console.log('Test 5: missing agent returns null') + assert.strictEqual(resolveAgentId('does not exist', agents), null) + console.log('✓ missing agent returns null\n') +} + +console.log('=== All agent ID resolution tests passed ===') diff --git a/skills/paseo-loop/SKILL.md b/skills/paseo-loop/SKILL.md index a7e034387..a019511f9 100644 --- a/skills/paseo-loop/SKILL.md +++ b/skills/paseo-loop/SKILL.md @@ -1,12 +1,12 @@ --- name: paseo-loop -description: Run a task in a worker→judge loop until acceptance criteria are met. Use when the user says "loop", "loop this", "keep trying until", or wants iterative autonomous execution. +description: Run a task in a loop until an exit condition is met. Use when the user says "loop", "loop this", "keep trying until", "babysit", "poll", or wants iterative autonomous execution. user-invocable: true --- # Loop Skill -You are setting up an autonomous worker→judge loop. A worker agent implements the plan, then a judge agent independently verifies the plan was implemented to the letter. +You are setting up an autonomous loop. An agent runs repeatedly until an exit condition is met. **User's arguments:** $ARGUMENTS @@ -18,14 +18,19 @@ Load the **Paseo skill** first — it contains the CLI reference for all agent c ## What Is a Loop -A loop is two roles — **worker** and **judge** — running in alternation: +A loop runs an agent repeatedly until it's done. There are two modes: -1. **Worker** receives the plan and implements it (fresh agent each iteration) -2. **Judge** receives the same plan and independently verifies it was implemented correctly, emitting a structured verdict: `{ criteria_met: boolean, reason: string }` -3. If `criteria_met` is false, a new worker is launched with the latest judge failure reason (latest only, not full history) -4. Loop exits when the judge says `criteria_met: true` or max iterations are reached +### Self-terminating (no verifier) -The judge verifies the **entire plan** — every step, every file change, every acceptance criterion. It does NOT suggest fixes or provide guidance. It only reports facts. +A single worker agent runs each iteration. It returns `{ done: boolean, reason: string }` via structured output. The loop exits when `done` is true. + +### Worker + Verifier + +A worker agent runs each iteration (detached, no structured output). After the worker finishes, a separate verifier agent evaluates the verification prompt and returns `{ done: boolean, reason: string }`. The loop exits when `done` is true. + +### Feedback between iterations + +In both modes, the `reason` from the previous iteration is fed back to the next worker as ``. This gives the worker context about what happened last time. ## Live Steering @@ -33,128 +38,137 @@ Each loop run persists state in: ```text ~/.paseo/loops// - plan.md - last_reason.md - history.log + worker-prompt.md # worker prompt (live-editable) + verifier-prompt.md # verifier prompt (live-editable, only when verifier is used) + last_reason.md # latest iteration result + history.log # per-iteration records ``` -Behavior: -- `plan.md` is read immediately before launching each worker and each judge. -- `last_reason.md` stores only the latest judge failure reason (not accumulated history). -- `history.log` appends per-iteration operational records. - -Edits to `plan.md` are picked up on the next worker/judge launch without restarting the loop. +Edits to prompt files are picked up on the next iteration without restarting the loop. ## Parsing Arguments Parse `$ARGUMENTS` to determine: -1. **Plan** — the full implementation plan (context, steps, acceptance criteria, constraints — all in one document) -2. **Worker model** — who does the work (default: Codex `gpt-5.4`) -3. **Judge model** — who verifies (default: Claude `sonnet`) -4. **Name** — a short name for tracking iterations -5. **Max iterations** — safety cap (default: 10) -6. **Worktree** — whether to run in an isolated git worktree +1. **Worker prompt** — what the worker does each iteration +2. **Verifier prompt** (optional) — what the verifier checks after each worker iteration +3. **Worker** — which agent does the work (default: Codex) +4. **Verifier** — which agent verifies (default: Claude sonnet) +5. **Sleep** (optional) — delay between iterations +6. **Name** — a short name for tracking +7. **Max iterations** — safety cap (default: unlimited) +8. **Archive** — whether to archive agents after each iteration +9. **Worktree** — whether to run in an isolated git worktree ### Examples ``` -/loop -→ Agent derives plan from conversation context +/loop babysit PR #42 until CI is green, check every 5 minutes +→ worker-prompt: "Check the CI status of PR #42 using `gh pr checks 42`. Report done when ALL checks + have passed (no pending, no failures). If any checks are still running or have failed, report not + done and list which checks are pending or failing." + No verifier (self-terminating: the worker inspects CI and reports done when green) + sleep: 5m + archive: yes + name: babysit-pr-42 -/loop the provider list bug -→ Plan: full context about the bug, steps to fix, acceptance criteria - Name: provider-list-fix +/loop implement the auth refactor from the plan in /tmp/plan.md +→ worker-prompt-file: /tmp/plan.md + verifier-prompt: "Verify every step of the plan was implemented. Check file changes, types, and + run `npm run typecheck` and `npm test`. Report each criterion individually with evidence." + name: auth-refactor + worktree: auth-refactor -/loop until tests pass -→ Plan: make the failing tests pass, acceptance criteria: all tests pass - Name: fix-tests +/loop run the test suite until it passes +→ worker-prompt: "Run `npm test`. If any tests fail, read the failure output, investigate the root + cause in the source code, and fix it. Report done when `npm test` exits with code 0 and all + tests pass. Report not done if any test still fails, and explain which tests failed and why." + No verifier (self-terminating: worker reports done when tests pass) + name: fix-tests -/loop in a worktree -→ Same as above but all agents run in a shared worktree +/loop watch error rates after deploy, check every 10 minutes +→ worker-prompt: "Check the error rate for the canary deployment by running `./scripts/check-canary.sh`. + Report done when the error rate has been below 0.1% for at least 2 consecutive checks. Report not + done with the current error rate and trend." + No verifier (self-terminating: worker reports done when error rate is stable) + sleep: 10m + max-iterations: 30 + archive: yes + name: canary-watch ``` ## Using the Script -The loop is implemented as a bash script at `~/.agents/skills/loop/bin/loop.sh`. +The loop is implemented as a bash script at `~/.claude/skills/paseo-loop/bin/loop.sh`. ```bash -~/.agents/skills/loop/bin/loop.sh \ - --plan "The full implementation plan" \ - --name "short-name" \ - --max-iterations 10 \ - --worker-provider codex \ - --worker-model gpt-5.4 \ - --judge-provider claude \ - --judge-model sonnet +# Self-terminating: worker checks PR CI and reports done when all checks pass +~/.claude/skills/paseo-loop/bin/loop.sh \ + --worker-prompt "Check CI status of PR #42 using gh pr checks 42. Report done when ALL checks pass. Report not done with a list of pending/failing checks." \ + --name "babysit-pr" \ + --sleep 5m \ + --archive + +# Worker + verifier: worker implements, verifier independently checks +~/.claude/skills/paseo-loop/bin/loop.sh \ + --worker-prompt "Implement the auth refactor: ..." \ + --verifier-prompt "Verify the auth refactor is complete. Run npm run typecheck and npm test. Check that all file changes match the plan. Report each criterion with evidence." \ + --name "auth-refactor" \ + --worktree "auth-refactor" ``` ### Arguments | Flag | Required | Default | Description | |---|---|---|---| -| `--plan` | Yes* | — | Full implementation plan (given to both worker and judge) | -| `--plan-file` | Yes* | — | Read the plan from a file | -| `--name` | Yes | — | Name prefix for agents (`name-work-N`, `name-verify-N`) | -| `--max-iterations` | No | 10 | Safety cap on iterations | -| `--worktree` | No | — | Worktree name. Created on first use, reused for all subsequent agents. | -| `--worker-provider` | No | `codex` | `codex` or `claude` | -| `--worker-model` | No | provider default | Model for worker | -| `--judge-provider` | No | `claude` | `codex` or `claude` | -| `--judge-model` | No | provider default | Model for judge | +| `--worker-prompt` | Yes* | — | Prompt given to the worker each iteration | +| `--worker-prompt-file` | Yes* | — | Read the worker prompt from a file | +| `--worker` | No | `codex` | Worker agent (`provider/model`, e.g. `codex/gpt-5.4`, `claude/sonnet`) | +| `--verifier-prompt` | No* | — | Verification prompt for a separate verifier agent | +| `--verifier-prompt-file` | No* | — | Read the verifier prompt from a file | +| `--verifier` | No | `claude/sonnet` | Verifier agent (`provider/model`) | +| `--name` | Yes | — | Name prefix for agents | +| `--sleep` | No | — | Delay between iterations (e.g. `30s`, `5m`, `1h`) | +| `--max-iterations` | No | unlimited | Safety cap on iterations | +| `--archive` | No | off | Archive agents after each iteration | +| `--worktree` | No | — | Worktree name. Created on first use, reused after. | | `--thinking` | No | `medium` | Thinking level for worker | -\* Provide exactly one of `--plan` or `--plan-file`. +\* Provide exactly one of `--worker-prompt` or `--worker-prompt-file`. Provide at most one of `--verifier-prompt` or `--verifier-prompt-file`. -### How the Plan Is Used +### Behavior by parameters -The same plan is given to both the worker and the judge, wrapped in XML tags: - -**Worker receives:** -``` -Implement the following plan exactly as specified. - - -[your plan here] - -``` - -If a previous iteration failed, the worker also receives: -``` - -[judge's failure reason from last iteration] - -``` - -**Judge receives:** -``` -Verify that every step of the plan has been implemented to the letter. - - -[your plan here] - -``` +| Parameters | Mode | Use case | +|---|---|---| +| `--worker-prompt` only | Self-terminating worker | Worker does work and decides when done | +| `--worker-prompt` + `--verifier-prompt` | Worker + verifier | Worker implements, verifier independently checks | +| `--worker-prompt` + `--sleep` | Polling with self-termination | Periodic check until a condition is met | +| `--worker-prompt` + `--verifier-prompt` + `--sleep` | Periodic work + verifier | Periodic work with independent verification | ### Agent Naming -Agents are named `{name}-work-{N}` and `{name}-verify-{N}` so the caller can track iterations: - +Without verifier: agents are named `{name}-{N}`: ``` -fix-bug-work-1 # First worker attempt -fix-bug-verify-1 # First judge check -fix-bug-work-2 # Second worker attempt (with failure context) -fix-bug-verify-2 # Second judge check +babysit-1 # First iteration +babysit-2 # Second iteration +``` + +With verifier: workers are `{name}-{N}`, verifiers are `{name}-verify-{N}`: +``` +feat-1 # First worker +feat-verify-1 # First verifier +feat-2 # Second worker (with previous result context) +feat-verify-2 # Second verifier ``` ### Worktree Support -When `--worktree` is passed, all workers and judges run in the same git worktree. The worktree is created on the first agent launch and reused for all subsequent agents (the `--worktree` flag on `paseo run` is idempotent — if the worktree exists, the agent runs in it without creating a new one). - -No `--base` is needed in the loop script — it automatically uses the current branch as the base. +When `--worktree` is passed, all agents run in the same git worktree. The worktree is created on first launch and reused for all subsequent agents. ```bash -~/.agents/skills/loop/bin/loop.sh \ - --plan-file /tmp/my-plan.md \ +~/.claude/skills/paseo-loop/bin/loop.sh \ + --worker-prompt "Implement the feature..." \ + --verifier-prompt "Verify the feature works and typecheck passes" \ --name "feature-x" \ --worktree "feature-x" ``` @@ -162,60 +176,73 @@ No `--base` is needed in the loop script — it automatically uses the current b ## Your Job 1. **Understand the task** from the conversation context and `$ARGUMENTS` -2. **Write the plan** — comprehensive, self-contained (the worker has zero context). The plan should include: - - Task description and context - - Relevant files and what they do - - Implementation steps (ordered, concrete) - - Acceptance criteria (factual, verifiable) - - Constraints (what NOT to do) -3. **Choose models** — default: Codex worker + Claude sonnet judge -4. **Choose a name** — short, descriptive -5. **Run the script** — call `loop.sh` with all the arguments +2. **Decide the mode** — does this need a separate verifier, or can the worker self-terminate? +3. **Write the worker prompt** — what the worker does each iteration. Must be self-contained (the agent has zero prior context). +4. **Write the verifier prompt** (if needed) — what the verifier checks. Should be factual and verifiable. +5. **Choose sleep** — if the task is polling/monitoring, add a sleep duration +6. **Choose archive** — use `--archive` for long-running or polling loops to keep the agent list clean +7. **Choose agents** — default: `codex` worker + `claude/sonnet` verifier +8. **Choose a name** — short, descriptive +9. **Run the script** — call `loop.sh` with all the arguments -### Steering Guidance +### When to use a verifier vs self-terminating -- Prefer tightening or clarifying the plan when important requirements are missing. -- Rework acceptance criteria when they are causing dead loops or are not independently verifiable. -- Avoid weakening user-defined criteria unless the user explicitly asks for that change. -- Keep the plan self-contained so the loop can succeed without relying on iterative reasoning context. +Use a verifier (`--verifier-prompt`) when: +- The worker's job is to implement something and you want independent verification +- You want the verification done by a different agent than the worker +- The worker should focus on doing work, not on judging its own work -### Writing a Good Plan +Use self-terminating (no verifier) when: +- The worker is checking/polling an external condition (CI status, deployment health) +- The worker can objectively determine when it's done (tests pass, file exists) +- You want a single agent doing both the work and the evaluation -The plan is the single source of truth for both the worker and the judge. It must be: +### Writing a Good Worker Prompt -1. **Self-contained** — The worker starts with zero context. Everything needed is in the plan. -2. **Specific** — Name files, functions, types. "Update the handler" is useless. "In `src/handlers/auth.ts`, modify the `handleLogin` function to..." is useful. -3. **Ordered** — Steps should be in implementation order. -4. **Verifiable** — Every step and criterion must be independently checkable by the judge. +The worker prompt is what the agent receives each iteration. It must be: -**Good plan elements:** -- "File `src/utils.ts` exports a function named `formatDate`" -- "Running `npm test` exits with code 0" -- "The component at `src/ProviderList.tsx` renders a list of providers from the API response" -- "Typecheck passes (`npm run typecheck`)" -- "`npm run typecheck` passes" +1. **Self-contained** — The agent starts with zero context. Everything needed is in the prompt. +2. **Specific** — Name files, functions, types, URLs. Be concrete. +3. **Action-oriented** — Tell the agent what to do, not what to think about. -**Bad plan elements:** -- "The code is clean" (subjective) -- "The implementation is correct" (vague) -- "It works" (unverifiable) +### Writing a Good Verifier Prompt + +The verifier prompt defines what the verifier checks after each worker iteration. It can range from strict factual checks to qualitative code review: + +**Factual / objective checks:** +- "Run `npm test` and `npm run typecheck`. Report done only if both pass with zero failures." +- "Check that PR #42 has all CI checks green and no unresolved review comments." +- "Verify the API responds to `GET /health` with status 200 within 500ms." + +**Code quality / style checks:** +- "Review the changes for DRY violations. Report done when there is no duplicated logic across files." +- "Check that the implementation does not over-engineer. No unnecessary abstractions, no premature generalization, no feature flags for single-use code." +- "Verify the code follows the project's conventions: functional style, no classes, explicit types, no `any`." + +**Performance / resource checks:** +- "Run the benchmark suite. Report done when p99 latency is under 100ms." +- "Check bundle size. Report done when the production build is under 500KB gzipped." + +**Comprehensive verification:** +- "Verify every step of the plan was implemented. Check file changes, types, test output. Run `npm run typecheck` and `npm test`. Report each criterion individually with evidence." + +The verifier should report facts with evidence, not suggest fixes. ### Skill Stacking -You can instruct the worker to use other skills in the plan: +You can instruct the worker to use other skills: ```bash -~/.agents/skills/loop/bin/loop.sh \ - --plan "Use the /committee skill to plan and then fix the provider list bug. The bug is..." \ +~/.claude/skills/paseo-loop/bin/loop.sh \ + --worker-prompt "Use /committee to plan, then fix the provider list bug. The bug is..." \ + --verifier-prompt "The provider list renders correctly and npm run typecheck passes" \ --name "provider-fix" ``` ### Composing with Handoff -A handoff can launch a loop in a worktree: +A handoff can launch a loop: ``` -/handoff a loop in a worktree +/handoff a loop in a worktree to babysit PR #42 ``` - -The handoff skill writes a prompt telling the new agent to use `/loop`, and the loop runs inside the worktree. diff --git a/skills/paseo-loop/bin/loop.sh b/skills/paseo-loop/bin/loop.sh index 5db1723bd..e4b9b37f1 100755 --- a/skills/paseo-loop/bin/loop.sh +++ b/skills/paseo-loop/bin/loop.sh @@ -3,49 +3,78 @@ set -euo pipefail usage() { cat <<'EOF' -Usage: loop.sh (--plan PROMPT | --plan-file PATH) --name NAME [options] +Usage: loop.sh --worker-prompt TEXT --name NAME [options] -Required: - --plan PROMPT The implementation plan (given to both worker and judge) - --plan-file PATH Read the plan from a file - --name NAME Name prefix for agents (e.g. "fix-bug" → "fix-bug-work-1", "fix-bug-verify-1") +Worker (required): + --worker-prompt TEXT Prompt given to the worker agent each iteration + --worker-prompt-file PATH Read the worker prompt from a file + --worker PROVIDER/MODEL Worker agent (e.g. codex, codex/gpt-5.4, claude/sonnet). Default: codex + +Verifier (optional): + --verifier-prompt TEXT Verification prompt evaluated by a separate agent after each iteration + --verifier-prompt-file PATH Read the verification prompt from a file + --verifier PROVIDER/MODEL Verifier agent (e.g. claude/sonnet, codex/gpt-5.4). Default: claude/sonnet + + When a verifier prompt is provided, a separate verifier agent evaluates the + condition after each worker iteration. When omitted, the worker itself returns + { done: boolean, reason: string } and decides when the loop is done. Options: - --max-iterations N Maximum loop iterations (default: 10) + --name NAME Name prefix for agents (required) + --sleep DURATION Sleep between iterations (e.g. 30s, 5m, 1h). Default: no delay. + --max-iterations N Maximum loop iterations (default: unlimited) + --archive Archive agents after each iteration --worktree NAME Run all agents in this worktree (created on first use, reused after) - --worker-provider P Worker provider: claude or codex (default: codex) - --worker-model M Worker model (default: provider's default) - --judge-provider P Judge provider: claude or codex (default: claude) - --judge-model M Judge model (default: provider's default) --thinking LEVEL Thinking level for worker (default: medium) EOF exit 1 } +parse_agent_spec() { + local spec="$1" + local default_provider="$2" + local default_model="$3" + + if [[ -z "$spec" ]]; then + echo "$default_provider" "$default_model" + return + fi + + if [[ "$spec" == */* ]]; then + echo "${spec%%/*}" "${spec#*/}" + else + echo "$spec" "" + fi +} + # Defaults -max_iterations=10 -worker_provider="codex" -worker_model="" -judge_provider="claude" -judge_model="sonnet" -plan="" -plan_file_input="" +max_iterations=0 +archive=false +worker_spec="" +verifier_spec="" +worker_prompt_input="" +worker_prompt_file_input="" +verifier_prompt_input="" +verifier_prompt_file_input="" name="" thinking="medium" worktree="" +sleep_duration="" state_root="${HOME}/.paseo/loops" while [[ $# -gt 0 ]]; do case "$1" in - --plan) plan="$2"; shift 2 ;; - --plan-file) plan_file_input="$2"; shift 2 ;; + --worker-prompt) worker_prompt_input="$2"; shift 2 ;; + --worker-prompt-file) worker_prompt_file_input="$2"; shift 2 ;; + --worker) worker_spec="$2"; shift 2 ;; + --verifier-prompt) verifier_prompt_input="$2"; shift 2 ;; + --verifier-prompt-file) verifier_prompt_file_input="$2"; shift 2 ;; + --verifier) verifier_spec="$2"; shift 2 ;; --name) name="$2"; shift 2 ;; --max-iterations) max_iterations="$2"; shift 2 ;; + --archive) archive=true; shift ;; + --sleep) sleep_duration="$2"; shift 2 ;; --worktree) worktree="$2"; shift 2 ;; - --worker-provider) worker_provider="$2"; shift 2 ;; - --worker-model) worker_model="$2"; shift 2 ;; - --judge-provider) judge_provider="$2"; shift 2 ;; - --judge-model) judge_model="$2"; shift 2 ;; --thinking) thinking="$2"; shift 2 ;; --help|-h) usage ;; *) echo "Unknown option: $1"; usage ;; @@ -56,6 +85,7 @@ load_prompt() { local inline_value="$1" local file_value="$2" local label="$3" + local required="$4" if [[ -n "$inline_value" && -n "$file_value" ]]; then echo "Error: use either --${label} or --${label}-file, not both" @@ -76,13 +106,29 @@ load_prompt() { return 0 fi - echo "Error: either --${label} or --${label}-file is required" - usage + if [[ "$required" == "true" ]]; then + echo "Error: either --${label} or --${label}-file is required" + usage + fi + + return 1 } -plan="$(load_prompt "$plan" "$plan_file_input" "plan")" +worker_prompt="$(load_prompt "$worker_prompt_input" "$worker_prompt_file_input" "worker-prompt" "true")" [[ -z "$name" ]] && { echo "Error: --name is required"; usage; } +# Load verifier prompt if provided (optional) +has_verifier=false +verifier_prompt_text="" +if [[ -n "$verifier_prompt_input" || -n "$verifier_prompt_file_input" ]]; then + verifier_prompt_text="$(load_prompt "$verifier_prompt_input" "$verifier_prompt_file_input" "verifier-prompt" "true")" + has_verifier=true +fi + +# Parse agent specs +read -r worker_provider worker_model <<< "$(parse_agent_spec "$worker_spec" "codex" "")" +read -r verifier_provider verifier_model <<< "$(parse_agent_spec "$verifier_spec" "claude" "sonnet")" + mkdir -p "$state_root" generate_loop_id() { @@ -98,32 +144,37 @@ done mkdir -p "$state_dir" -plan_file="${state_dir}/plan.md" +prompt_file="${state_dir}/worker-prompt.md" last_reason_file="${state_dir}/last_reason.md" history_log="${state_dir}/history.log" -printf '%s\n' "$plan" > "$plan_file" +printf '%s\n' "$worker_prompt" > "$prompt_file" printf '' > "$last_reason_file" printf '' > "$history_log" +if [[ "$has_verifier" == true ]]; then + verifier_prompt_file="${state_dir}/verifier-prompt.md" + printf '%s\n' "$verifier_prompt_text" > "$verifier_prompt_file" +fi + # Build worker flags worker_flags=() if [[ "$worker_provider" == "codex" ]]; then worker_flags+=(--mode full-access --provider codex) elif [[ "$worker_provider" == "claude" ]]; then - worker_flags+=(--mode bypass) + worker_flags+=(--mode bypassPermissions) fi [[ -n "$worker_model" ]] && worker_flags+=(--model "$worker_model") [[ -n "$thinking" ]] && worker_flags+=(--thinking "$thinking") -# Build judge flags -judge_flags=() -if [[ "$judge_provider" == "codex" ]]; then - judge_flags+=(--mode full-access --provider codex) -elif [[ "$judge_provider" == "claude" ]]; then - judge_flags+=(--mode bypass) +# Build verifier flags +verifier_flags=() +if [[ "$verifier_provider" == "codex" ]]; then + verifier_flags+=(--mode full-access --provider codex) +elif [[ "$verifier_provider" == "claude" ]]; then + verifier_flags+=(--mode bypassPermissions) fi -[[ -n "$judge_model" ]] && judge_flags+=(--model "$judge_model") +[[ -n "$verifier_model" ]] && verifier_flags+=(--model "$verifier_model") # Worktree flags — passed to every paseo run call worktree_flags=() @@ -132,124 +183,164 @@ if [[ -n "$worktree" ]]; then worktree_flags+=(--worktree "$worktree" --base "$base_branch") fi -# Judge output schema -judge_schema='{"type":"object","properties":{"criteria_met":{"type":"boolean"},"reason":{"type":"string"}},"required":["criteria_met","reason"],"additionalProperties":false}' +# Structured output schema +done_schema='{"type":"object","properties":{"done":{"type":"boolean"},"reason":{"type":"string"}},"required":["done","reason"],"additionalProperties":false}' iteration=0 echo "=== Loop started: $name ===" echo " Loop ID: $loop_id" echo " State dir: $state_dir" -echo " Plan file: $plan_file" +echo " Worker prompt: $prompt_file (live-editable)" +if [[ "$has_verifier" == true ]]; then + echo " Verifier prompt: $verifier_prompt_file (live-editable)" + echo " Mode: worker + verifier" +else + echo " Mode: self-terminating worker" +fi echo " Last reason file: $last_reason_file" echo " History log: $history_log" -echo " Steering: edits to plan.md are picked up on the next worker/judge launch." -echo " Worker: $worker_provider ${worker_model:-(default)}" -echo " Judge: $judge_provider ${judge_model:-(default)}" +echo " Worker: $worker_provider/${worker_model:-(default)}" +if [[ "$has_verifier" == true ]]; then + echo " Verifier: $verifier_provider/${verifier_model:-(default)}" +fi +if [[ -n "$sleep_duration" ]]; then + echo " Sleep: $sleep_duration between iterations" +fi +if [[ "$archive" == true ]]; then + echo " Archive: agents archived after each iteration" +fi if [[ -n "$worktree" ]]; then echo " Worktree: $worktree (base: $base_branch)" fi -echo " Max iterations: $max_iterations" +if [[ $max_iterations -gt 0 ]]; then + echo " Max iterations: $max_iterations" +else + echo " Max iterations: unlimited" +fi echo "" -while [[ $iteration -lt $max_iterations ]]; do +while [[ $max_iterations -eq 0 || $iteration -lt $max_iterations ]]; do iteration=$((iteration + 1)) - echo "--- Iteration $iteration/$max_iterations ---" + if [[ $max_iterations -gt 0 ]]; then + echo "--- Iteration $iteration/$max_iterations ---" + else + echo "--- Iteration $iteration ---" + fi - if [[ ! -s "$plan_file" ]]; then - echo "Error: plan file is missing or empty: $plan_file" + if [[ ! -s "$prompt_file" ]]; then + echo "Error: worker prompt file is missing or empty: $prompt_file" exit 1 fi - current_plan="$(cat "$plan_file")" - last_reason="$(cat "$last_reason_file")" + current_prompt="$(cat "$prompt_file")" + last_reason="$(cat "$last_reason_file" 2>/dev/null || true)" - # Build worker prompt with plan in XML tags - worker_prompt="Implement the following plan exactly as specified. - - -$current_plan -" + # Build the full worker prompt + full_worker_prompt="$current_prompt" if [[ -n "$last_reason" ]]; then - worker_prompt="$worker_prompt + full_worker_prompt="$full_worker_prompt - -The previous attempt was verified and did NOT pass. Address the following issues before anything else: + +The previous iteration reported the following: $last_reason -" +" fi - # Launch worker - worker_name="${name}-work-${iteration}" - echo "Launching worker: $worker_name" - worker_id=$(paseo run -d "${worker_flags[@]}" "${worktree_flags[@]}" --name "$worker_name" "$worker_prompt" -q) - echo "Worker [$worker_name] launched. ID: $worker_id" - echo " Stream logs: paseo logs $worker_id -f" - echo " Inspect: paseo inspect $worker_id" - echo " Wait: paseo wait $worker_id" + worker_name="${name}-${iteration}" - # Wait for worker - echo "" - echo "Waiting for worker to complete..." - paseo wait "$worker_id" - echo "Worker done." + if [[ "$has_verifier" == true ]]; then + # --- Mode: worker + separate verifier --- - # Build judge prompt with plan in XML tags - judge_prompt="You are a fact checker. Your ONLY job is to verify whether the plan below was implemented correctly and completely. + # Worker runs detached (no structured output) + echo "Launching worker: $worker_name" + worker_id=$(paseo run -d "${worker_flags[@]}" "${worktree_flags[@]}" --name "$worker_name" "$full_worker_prompt" -q) + echo "Worker [$worker_name] launched. ID: $worker_id" + echo " Stream logs: paseo logs $worker_id -f" + echo " Inspect: paseo inspect $worker_id" -Inspect the codebase independently. Do NOT fix anything. Do NOT suggest fixes. Do NOT provide guidance. - - -$current_plan - - - -Verify that every step of the plan has been implemented to the letter. If the plan contains acceptance criteria, check each one individually. - -Report ONLY facts: -- criteria_met: true if the ENTIRE plan is implemented correctly, false if ANY part is missing or wrong -- reason: for each item in the plan, state whether it was implemented correctly with evidence (file contents, test output, etc.). Be specific about what passed and what failed. -" - - # Launch judge (blocking, structured output) - judge_name="${name}-verify-${iteration}" - echo "" - echo "Launching judge: $judge_name (synchronous, blocks until verdict)" - verdict=$(paseo run "${judge_flags[@]}" "${worktree_flags[@]}" --name "$judge_name" --output-schema "$judge_schema" "$judge_prompt") - echo "Verdict: $verdict" - - # Parse verdict - criteria_met=$(echo "$verdict" | jq -r '.criteria_met') - reason=$(echo "$verdict" | jq -r '.reason') - printf '[%s] iteration=%s worker_id=%s judge=%s criteria_met=%s reason=%s\n' \ - "$(date -u +'%Y-%m-%dT%H:%M:%SZ')" \ - "$iteration" \ - "$worker_id" \ - "$judge_name" \ - "$criteria_met" \ - "$(echo "$reason" | tr '\n' ' ')" >> "$history_log" - - if [[ "$criteria_met" == "true" ]]; then echo "" - echo "=== Loop complete: criteria met on iteration $iteration ===" + echo "Waiting for worker to complete..." + paseo wait "$worker_id" + echo "Worker done." + + if [[ "$archive" == true ]]; then + paseo agent archive "$worker_name" 2>/dev/null || true + fi + + # Re-read verifier prompt for live steering + current_verifier_prompt="$(cat "$verifier_prompt_file")" + + # Verifier runs synchronously with structured output + verifier_name="${name}-verify-${iteration}" + full_verifier_prompt="$current_verifier_prompt + +Respond with { \"done\": true/false, \"reason\": \"...\" }. +- done: true if the condition is met, false otherwise +- reason: explain what you found with evidence" + + echo "" + echo "Launching verifier: $verifier_name" + verdict=$(paseo run "${verifier_flags[@]}" "${worktree_flags[@]}" --name "$verifier_name" --output-schema "$done_schema" "$full_verifier_prompt") + echo "Verdict: $verdict" + + done_value=$(echo "$verdict" | jq -r '.done') + reason=$(echo "$verdict" | jq -r '.reason') + + if [[ "$archive" == true ]]; then + paseo agent archive "$verifier_name" 2>/dev/null || true + fi + + printf '[%s] iteration=%s worker=%s verifier=%s done=%s reason=%s\n' \ + "$(date -u +'%Y-%m-%dT%H:%M:%SZ')" \ + "$iteration" \ + "$worker_name" \ + "$verifier_name" \ + "$done_value" \ + "$(echo "$reason" | tr '\n' ' ')" >> "$history_log" + + else + # --- Mode: self-terminating worker --- + + # Worker runs synchronously with structured output + echo "Launching worker: $worker_name" + verdict=$(paseo run "${worker_flags[@]}" "${worktree_flags[@]}" --name "$worker_name" --output-schema "$done_schema" "$full_worker_prompt") + echo "Result: $verdict" + + done_value=$(echo "$verdict" | jq -r '.done') + reason=$(echo "$verdict" | jq -r '.reason') + + if [[ "$archive" == true ]]; then + paseo agent archive "$worker_name" 2>/dev/null || true + fi + + printf '[%s] iteration=%s worker=%s done=%s reason=%s\n' \ + "$(date -u +'%Y-%m-%dT%H:%M:%SZ')" \ + "$iteration" \ + "$worker_name" \ + "$done_value" \ + "$(echo "$reason" | tr '\n' ' ')" >> "$history_log" + fi + + if [[ "$done_value" == "true" ]]; then + echo "" + echo "=== Loop complete: done on iteration $iteration ===" echo "Reason: $reason" - echo "" - echo "Review the final worker's changes:" - echo " paseo logs $worker_id" - echo " paseo inspect $worker_id" exit 0 fi - echo "Criteria not met: $reason" - echo "Starting next iteration..." - - # Keep only the latest judge reason for the next worker iteration. + echo "Not done: $reason" printf '%s\n' "$reason" > "$last_reason_file" + if [[ -n "$sleep_duration" ]] && [[ $max_iterations -eq 0 || $iteration -lt $max_iterations ]]; then + echo "Sleeping $sleep_duration before next iteration..." + sleep "$sleep_duration" + fi + echo "" done -echo "=== Loop exhausted: $max_iterations iterations without meeting criteria ===" +echo "=== Loop exhausted: $max_iterations iterations without completing ===" exit 1