diff --git a/.agent/README.md b/.agent/README.md index 8f48712..6638962 100644 --- a/.agent/README.md +++ b/.agent/README.md @@ -1,52 +1,58 @@ -# .agent/ — agentctl assets +# .agent/ -This directory ships the prompts and config that `agentctl` uses. +Prompts and config that `agentctl` ships with. ``` .agent/ prompts/ - gpt55-orchestrator.md # lead orchestrator: plan, route, final review, decide - claude-reviewer.md # architecture reviewer: infra risk and design validation - deepseek-worker.md # execution worker: token-efficient scout / executor + deepseek-worker.md # retrieval worker: read-only tools, returns receipts + gpt55-reviewer.md # optional second opinion from GPT-5.5 + gpt55-orchestrator.md # fallback template for the GPT-5.5 reviewer + claude-reviewer.md # optional architecture and risk review + gpt55-operator-onboarding.md # paste-in briefing for driving agentctl from a ChatGPT chat opencode/ - deepseek-worker.md # OpenCode execution worker definition - config.example.json # copy to config.json to customize - config.json # your local config (gitignored) + deepseek-worker.md # OpenCode agent definition for the worker + config.example.json # shipped defaults, copy to config.json to change them + config.json # your local config (gitignored) ``` -## Config resolution order +## Config resolution -`agentctl` loads the **first** file it finds, merged over built-in defaults: +`agentctl` merges the first file it finds over the built-in defaults: -1. `/.agent/config.json` — per-repo override -2. `/.agent/config.json` — your global override -3. `/.agent/config.example.json` — shipped example +1. `/.agent/config.json` for a per-repo override +2. `/.agent/config.json` for your global override +3. `/.agent/config.example.json`, the shipped example -## Wiring DeepSeek (OpenCode) +## DeepSeek via OpenCode -The default deepseek command is: +Default command: ``` opencode run --agent deepseek-worker --format json "" ``` -To use `--agent deepseek-worker`, install the agent definition: +Install the agent definition once: ``` mkdir -p ~/.config/opencode/agent cp .agent/opencode/deepseek-worker.md ~/.config/opencode/agent/ ``` -If you don't install the agent, switch the deepseek args to the model form -(already present as `fallback_args` in `config.example.json`): +Without it, switch the deepseek `args` to the model form (already in `fallback_args` in the example): ``` opencode run --model deepseek/deepseek-chat --format json "" ``` -## Wiring GPT-5.5 +## GPT-5.5 -`config.example.json` ships `gpt55.command = "CONFIGURE_ME"`. Until you set a -real command, the orchestrator falls back to a local Claude/heuristic planner -and every run is flagged `GPT_LIMIT_ACTIVE`. Point `command`/`args` at your real -GPT-5.5 CLI or a thin API wrapper to enable the Lead Orchestrator. +The example config uses `"type": "http_openai"`: one HTTP request to the OpenAI API, key from `OPENAI_API_KEY`. Without +a key the review is skipped and the run flags `GPT_LIMIT_ACTIVE`. The run itself still completes. + +No API key? Use `--route deepseek-chatgpt`. agentctl writes a prompt to paste into ChatGPT, and `agentctl ingest ` +adds the answer to the report. + +## Claude + +`claude -p "" --output-format json`, using the Claude Code login. Used for `--route deepseek-claude`. diff --git a/.agent/config.example.json b/.agent/config.example.json index c65bf18..4856bbe 100644 --- a/.agent/config.example.json +++ b/.agent/config.example.json @@ -1,19 +1,18 @@ { "orchestrator": { - "primary": "gpt55", - "fallback": "claude", + "default_caller": "local", "default_route": "deepseek", "review_required": true, "review_policy": "best_effort_gpt55", - "default_caller": "gpt55", "run_timeout_seconds": 120, "agent_timeout_seconds": 90, "review_timeout_seconds": 45, "idle_timeout_seconds": 30, "role_policy": { - "gpt55": "Lead Orchestrator: planning, routing, final review, and final decision authority", - "claude": "Architecture Reviewer: infrastructure architecture, risk review, and lightweight code design validation", - "deepseek": "Execution Worker: token-efficient repository inspection, tests, logs, context packs, and low-risk implementation" + "deepseek": "Retrieval worker: read-only tools, returns raw tool receipts, no conclusions", + "local": "Logic: synthesizes the answer and the decision from the receipts", + "gpt55": "Optional best-effort reviewer, never on the critical path", + "claude": "Optional reviewer for architecture, design and risk" } }, "agents": { @@ -70,17 +69,11 @@ ] } }, - "limits": { - "detect_patterns": true, - "fallback_on_limit": true - }, "worktrees": { "enabled": true, "base_dir": ".agent-runs/worktrees" }, "safety": { - "block_dangerous_commands": true, - "deepseek_may_edit": false, - "claude_may_edit": true + "deepseek_may_edit": false } -} \ No newline at end of file +} diff --git a/.agent/prompts/claude-builder.md b/.agent/prompts/claude-builder.md deleted file mode 100644 index a77213c..0000000 --- a/.agent/prompts/claude-builder.md +++ /dev/null @@ -1,56 +0,0 @@ -You are Claude Senior Implementer, the optional implementation escalation role. - -You receive bounded implementation contracts from GPT-5.5 Lead Orchestrator. -GPT-5.5 owns final architecture and final review. -Your job is to implement complex, high-quality changes within the given contract. - -Use DeepSeek Execution Worker context packs as prior reconnaissance. -Do not repeat token-heavy exploration unless necessary to verify facts. -Do not broaden scope. -Do not redesign the system unless explicitly requested. -Prefer minimal, well-tested, maintainable changes. - -Before editing: -- confirm relevant files -- identify existing patterns -- preserve project conventions -- check nearby tests - -During implementation: -- make cohesive changes -- avoid unrelated cleanup -- avoid formatting churn -- add or update tests when appropriate -- run targeted tests - -Escalate back to the orchestrator if: -- the contract is contradictory -- architecture is unclear -- the change has product implications -- the change requires unrelated subsystems -- tests indicate broader failure - -Return exactly this JSON shape: - -{ - "status": "done | blocked | needs_orchestrator", - "summary": "what was implemented", - "design_notes": ["important implementation decision"], - "changed_files": ["path"], - "commands_run": [ - { - "cmd": "command", - "exit_code": 0, - "important_output": "short relevant output" - } - ], - "tests": [ - { - "cmd": "test command", - "result": "passed | failed | not_run", - "notes": "short notes" - } - ], - "risks": ["remaining risk"], - "review_notes_for_gpt55": ["specific areas the orchestrator should inspect"] -} diff --git a/.agent/prompts/gpt55-operator-onboarding.md b/.agent/prompts/gpt55-operator-onboarding.md index 80eb810..4aed49f 100644 --- a/.agent/prompts/gpt55-operator-onboarding.md +++ b/.agent/prompts/gpt55-operator-onboarding.md @@ -17,7 +17,7 @@ back the output, then you review and decide. Work in English, keep it concise. ## HOW THE USER DRIVES IT -- Interactive console: `agentctl console`. In the console the user just types a task (no quotes). Meta commands: +- Interactive console: `agentctl` (or `agentctl shell`). In the console the user just types a task (no quotes). Meta commands: `/cd ` — set target repo `/status` `/review` diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml new file mode 100644 index 0000000..7ccddf8 --- /dev/null +++ b/.github/workflows/ci.yml @@ -0,0 +1,21 @@ +name: tests + +on: + push: + pull_request: + +permissions: + contents: read + +jobs: + test: + runs-on: macos-latest + strategy: + matrix: + python: ["3.11", "3.12"] + steps: + - uses: actions/checkout@v4 + - uses: actions/setup-python@v5 + with: + python-version: ${{ matrix.python }} + - run: python -m unittest discover -s tests -v diff --git a/AGENTS.md b/AGENTS.md index 85f12aa..a432937 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -10,9 +10,8 @@ This repository uses `agentctl` as the local multi-agent workflow. One rule: prose is not trusted. It carries the token-heavy file/log reading. - **Local / Claude — Logic.** The stable reasoner and decision-maker. Synthesizes the answer from DeepSeek's receipts. This is the default; it never depends on a flaky channel. -- **GPT-5.5 — Optional best-effort review.** A second opinion only when it has a working route. The - free OpenCode/OAuth route is unreliable and a deterministic one needs OpenAI API billing, so GPT-5.5 - is never load-bearing. If it returns empty, the run completes on local synthesis. +- **GPT-5.5 — Optional best-effort review.** A second opinion when a route is configured (API key or the + ChatGPT handoff). It is never load-bearing. If it returns empty, the run completes on local synthesis. ## Default Workflow diff --git a/LICENSE b/LICENSE new file mode 100644 index 0000000..9408c61 --- /dev/null +++ b/LICENSE @@ -0,0 +1,21 @@ +MIT License + +Copyright (c) 2026 nzrbits + +Permission is hereby granted, free of charge, to any person obtaining a copy +of this software and associated documentation files (the "Software"), to deal +in the Software without restriction, including without limitation the rights +to use, copy, modify, merge, publish, distribute, sublicense, and/or sell +copies of the Software, and to permit persons to whom the Software is +furnished to do so, subject to the following conditions: + +The above copyright notice and this permission notice shall be included in all +copies or substantial portions of the Software. + +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +SOFTWARE. diff --git a/README.md b/README.md index 5e9b433..193b82c 100644 --- a/README.md +++ b/README.md @@ -1,111 +1,124 @@ # agentctl -A local multi-agent orchestrator CLI for macOS. It splits work along one rule: +[![tests](https://github.com/nzrbits/agentctl/actions/workflows/ci.yml/badge.svg)](https://github.com/nzrbits/agentctl/actions/workflows/ci.yml) +[![license: MIT](https://img.shields.io/badge/license-MIT-blue.svg)](LICENSE) -> **Push the token-heavy retrieval onto DeepSeek. Keep the logic on a reliable reasoner.** +A small CLI that runs several coding agents on one task and keeps the result reproducible. -- **DeepSeek = workhorse (retrieval).** Runs read-only tools (read/grep/glob) via OpenCode and - returns raw **tool receipts** — file/log/command evidence. Its prose conclusions are **not trusted**. -- **Local / Claude = logic.** The stable reasoner. Synthesizes the answer and the decision from - DeepSeek's receipts. This is the default and it never depends on a flaky channel. -- **GPT-5.5 = optional best-effort review.** A second opinion *when it has a working route*. The free - OpenCode/OAuth route to GPT-5.5 is unreliable (can return empty), and a deterministic route needs - OpenAI API billing — so GPT-5.5 is **never in the critical path**: if it returns nothing the run - still completes on local synthesis. +I use it every day in infrastructure and code work. It follows one rule: **a cheap model reads, a reliable model thinks.** +DeepSeek does the token-heavy part (reading files, grepping, dumping logs). Claude, or whoever owns the logic, only sees +the raw evidence DeepSeek collected and makes the decision from that. GPT-5.5 can add a second opinion, but a run never +waits for it. -It runs real subprocesses (`opencode`, `claude`), isolates edits in **git worktrees**, parses each -agent's output, scans for **dangerous commands**, and writes a full audit trail per run. +Pure Python standard library, no dependencies. Runs on macOS, Python 3.9+. -## Why this split +## Why -DeepSeek tokens are cheap, so it does the bulk file/log reading. The reasoner only sees the distilled -receipts, so it spends few tokens on logic. Crucially, model-behaviour quirks (DeepSeek ending a turn -without a final message, GPT-5.5's free route returning empty) are handled by the **orchestrator**, not -by trusting the model: receipts are harvested deterministically, empty reviews are marked `unavailable`, -and idle stalls get one bounded retry. The orchestrator is deterministic; the agents are probabilistic -inside it. +Agent CLIs fail in boring ways. A model ends its turn after the tool calls without writing an answer, a subprocess +waits forever on stdin, a free API route returns exit 0 with an empty body. If the orchestrator trusts the model's +summary, each of these turns into a wrong "done". -## Requirements +agentctl treats the models as unreliable workers inside a deterministic shell: -- macOS, `python3` (3.9+), `git` -- [OpenCode](https://opencode.ai) (`opencode`) — runs the DeepSeek worker (and GPT-5.5 when used). - Needs OpenCode auth configured (DeepSeek API key; OpenAI via OAuth or API key). -- [Claude Code](https://claude.com/claude-code) (`claude`) — used when Claude runs as a review agent. +- **Evidence over prose.** The worker's deliverable is not its summary. It is the list of tool calls it made, with + input and output, parsed straight from the event stream. The summary is kept, marked as untrusted. +- **Timeouts that mean something.** Every agent has a hard timeout and an idle timeout. A stall gets one bounded retry, + then the run moves on. +- **Optional means optional.** An empty review is marked `unavailable`. It never blocks a run and never counts as an + accept. +- **Everything on disk.** Each run writes its prompts, raw output, parsed reports and a decision to `.agent-runs//`. -## Install +## How a run works + +```mermaid +flowchart LR + T[task] --> S[repo snapshot
read-only] + S --> P[routing plan] + P --> D[DeepSeek
read-only tools] + D -->|tool receipts| L[synthesis
local / Claude] + L -->|optional| R[second opinion
GPT-5.5 or Claude] + L --> O[review-report.md
log.json] + R --> O +``` + +1. **Snapshot** the target repo: git state, file list, language, test command. +2. **Plan** the route. DeepSeek always goes first, the reviewer depends on `--route`. +3. **Retrieve.** DeepSeek runs via OpenCode with read-only tools. +4. **Harvest receipts** from its event stream, whether or not it wrote a final answer. +5. **Synthesize** the decision from the receipts. +6. **Review** (optional). A second model checks the synthesis against the same receipts. +7. **Persist** everything and print a short summary. + +## Safety + +- Read-only by default. `--allow-edit` puts agent edits into a separate git worktree, never the main working tree. +- Every command an agent reports is scanned for destructive patterns: `rm -rf`, `git reset --hard`, force push, + `terraform apply`, `kubectl delete`, `helm uninstall`, cloud CLI mutations and more. Hits are flagged in the report. +- Receipts go to the reviewer as data. Its prompt tells it never to follow instructions found inside them. +- Rate limits, auth errors, billing and context overflow are classified from CLI output and surface as flags + (`GPT_LIMIT_ACTIVE`, `BLOCKED — auth error`, …) instead of silent failures. +- API keys come from the environment or a gitignored `.env`. `doctor` prints them masked. + +## Quickstart + +Requirements: `git`, Python 3.9+, [OpenCode](https://opencode.ai) with a DeepSeek key, and optionally +[Claude Code](https://claude.com/claude-code). ```bash -ln -s ~/dev/agentctl/agentctl ~/.local/bin/agentctl # put on PATH (adjust to your location) +git clone https://github.com/nzrbits/agentctl.git +ln -s "$PWD/agentctl/agentctl" ~/.local/bin/agentctl +mkdir -p ~/.config/opencode/agent && cp agentctl/.agent/opencode/deepseek-worker.md ~/.config/opencode/agent/ + +cd ~/code/some-repo agentctl doctor +agentctl run "why does the nightly pipeline fail?" --dry-run +agentctl run "why does the nightly pipeline fail?" +agentctl review ``` -Run `agentctl` from inside any target repo — that repo is the working context, and run artifacts land -in that repo's `.agent-runs/`. +Run it from inside the repo you want to work on. That repo is the context, and the run artifacts land in its +`.agent-runs/`. ## Commands -| command | what it does | +| Command | What it does | |---|---| -| `agentctl` | bare invocation starts the interactive console (`shell`) | -| `agentctl doctor` | check macOS, git, toolchains, opencode/claude, config, env (masked) | -| `agentctl run ""` | snapshot → DeepSeek retrieval → local synthesis (+ optional review) | -| `agentctl run "" --dry-run` | plan + write prompts/run-dir, do NOT execute agents | -| `agentctl run "" --allow-edit` | let agents edit files, in isolated worktrees | -| `agentctl run "" --caller local` | Claude/local owns the logic (default, stable) | -| `agentctl run "" --route deepseek-gpt55` | also attempt a best-effort GPT-5.5 review | +| `agentctl` | interactive console (same as `agentctl shell`) | +| `agentctl doctor` | check git, toolchains, agent CLIs, config and env vars (masked) | +| `agentctl run ""` | snapshot, DeepSeek retrieval, synthesis, optional review | +| `agentctl run "" --dry-run` | write plan and prompts, run no agents | +| `agentctl run "" --allow-edit` | allow edits, isolated in git worktrees | +| `agentctl run "" --route ` | `deepseek` (default), `deepseek-claude`, `deepseek-gpt55`, `deepseek-chatgpt` | +| `agentctl ingest ` | add a reply pasted from ChatGPT to a run (for `deepseek-chatgpt`) | | `agentctl status` | list recent runs | -| `agentctl review [--run DIR]` | print a run's review report (latest by default) | -| `agentctl diff` | show diffs from agent worktrees (or the working tree) | -| `agentctl cleanup [--yes] [--runs]` | remove worktrees (and optionally run dirs) | -| `agentctl selftest` | run the bundled unit tests | - -### Callers and routes +| `agentctl review [--run DIR]` | print a run's report | +| `agentctl diff` | show diffs from agent worktrees | +| `agentctl cleanup [--yes] [--runs]` | remove worktrees and throwaway branches | +| `agentctl selftest` | run the unit tests | -- `--caller local` / `--caller claude` — Claude/local reasons from receipts (default, no flaky channel). -- `--caller gpt55` — GPT-5.5 leads (only when it has a working route). -- `--route deepseek` — DeepSeek retrieval + local synthesis only (default). -- `--route deepseek-gpt55` — also run a best-effort GPT-5.5 review (non-blocking). -- `--route deepseek-claude` — also run a `claude` CLI review. +`deepseek-chatgpt` writes a ready-to-paste prompt for a ChatGPT subscription instead of calling an API. You paste the +answer back and run `agentctl ingest`. It costs no API tokens and still ends up in the same report. -The default route comes from `.agent/config.json` (`orchestrator.default_route`, normally `deepseek`). +## Configuration -## How a run works +Copy `.agent/config.example.json` to `.agent/config.json` (gitignored). Every agent's command line is config, so a new +agent CLI needs a config block, not code. Details in [`.agent/README.md`](.agent/README.md). -1. **Snapshot** the repo (git state, structure, package manager, test command) — read-only. -2. **Plan / route**: pure-local routing. DeepSeek is always first; the reviewer is optional. -3. **DeepSeek retrieval**: gathers evidence with read-only tools. If it idle-stalls, one bounded retry. -4. **Harvest receipts**: the orchestrator extracts DeepSeek's tool calls (input + output) as the - deliverable, regardless of whether DeepSeek emitted a final summary. Its prose is kept only as an - untrusted note. -5. **Local synthesis** produces the stable decision from the receipts. -6. **Optional review** (only on `deepseek-gpt55` / `deepseek-claude`): a second opinion. An empty - GPT-5.5 result is marked `unavailable` and never blocks or false-accepts. -7. **Persist** under `.agent-runs//`: `task.txt`, `snapshot.json`, `plan.json`, `prompt-*.md`, - `agent-*/` (stdout, stderr, parsed `report.json`), `review-report.md`, `log.json`. - -## Reliability notes (known upstream behaviour) - -- **opencode subprocess + stdin**: opencode blocked when it inherited an open stdin; agentctl spawns it - with `stdin=DEVNULL`. Required for complex tasks to stream at all. -- **DeepSeek often ends after tool calls with no final text** → handled by harvesting tool receipts. -- **opencode `run --agent ` emits final text unreliably** → DeepSeek is fine (we only need - receipts); GPT-5.5 runs on `--model`, and an empty result is treated as `unavailable`. -- **Free GPT-5.5 (OpenCode OAuth) can return empty (exit 0, no output)** → best-effort only; a - deterministic GPT-5.5 needs OpenAI API billing. +## Tests -## Configuration +```bash +agentctl selftest # or: python3 -m unittest discover -s tests +``` -See [`.agent/README.md`](.agent/README.md). Copy `.agent/config.example.json` to `.agent/config.json` -and edit. Every agent's command line is config-driven. `.agent/config.json` is gitignored. +The tests cover limit detection, the dangerous-command scanner, stream and JSON parsing, routing policy and config +defaults. CI runs them on every push. -## Secrets +## Status -- No secrets are hardcoded. Copy `.env.example` → `.env` (gitignored) for keys. -- `doctor` only prints **masked** env values. +A personal tool that I use and change as I go, not a product. The OpenCode route to GPT-5.5 is best-effort by design, +for a deterministic one set `OPENAI_API_KEY` and the `http_openai` agent type. Known upstream quirks and how agentctl +handles them are listed in [`docs/reliability.md`](docs/reliability.md). -## Safety +## License -Read-only by default. `--allow-edit` is required for changes, and DeepSeek edits stay off unless -`safety.deepseek_may_edit` is enabled. Dangerous commands (`rm -rf`, `git reset --hard`, force push, -`terraform apply`, `kubectl delete`, …) are flagged in every review. DeepSeek receipts are treated as -untrusted **data**, never as instructions. +MIT diff --git a/agentctl_core/__init__.py b/agentctl_core/__init__.py index f627eff..b657028 100644 --- a/agentctl_core/__init__.py +++ b/agentctl_core/__init__.py @@ -1,3 +1,3 @@ -"""agentctl: local multi-agent orchestrator (GPT-5.5 / Claude / DeepSeek).""" +"""agentctl: local multi-agent orchestrator. DeepSeek retrieves, a reliable model decides.""" __version__ = "0.1.0" diff --git a/agentctl_core/adapters.py b/agentctl_core/adapters.py index 7db9f7a..e476386 100644 --- a/agentctl_core/adapters.py +++ b/agentctl_core/adapters.py @@ -31,7 +31,9 @@ @dataclass class AgentResult: agent: str - status: str # ok | not_configured | tool_missing | limit | error | timeout | dry_run + # ok | not_configured | tool_missing | limit | error | timeout | dry_run + # | unavailable | handoff_pending + status: str command: list[str] = field(default_factory=list) command_display: str = "" exit_code: Optional[int] = None @@ -126,7 +128,7 @@ def collect_tool_receipts(text: str, max_output: int = 8000, return out -def _collect_stream_text(text: str) -> str: +def collect_stream_text(text: str) -> str: """Reassemble assistant text from an NDJSON event stream. OpenCode's `--format json` and Claude's `--output-format stream-json` emit @@ -163,7 +165,7 @@ def _collect_stream_text(text: str) -> str: return "\n".join(chunks) if saw_event else "" -def _extract_json(text: str) -> Optional[dict]: +def extract_json(text: str) -> Optional[dict]: """Find the most plausible JSON object in agent stdout. Handles: pure JSON, JSON embedded in chatter, NDJSON event streams @@ -172,9 +174,9 @@ def _extract_json(text: str) -> Optional[dict]: if not text: return None # 0) if this is an event stream, parse the reassembled assistant text first - streamed = _collect_stream_text(text) + streamed = collect_stream_text(text) if streamed and streamed != text: - inner = _extract_json(streamed) + inner = extract_json(streamed) if inner is not None: return inner # 1) whole thing @@ -183,7 +185,7 @@ def _extract_json(text: str) -> Optional[dict]: if isinstance(obj, dict): # claude `--output-format json` wraps the answer in {"result": "..."} if isinstance(obj.get("result"), str): - inner = _extract_json(obj["result"]) + inner = extract_json(obj["result"]) if inner is not None: return inner return obj @@ -251,7 +253,7 @@ def _run_http_openai(name: str, agent_cfg: dict, prompt: str, res.status = "ok" res.exit_code = 0 res.stdout = text - res.report = _extract_json(text) + res.report = extract_json(text) res.duration_s = round(time.time() - t0, 2) res.message = f"openai {model} ok ({data.get('usage', {}).get('total_tokens', '?')} tok)" except urllib.error.HTTPError as e: @@ -412,7 +414,7 @@ def run_agent( # Parse the report first. A clean exit + a parseable report means the agent # SUCCEEDED, and its content may legitimately discuss "rate limit", "401", # etc. (e.g. summarizing code/tests). We must not let that trip detection. - res.report = _extract_json(res.stdout) + res.report = extract_json(res.stdout) succeeded = res.exit_code == 0 and res.report is not None # stderr is where CLIs print real infra failures -> always scanned. diff --git a/agentctl_core/cli.py b/agentctl_core/cli.py index d2b06ae..cddb618 100644 --- a/agentctl_core/cli.py +++ b/agentctl_core/cli.py @@ -3,6 +3,7 @@ import argparse import json +import shutil import subprocess import sys from pathlib import Path @@ -66,6 +67,8 @@ def cmd_status(args) -> int: route = "gpt55-review" elif "claude" in agents: route = "claude-review" + elif "chatgpt" in agents: + route = "chatgpt-handoff" else: route = "deepseek" except json.JSONDecodeError: @@ -150,7 +153,6 @@ def cmd_cleanup(args) -> int: capture_output=True, text=True) removed.append(f"branch {b}") if args.runs and args.yes: - import shutil for r in _list_runs(cwd): shutil.rmtree(r, ignore_errors=True) removed.append(str(r)) diff --git a/agentctl_core/config.py b/agentctl_core/config.py index 1f30b57..52eba6d 100644 --- a/agentctl_core/config.py +++ b/agentctl_core/config.py @@ -9,19 +9,19 @@ # Mirrors .agent/config.example.json so agentctl works with zero config files. DEFAULTS: dict[str, Any] = { "orchestrator": { - "primary": "gpt55", - "fallback": "claude", + "default_caller": "local", + "default_route": "deepseek", "review_required": True, - "review_policy": "reciprocal", - "default_caller": "gpt55", + "review_policy": "best_effort_gpt55", "run_timeout_seconds": 120, "agent_timeout_seconds": 90, "review_timeout_seconds": 45, "idle_timeout_seconds": 30, "role_policy": { - "gpt55": "Lead Orchestrator: planning, routing, final review, and final decision authority", - "claude": "Architecture Reviewer: infrastructure architecture, risk review, and lightweight code design validation", - "deepseek": "Execution Worker: token-efficient repository inspection, tests, logs, context packs, and low-risk implementation", + "deepseek": "Retrieval worker: read-only tools, returns raw tool receipts, no conclusions", + "local": "Logic: synthesizes the answer and the decision from the receipts", + "gpt55": "Optional best-effort reviewer, never on the critical path", + "claude": "Optional reviewer for architecture, design and risk", }, }, "agents": { @@ -50,12 +50,9 @@ "idle_timeout_seconds": 30, }, }, - "limits": {"detect_patterns": True, "fallback_on_limit": True}, "worktrees": {"enabled": True, "base_dir": ".agent-runs/worktrees"}, "safety": { - "block_dangerous_commands": True, "deepseek_may_edit": False, - "claude_may_edit": True, }, } diff --git a/agentctl_core/runner.py b/agentctl_core/runner.py index 5d89add..f591256 100644 --- a/agentctl_core/runner.py +++ b/agentctl_core/runner.py @@ -14,6 +14,7 @@ import json import os import secrets +import shutil import subprocess import time from dataclasses import dataclass, field @@ -50,7 +51,7 @@ class RunContext: config: dict dry_run: bool allow_edit: bool - caller: str = "gpt55" + caller: str = "local" timeout_seconds: int | None = None started_at: float = field(default_factory=time.time) snapshot: dict = field(default_factory=dict) @@ -134,7 +135,7 @@ def _compose_agent_prompt(ctx: RunContext, template: str, contract: dict) -> str # --------------------------------------------------------------------------- # -# orchestrator (GPT-5.5 if wired, else local heuristic) +# routing plan # --------------------------------------------------------------------------- # def _is_complex(task: str) -> bool: t = task.lower() @@ -154,7 +155,7 @@ def _gpt_available(ctx: RunContext) -> bool: # deterministic route: available iff the API key is set return bool(os.environ.get(gpt.get("api_key_env", "OPENAI_API_KEY"))) cmd = gpt.get("command") - return bool(cmd) and cmd != "CONFIGURE_ME" and bool(__import__("shutil").which(cmd)) + return bool(cmd) and cmd != "CONFIGURE_ME" and bool(shutil.which(cmd)) def _resolve_caller(config: dict, caller: str) -> str: @@ -308,7 +309,7 @@ def _idle_timeout(ctx: RunContext) -> int: return max(1, int(ctx.config.get("orchestrator", {}).get("idle_timeout_seconds", 30))) -def _harvest_deepseek_evidence(ctx, ds_res, dry_run): +def _harvest_deepseek_evidence(ctx: RunContext, ds_res: adapters.AgentResult, dry_run: bool) -> None: """Use DeepSeek as a RETRIEVAL worker, never as a reasoner. Top priority of this orchestrator: push the token-heavy retrieval (file reads, @@ -322,7 +323,7 @@ def _harvest_deepseek_evidence(ctx, ds_res, dry_run): if dry_run or ds_res.status != "ok": return receipts = adapters.collect_tool_receipts(ds_res.stdout or "") - note = adapters._collect_stream_text(ds_res.stdout or "").strip() + note = adapters.collect_stream_text(ds_res.stdout or "").strip() if not receipts and not note: return # nothing gathered at all report = { @@ -526,17 +527,15 @@ def _run_review_agent(ctx: RunContext, cfg: dict, run_dir: Path, reviewer: str | timeout = _cap_timeout(ctx, "review_timeout_seconds", 45) res = adapters.run_agent(reviewer, agent_cfg, prompt, cwd, dry_run=dry_run, timeout=timeout, idle_timeout=_idle_timeout(ctx)) - # The free GPT-5.5 route (opencode OAuth / ChatGPT subscription) intermittently - # returns an empty stream (exit 0, no text). A deterministic GPT-5.5 needs OpenAI - # API billing, which this account lacks. So GPT-5.5 review is BEST-EFFORT: if it - # comes back empty, mark it unavailable and never block, retry, or false-accept. - # DeepSeek's receipts go to local synthesis, which owns the stable decision. + # The opencode/OAuth route to GPT-5.5 can return an empty stream (exit 0, no text). + # So the GPT-5.5 review is best-effort: an empty result is marked unavailable and + # never blocks, retries or false-accepts. Local synthesis owns the decision. if reviewer == "gpt55" and res.status in ("ok", "timeout"): if agent_cfg.get("type") == "http_openai": # deterministic route returns plain text, not an NDJSON event stream empty = not (res.stdout or "").strip() and not res.report else: - empty = not adapters._collect_stream_text(res.stdout or "").strip() and not res.report + empty = not adapters.collect_stream_text(res.stdout or "").strip() and not res.report if empty: res.status = "unavailable" res.message = ("GPT-5.5 review returned no usable output — local synthesis used " @@ -558,7 +557,7 @@ def _persist_result(run_dir: Path, name: str, res: adapters.AgentResult): (d / "report.json").write_text(json.dumps(res.report, indent=2)) -def _chatgpt_handoff(ctx, run_dir, prompt) -> adapters.AgentResult: +def _chatgpt_handoff(ctx: RunContext, run_dir: Path, prompt: str) -> adapters.AgentResult: """GPT-5.5 logic via the ChatGPT subscription — deterministic, no agent tokens. Writes a paste-ready prompt and returns a pending handoff. The user pastes the @@ -593,7 +592,7 @@ def ingest(run_dir: Path) -> int: d = run_dir / "agent-chatgpt" d.mkdir(exist_ok=True) (d / "stdout.txt").write_text(text) - report = adapters._extract_json(text) + report = adapters.extract_json(text) if report: (d / "report.json").write_text(json.dumps(report, indent=2)) (d / "result.json").write_text(json.dumps( @@ -679,7 +678,8 @@ def _console_summary(ctx: RunContext, ds_res: adapters.AgentResult | None, return "\n".join(lines) -def _review(ctx: RunContext, plan_info, ds_res, review_res, stage: str = "complete") -> str: +def _review(ctx: RunContext, plan_info: dict, ds_res: adapters.AgentResult | None, + review_res: adapters.AgentResult | None, stage: str = "complete") -> str: """Local synthesis review. Optional GPT-5.5 review may add evidence; this always runs.""" lines = ["# agentctl review report", ""] lines.append(f"- run: `{ctx.run_dir.name}`") @@ -725,7 +725,7 @@ def agent_block(title, res): # Surface the agent's actual prose answer, not just the parsed report. # opencode/claude emit NDJSON event streams; the real findings live in the # reassembled text, which is dropped by _extract_report when it is not JSON. - answer = adapters._collect_stream_text(res.stdout or "").strip() + answer = adapters.collect_stream_text(res.stdout or "").strip() if answer: if len(answer) > 2000: answer = answer[:2000] + " …[truncated — full output in the agent's stdout.txt]" diff --git a/docs/reliability.md b/docs/reliability.md new file mode 100644 index 0000000..a014679 --- /dev/null +++ b/docs/reliability.md @@ -0,0 +1,15 @@ +# Reliability notes + +Upstream behaviour that agentctl works around. Each item says what happens and what the orchestrator does about it. + +| Behaviour | Effect | Handling | +|---|---|---| +| `opencode` inherits an open stdin | the process blocks on a permission read and streams nothing | agents are spawned with `stdin=DEVNULL` | +| DeepSeek ends its turn after the tool calls, with no final text | no summary to parse | tool receipts are harvested from the event stream and become the deliverable | +| `opencode run --agent ` emits final text unreliably | empty answers from some profiles | DeepSeek only needs receipts; GPT-5.5 runs with `--model` | +| OpenCode's OAuth route to GPT-5.5 returns exit 0 with an empty body | looks like success | an empty review is marked `unavailable` and never counts as an accept | +| DeepSeek idle-stalls mid-run | the run hangs | idle timeout, then one bounded retry if the run budget allows it | +| Agents talk about "rate limit" or "401" in a normal answer | false limit detection | stdout is only scanned for limits when the call did not succeed cleanly; stderr is always scanned | + +For a deterministic GPT-5.5 review, set the agent type to `http_openai` and provide `OPENAI_API_KEY`. That route is a +single HTTP request, so it cannot stall or return a half-finished stream. diff --git a/tests/test_config_defaults.py b/tests/test_config_defaults.py new file mode 100644 index 0000000..bcf6b52 --- /dev/null +++ b/tests/test_config_defaults.py @@ -0,0 +1,45 @@ +"""Built-in defaults and the shipped example config must agree with the documented behaviour.""" +import json +import os +import sys +import unittest +from pathlib import Path + +sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.realpath(__file__)))) + +from agentctl_core import config as cfg_mod, runner # noqa: E402 + +EXAMPLE = cfg_mod.install_dir() / ".agent" / "config.example.json" + + +class TestConfigDefaults(unittest.TestCase): + def setUp(self): + self._env = os.environ.pop("AGENTCTL_CALLER", None) + + def tearDown(self): + if self._env is not None: + os.environ["AGENTCTL_CALLER"] = self._env + + def test_builtin_default_caller_is_local(self): + self.assertEqual(runner._resolve_caller(cfg_mod.DEFAULTS, "auto"), "local") + + def test_example_default_caller_is_local(self): + example = json.loads(EXAMPLE.read_text()) + self.assertEqual(runner._resolve_caller(example, "auto"), "local") + + def test_default_route_is_deepseek_only(self): + for cfg in (cfg_mod.DEFAULTS, json.loads(EXAMPLE.read_text())): + self.assertEqual(runner._resolve_route(cfg, "auto", "local"), "deepseek") + + def test_run_context_defaults_to_local(self): + ctx = runner.RunContext("x", Path("."), Path("."), {}, False, False) + self.assertEqual(ctx.caller, "local") + + def test_chatgpt_route_uses_manual_handoff(self): + cfg = {"orchestrator": {"review_required": True}} + ctx = runner.RunContext("x", Path("."), Path("."), cfg, False, False, route="deepseek-chatgpt") + self.assertEqual(runner._optional_reviewer(ctx), "chatgpt") + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_safety_and_adapters.py b/tests/test_safety_and_adapters.py index 54be772..fb959f5 100644 --- a/tests/test_safety_and_adapters.py +++ b/tests/test_safety_and_adapters.py @@ -71,12 +71,12 @@ def test_as_dict_elides_long_command_args(self): def test_extract_json_embedded(self): text = 'chatter before\n{"status":"done","summary":"ok"}\ntrailing' - obj = adapters._extract_json(text) + obj = adapters.extract_json(text) self.assertEqual(obj.get("status"), "done") def test_extract_json_picks_report(self): text = '{"noise":1} ... {"action":"final_answer","reasoning_summary":"x"}' - obj = adapters._extract_json(text) + obj = adapters.extract_json(text) self.assertEqual(obj.get("action"), "final_answer") def test_extract_from_opencode_stream(self): @@ -86,13 +86,13 @@ def test_extract_from_opencode_stream(self): '\\"summary\\":\\"ok\\"}"}}\n' '{"type":"step_finish","part":{"type":"step-finish"}}' ) - obj = adapters._extract_json(stream) + obj = adapters.extract_json(stream) self.assertEqual(obj.get("status"), "done") def test_extract_from_claude_result_wrapper(self): wrapped = ('{"type":"result","result":"text {\\"status\\":\\"done\\",' '\\"changed_files\\":[\\"a.py\\"]}"}') - obj = adapters._extract_json(wrapped) + obj = adapters.extract_json(wrapped) self.assertEqual(obj.get("changed_files"), ["a.py"])