diff --git a/.agents/plugins/marketplace.json b/.agents/plugins/marketplace.json
new file mode 100644
index 0000000..1d8194d
--- /dev/null
+++ b/.agents/plugins/marketplace.json
@@ -0,0 +1,20 @@
+{
+ "name": "planman",
+ "interface": {
+ "displayName": "Planman"
+ },
+ "plugins": [
+ {
+ "name": "planman",
+ "source": {
+ "source": "local",
+ "path": "./plugins/planman"
+ },
+ "policy": {
+ "installation": "AVAILABLE",
+ "authentication": "ON_INSTALL"
+ },
+ "category": "Developer Tools"
+ }
+ ]
+}
diff --git a/.claude-plugin/marketplace.json b/.claude-plugin/marketplace.json
index 07a80f3..1491b2e 100644
--- a/.claude-plugin/marketplace.json
+++ b/.claude-plugin/marketplace.json
@@ -4,13 +4,13 @@
"name": "rusdyn"
},
"metadata": {
- "description": "Evaluates Claude Code plans using OpenAI Codex CLI",
+ "description": "Evaluates AI coding-agent plans using another agent",
"version": "0.4.9"
},
"plugins": [
{
"name": "planman",
- "description": "Evaluates Claude Code plans using OpenAI Codex CLI",
+ "description": "Evaluates AI coding-agent plans using another agent",
"version": "0.4.9",
"source": "./",
"author": {
diff --git a/.claude-plugin/plugin.json b/.claude-plugin/plugin.json
index c6b42d9..914f9cc 100644
--- a/.claude-plugin/plugin.json
+++ b/.claude-plugin/plugin.json
@@ -1,10 +1,10 @@
{
"name": "planman",
"version": "0.4.9",
- "description": "Evaluates Claude Code plans using OpenAI Codex CLI and rejects low-scoring plans with feedback",
+ "description": "Evaluates AI coding-agent plans using another agent and rejects low-scoring plans with feedback",
"author": { "name": "dyn" },
"license": "MIT",
- "keywords": ["plan", "evaluation", "quality-gate", "codex"],
+ "keywords": ["plan", "evaluation", "quality-gate", "codex", "claude"],
"homepage": "https://github.com/RusDyn/planman",
"repository": "https://github.com/RusDyn/planman"
}
diff --git a/.codex-plugin/plugin.json b/.codex-plugin/plugin.json
new file mode 100644
index 0000000..4e0ae21
--- /dev/null
+++ b/.codex-plugin/plugin.json
@@ -0,0 +1,24 @@
+{
+ "name": "planman",
+ "version": "0.4.9",
+ "description": "Evaluates AI coding-agent plans using another agent",
+ "author": { "name": "dyn" },
+ "license": "MIT",
+ "keywords": ["plan", "evaluation", "quality-gate", "codex", "claude"],
+ "homepage": "https://github.com/RusDyn/planman",
+ "repository": "https://github.com/RusDyn/planman",
+ "interface": {
+ "displayName": "Planman",
+ "shortDescription": "Evaluates coding-agent plans before execution",
+ "longDescription": "Planman evaluates implementation plans from Codex or Claude Code with another local coding-agent CLI and blocks weak plans with actionable feedback.",
+ "developerName": "dyn",
+ "category": "Developer Tools",
+ "capabilities": ["Read", "Interactive"],
+ "defaultPrompt": [
+ "Evaluate my implementation plan",
+ "Stress-test this coding plan",
+ "Show planman status"
+ ],
+ "brandColor": "#2563EB"
+ }
+}
diff --git a/.gitignore b/.gitignore
index 5c5981b..1e35f16 100644
--- a/.gitignore
+++ b/.gitignore
@@ -8,16 +8,16 @@ build/
.env
# Claude Code local state
-.claude/planman.log
-.claude/plans/
-.claude/settings.local.json
+.claude/
# Playwright MCP auto-generated logs
.playwright-mcp/
# Benchmark data (large/ephemeral)
-benchmark/exercises/
-benchmark/results/
-benchmark/swebench/results/
logs/
planman-*.json
+submission
+benchmark/
+test_*.log
+.omc/
+sb-cli-reports/
diff --git a/README.md b/README.md
index e00771f..e001c30 100644
--- a/README.md
+++ b/README.md
@@ -7,16 +7,16 @@
-
+
-A quality gate for AI-generated plans. This [Claude Code](https://docs.anthropic.com/en/docs/claude-code) plugin sends implementation plans to [OpenAI Codex CLI](https://github.com/openai/codex) for independent scoring, automatically rejecting low-quality plans with actionable feedback — so Claude iterates before you review.
+A quality gate for AI-generated plans. This plugin evaluates implementation plans from Claude Code or Codex with another coding agent, automatically rejecting low-quality plans with actionable feedback before you review.
> *"Every plan deserves a second opinion."*
-When Claude presents an implementation plan, planman intercepts it, sends it to [OpenAI Codex CLI](https://github.com/openai/codex) for scoring, and rejects low-scoring plans with actionable feedback. Claude revises and re-presents. After a configurable number of rounds, you decide.
+When an agent presents an implementation plan, planman intercepts it, sends it to the configured evaluator for scoring, and rejects low-scoring plans with actionable feedback. By default, Claude Code plans are reviewed by Codex and Codex plans are reviewed by Claude Code. After a configurable number of rounds, you decide.
-**No API keys required** — uses your ChatGPT subscription via the `codex` CLI.
+**No API keys required for Codex review** — uses your ChatGPT subscription via the `codex` CLI. Claude review uses your local Claude Code CLI authentication.
## How It Works
@@ -26,7 +26,7 @@ Planman uses **two hooks** in a plan-mode-only architecture:
PostToolUse(Write) — records plan file path when Claude writes to .claude/plans/
│
▼
-PreToolUse(ExitPlanMode) — evaluates plan via codex when Claude exits plan mode
+PreToolUse(ExitPlanMode) — evaluates plan via configured evaluator when Claude exits plan mode
│
├── Round 1: mandatory review — plan always gets scored feedback
│ │
@@ -41,6 +41,7 @@ PreToolUse(ExitPlanMode) — evaluates plan via codex when Claude exits plan mod
```
**Deterministic:** Files in `.claude/plans/` are always treated as plans — no LLM-based plan detection.
+For Codex-hosted sessions, planman uses Codex lifecycle hooks and evaluates only clearly plan-like content; if no plan is found, it fails open.
## Quick Start
@@ -54,9 +55,14 @@ PreToolUse(ExitPlanMode) — evaluates plan via codex when Claude exits plan mod
/plugin marketplace add RusDyn/planman
/plugin install planman@planman
```
-4. Restart Claude Code
+4. Or add planman to Codex:
+ ```bash
+ codex plugin marketplace add RusDyn/planman
+ codex plugin add planman@planman
+ ```
+5. Restart the host agent
-That's it. The next time Claude exits plan mode, planman evaluates the plan and blocks with feedback if the score is below threshold (default 7/10). Run `/planman:init` to customize settings.
+That's it. The next time the host agent presents a detected plan, planman evaluates it and blocks with feedback if the score is below threshold (default 7/10). Run `/planman:init` to customize settings.
> *"You're four steps from better plans."*
@@ -77,6 +83,8 @@ That's it. The next time Claude exits plan mode, planman evaluates the plan and
### From GitHub
+Claude Code:
+
```bash
# Add the marketplace
/plugin marketplace add RusDyn/planman
@@ -85,8 +93,17 @@ That's it. The next time Claude exits plan mode, planman evaluates the plan and
/plugin install planman@planman
```
+Codex:
+
+```bash
+codex plugin marketplace add RusDyn/planman
+codex plugin add planman@planman
+```
+
### Local Development
+Claude Code:
+
```bash
# Add the local directory as a marketplace
/plugin marketplace add /path/to/planman
@@ -95,28 +112,38 @@ That's it. The next time Claude exits plan mode, planman evaluates the plan and
/plugin install planman@planman
```
-Then restart Claude Code.
+Codex:
+
+```bash
+codex plugin marketplace add /path/to/planman
+codex plugin add planman@planman
+```
+
+Then restart the host agent.
## Configuration
-Settings are loaded from env vars (highest priority) or `.claude/planman.jsonc`:
+Settings are loaded from env vars (highest priority), `.planman.jsonc`, or legacy `.claude/planman.jsonc`:
| Setting | Env Var | Default | Description |
|---------|---------|---------|-------------|
| `threshold` | `PLANMAN_THRESHOLD` | `7` | Minimum score (0-10) to pass; 0 = pass all |
| `max_rounds` | `PLANMAN_MAX_ROUNDS` | `3` | Evaluation rounds before you decide (1-100) |
-| `model` | `PLANMAN_MODEL` | *(codex default)* | Override Codex model (`-m` flag) |
-| `fail_open` | `PLANMAN_FAIL_OPEN` | `true` | Pass through if Codex fails |
+| `model` | `PLANMAN_MODEL` | *(evaluator default)* | Override evaluator model |
+| `evaluator` | `PLANMAN_EVALUATOR` | `auto` | `auto`, `codex`, or `claude`; auto picks the other agent |
+| `codex_bin` | `PLANMAN_CODEX_BIN` | `codex` | Codex CLI binary/path |
+| `claude_bin` | `PLANMAN_CLAUDE_BIN` | `claude` | Claude Code CLI binary/path |
+| `fail_open` | `PLANMAN_FAIL_OPEN` | `true` | Pass through if evaluator fails |
| `enabled` | `PLANMAN_ENABLED` | `true` | Master switch |
| `custom_rubric` | `PLANMAN_RUBRIC` | *(built-in)* | Custom evaluation rubric |
| `verbose` | `PLANMAN_VERBOSE` | `false` | Debug output to stderr |
-| `source_verify` | `PLANMAN_SOURCE_VERIFY` | `true` | Codex verifies plan against actual source files |
+| `source_verify` | `PLANMAN_SOURCE_VERIFY` | `true` | Evaluator verifies plan against actual source files |
| `stress_test` | `PLANMAN_STRESS_TEST` | `false` | Stress-test rounds (`false`/`true`/number N) |
| `context` | `PLANMAN_CONTEXT` | *(empty)* | Project context injected into evaluation prompt |
-`stress_test` accepts `false` (off), `true` (1 round), or a number N (N stress-test rounds). Stress-test rounds skip Codex and auto-reject with the stress-test prompt. Codex evaluation begins at round N+1.
+`stress_test` accepts `false` (off), `true` (1 round), or a number N (N stress-test rounds). Stress-test rounds skip the external evaluator and auto-reject with the stress-test prompt. External evaluation begins at round N+1.
-Run `/planman:init` to generate `.claude/planman.jsonc` with all settings and inline documentation.
+Run `/planman:init` in Claude Code to generate `.planman.jsonc` with all settings and inline documentation. In Codex, create the same file manually or use `PLANMAN_*` environment variables.
## Scoring Rubric
@@ -140,7 +167,7 @@ Override the built-in rubric for domain-specific evaluation:
export PLANMAN_RUBRIC="Score the plan focusing on security implications, test coverage, and backwards compatibility. Be strict about migration safety."
```
-Or in `.claude/planman.jsonc`:
+Or in `.planman.jsonc`:
```json
{
@@ -150,13 +177,13 @@ Or in `.claude/planman.jsonc`:
## Commands
-All commands use the `planman:` namespace prefix:
+Claude Code slash commands use the `planman:` namespace prefix. Codex uses the installed Stop hook plus `.planman.jsonc` or `PLANMAN_*` environment variables.
| Command | Description |
|---------|-------------|
-| `/planman:status` | Show status, codex version, effective config |
+| `/planman:status` | Show status, evaluator versions, effective config |
| `/planman:help` | Full usage guide |
-| `/planman:init` | Create `.claude/planman.jsonc` with all defaults |
+| `/planman:init` | Create `.planman.jsonc` with all defaults |
| `/planman:clear` | Clear session state (reset evaluation rounds) |
## Multi-Round Behavior
@@ -168,10 +195,11 @@ All commands use the `planman:` namespace prefix:
## Zero Friction Design
-- **No API keys** — uses ChatGPT subscription via `codex` CLI
+- **Cross-agent by default** — Claude plans use Codex review; Codex plans use Claude review
+- **No API keys for Codex review** — uses ChatGPT subscription via `codex` CLI
- **No pip dependencies** — stdlib only (Python 3.8+)
-- **Fail-open by default** — Codex errors never block your workflow
-- **Auto-detect codex** — if `codex` isn't installed, hook silently passes through
+- **Fail-open by default** — evaluator errors never block your workflow
+- **Auto-detect evaluator** — if the resolved evaluator CLI isn't installed, hook silently passes through
- **Plan-mode only** — deterministic detection via `.claude/plans/` path
## State Files
@@ -183,19 +211,22 @@ Session state is stored in the system temp directory (run `python3 -c "import te
## Plugin Structure
-- `.claude-plugin/marketplace.json` — marketplace registry (used by `/plugin marketplace add`)
-- `.claude-plugin/plugin.json` — plugin definition (hooks, commands, schemas)
-- `hooks/hooks.json` — two hooks: PostToolUse(Write) + PreToolUse(ExitPlanMode)
+- `.claude-plugin/marketplace.json` — Claude Code marketplace registry
+- `.claude-plugin/plugin.json` — Claude Code plugin definition
+- `.codex-plugin/plugin.json` — Codex plugin definition
+- `.agents/plugins/marketplace.json` — Codex marketplace registry
+- `hooks/hooks.json` — Claude Code hooks: PostToolUse(Write) + PreToolUse(ExitPlanMode)
+- `hooks.json` — Codex Stop hook
- `scripts/` — hook implementation (Python, stdlib only)
- `post_tool_hook.py` — records plan file path
- - `pre_exit_plan_hook.py` — evaluates plan via codex
+ - `pre_exit_plan_hook.py` — evaluates plan via configured evaluator
- `hook_utils.py` — shared evaluation logic
- - `evaluator.py` — codex subprocess wrapper
+ - `evaluator.py` — evaluator subprocess wrappers
- `state.py` — multi-round session state
- `config.py` — configuration loader
- `clear_state.py` — session cleanup utility
-- `schemas/` — JSON output schema for codex structured output
-- `commands/` — slash commands (`/planman:status`, `/planman:help`, `/planman:init`, `/planman:clear`)
+- `schemas/` — JSON output schema for structured evaluator output
+- `commands/` — Claude Code slash commands (`/planman:status`, `/planman:help`, `/planman:init`, `/planman:clear`)
## Uninstalling
@@ -203,14 +234,14 @@ Session state is stored in the system temp directory (run `python3 -c "import te
/plugin uninstall planman@planman
```
-This removes planman's hooks and commands. Your `.claude/planman.jsonc` config file is preserved — delete it manually if no longer needed.
+This removes planman's hooks and commands. Your `.planman.jsonc` or legacy `.claude/planman.jsonc` config file is preserved — delete it manually if no longer needed.
## Troubleshooting
### "Nothing happens" when Claude presents a plan
1. **Check planman is installed**: Run `/planman:status` — it should show status and config
-2. **Enable verbose mode**: Set `PLANMAN_VERBOSE=true` in your env or `.claude/planman.jsonc`
+2. **Enable verbose mode**: Set `PLANMAN_VERBOSE=true` in your env or `.planman.jsonc`
3. **Check threshold**: A threshold of `0` disables evaluation. Set `PLANMAN_THRESHOLD=1` for testing
### I set `PLANMAN_VERBOSE=true` but see no output
diff --git a/commands/clear.md b/commands/clear.md
index c960c16..1e060ae 100644
--- a/commands/clear.md
+++ b/commands/clear.md
@@ -2,7 +2,7 @@
description: Clear planman session state (reset evaluation rounds)
---
Clear all planman session state files by running:
-`python3 ${CLAUDE_PLUGIN_ROOT}/scripts/clear_state.py`
+`python3 -c "import os, subprocess; root = os.environ.get('CLAUDE_PLUGIN_ROOT') or os.environ.get('CODEX_PLUGIN_ROOT') or os.getcwd(); subprocess.run(['python3', os.path.join(root, 'scripts', 'clear_state.py')])"`
## What Gets Cleared
diff --git a/commands/help.md b/commands/help.md
index 94daec7..c037fc6 100644
--- a/commands/help.md
+++ b/commands/help.md
@@ -10,43 +10,46 @@ Show the user the following help information:
## What is Planman?
-Planman is a Claude Code plugin that evaluates your implementation plans before you approve them. It uses **OpenAI Codex CLI** as an external evaluator — when Claude exits plan mode, Planman intercepts the ExitPlanMode call, sends the plan to Codex for scoring, and rejects low-scoring plans with actionable feedback. Claude then revises and re-presents. After a configurable number of rounds, you decide.
+Planman evaluates AI coding-agent implementation plans before you approve them. By default it uses cross-agent review: Claude Code plans are reviewed by Codex, and Codex plans are reviewed by Claude Code. Low-scoring plans are rejected with actionable feedback. After a configurable number of rounds, you decide.
## Prerequisites
-1. **OpenAI Codex CLI**: `npm install -g @openai/codex`
-2. **ChatGPT subscription** (Plus, Pro, or Team)
-3. **Login once**: Run `codex` and authenticate via browser
+1. **OpenAI Codex CLI** for Codex review: `npm install -g @openai/codex`
+2. **Claude Code CLI** for Claude review
+3. Authenticate the evaluator CLI you intend to use
-No API keys needed — Codex uses your ChatGPT subscription.
+No API keys are needed for Codex review — Codex uses your ChatGPT subscription. Claude review uses your local Claude Code authentication.
## How It Works
-1. Claude writes a plan to `.claude/plans/` (PostToolUse(Write) records the path)
-2. Claude calls ExitPlanMode to present the plan
-3. Planman intercepts ExitPlanMode and sends the plan to `codex exec` for evaluation
-4. Codex scores the plan on 5 criteria (completeness, correctness, sequencing, risk awareness, clarity)
+1. The host agent produces a plan
+2. Planman detects the host automatically
+3. Planman sends the plan to the resolved evaluator (`auto`, `codex`, or `claude`)
+4. The evaluator scores the plan on 5 criteria (completeness, correctness, sequencing, risk awareness, clarity)
5. **Round 1**: Mandatory review — plan always gets scored feedback, regardless of score
-6. **Round 2+**: Score >= threshold (default 7/10) → plan passes. Below → rejected with feedback, Claude revises
+6. **Round 2+**: Score >= threshold (default 7/10) → plan passes. Below → rejected with feedback, the host agent revises
7. After max rounds (default 3): You decide whether to proceed
-**Plan-mode only.** Files in `.claude/plans/` are deterministically treated as plans — no LLM-based detection.
+Claude Code files in `.claude/plans/` are deterministically treated as plans. Codex-hosted sessions evaluate only clearly plan-like content and fail open if no plan is found.
## Configuration
-Settings are loaded from env vars (highest priority) or `.claude/planman.jsonc`:
+Settings are loaded from env vars (highest priority), `.planman.jsonc`, or legacy `.claude/planman.jsonc`:
| Setting | Env Var | Default | Description |
|---------|---------|---------|-------------|
| `threshold` | `PLANMAN_THRESHOLD` | `7` | Minimum score (1-10) to pass |
| `max_rounds` | `PLANMAN_MAX_ROUNDS` | `3` | Rounds before you decide |
| `min_rounds` | `PLANMAN_MIN_ROUNDS` | `0` | Minimum rounds before plan can pass |
-| `model` | `PLANMAN_MODEL` | *(codex default)* | Override Codex model |
-| `fail_open` | `PLANMAN_FAIL_OPEN` | `true` | Pass if Codex fails |
+| `model` | `PLANMAN_MODEL` | *(evaluator default)* | Override evaluator model |
+| `evaluator` | `PLANMAN_EVALUATOR` | `auto` | `auto`, `codex`, or `claude` |
+| `codex_bin` | `PLANMAN_CODEX_BIN` | `codex` | Codex CLI binary/path |
+| `claude_bin` | `PLANMAN_CLAUDE_BIN` | `claude` | Claude Code CLI binary/path |
+| `fail_open` | `PLANMAN_FAIL_OPEN` | `true` | Pass if evaluator fails |
| `enabled` | `PLANMAN_ENABLED` | `true` | Master switch |
| `custom_rubric` | `PLANMAN_RUBRIC` | *(built-in)* | Custom evaluation rubric |
| `verbose` | `PLANMAN_VERBOSE` | `false` | Debug output to stderr |
-| `source_verify` | `PLANMAN_SOURCE_VERIFY` | `true` | Codex verifies plan against actual source files |
+| `source_verify` | `PLANMAN_SOURCE_VERIFY` | `true` | Evaluator verifies plan against actual source files |
| `stress_test` | `PLANMAN_STRESS_TEST` | `false` | Stress-test rounds (`false`/`true`/number N) |
| `context` | `PLANMAN_CONTEXT` | *(empty)* | Project context for evaluator |
| `plan_dirs` | `PLANMAN_PLAN_DIRS` | `[]` | Additional plan directories to monitor |
@@ -55,9 +58,9 @@ Settings are loaded from env vars (highest priority) or `.claude/planman.jsonc`:
### Quick Start
-Run `/planman:init` to create `.claude/planman.jsonc` with all settings and descriptions.
+Run `/planman:init` in Claude Code to create `.planman.jsonc` with all settings and descriptions. In Codex, create the same file manually or use `PLANMAN_*` environment variables.
-### Example `.claude/planman.jsonc`
+### Example `.planman.jsonc`
```jsonc
{
@@ -68,7 +71,7 @@ Run `/planman:init` to create `.claude/planman.jsonc` with all settings and desc
}
```
-`stress_test` accepts `false` (off), `true` (1 round), or a number N (N stress-test rounds). Stress-test rounds skip Codex and auto-reject with the stress-test prompt. Codex evaluation begins at round N+1.
+`stress_test` accepts `false` (off), `true` (1 round), or a number N (N stress-test rounds). Stress-test rounds skip the evaluator and auto-reject with the stress-test prompt. External evaluation begins at round N+1.
## Tips
@@ -101,7 +104,9 @@ Env vars accept comma-separated values or JSON arrays: `PLANMAN_EXEC_PATTERNS='[
## Commands
+These are Claude Code slash commands. Codex uses the installed Stop hook plus `.planman.jsonc` or `PLANMAN_*` environment variables.
+
- `/planman:status` — Show status and effective configuration
- `/planman:help` — This help page
-- `/planman:init` — Create `.claude/planman.jsonc` with all defaults
+- `/planman:init` — Create `.planman.jsonc` with all defaults
- `/planman:clear` — Clear session state (reset evaluation rounds)
diff --git a/commands/init.md b/commands/init.md
index f57ddaa..5b5e650 100644
--- a/commands/init.md
+++ b/commands/init.md
@@ -1,10 +1,10 @@
---
-description: Create .claude/planman.jsonc with commented defaults
+description: Create .planman.jsonc with commented defaults
---
# Planman Init
-Create a starter `.claude/planman.jsonc` with all available settings and descriptions.
+Create a starter `.planman.jsonc` with all available settings and descriptions.
## Instructions
@@ -13,10 +13,13 @@ Run this command:
```bash
python3 -c "
import sys, os
-sys.path.insert(0, '${CLAUDE_PLUGIN_ROOT}/scripts')
+plugin_root = os.environ.get('CLAUDE_PLUGIN_ROOT') or os.environ.get('CODEX_PLUGIN_ROOT') or os.getcwd()
+sys.path.insert(0, os.path.join(plugin_root, 'scripts'))
-jsonc_path = os.path.join('.claude', 'planman.jsonc')
-json_path = os.path.join('.claude', 'planman.json')
+jsonc_path = '.planman.jsonc'
+json_path = '.planman.json'
+legacy_jsonc_path = os.path.join('.claude', 'planman.jsonc')
+legacy_json_path = os.path.join('.claude', 'planman.json')
if os.path.exists(jsonc_path):
print(f'Already exists: {jsonc_path}')
print('Delete it first if you want to regenerate.')
@@ -25,6 +28,10 @@ if os.path.exists(json_path):
print(f'Found existing {json_path} — rename or delete it first.')
print('planman now uses .jsonc (supports // comments).')
sys.exit(0)
+if os.path.exists(legacy_jsonc_path) or os.path.exists(legacy_json_path):
+ print('Found existing legacy .claude/planman config.')
+ print('Keeping it. Delete or move it first if you want to regenerate .planman.jsonc.')
+ sys.exit(0)
content = '''// Planman configuration
// Docs: /planman:help | All settings are optional — defaults shown below
@@ -35,9 +42,15 @@ content = '''// Planman configuration
\"max_rounds\": 3,
// Minimum rounds before a plan can pass (0 = no minimum)
\"min_rounds\": 0,
- // Override Codex model (empty = codex default)
+ // Override evaluator model (empty = evaluator default)
\"model\": \"\",
- // Pass through if Codex fails
+ // Evaluator provider (auto = Codex reviews Claude, Claude reviews Codex)
+ \"evaluator\": \"auto\",
+ // Codex CLI binary/path
+ \"codex_bin\": \"codex\",
+ // Claude Code CLI binary/path
+ \"claude_bin\": \"claude\",
+ // Pass through if the evaluator fails
\"fail_open\": true,
// Master switch
\"enabled\": true,
@@ -45,7 +58,7 @@ content = '''// Planman configuration
\"custom_rubric\": \"\",
// Debug output to stderr + log file
\"verbose\": false,
- // Codex verifies plan against actual source files
+ // Evaluator verifies plan against actual source files
\"source_verify\": true,
// Stress-test rounds before Codex evaluation (false=off, true=1, or number N)
\"stress_test\": false,
@@ -64,7 +77,6 @@ content = '''// Planman configuration
}
'''
-os.makedirs('.claude', exist_ok=True)
with open(jsonc_path, 'w') as f:
f.write(content)
print(f'Created {jsonc_path}')
@@ -72,4 +84,4 @@ print('Edit the values you want to change. Run /planman:status to verify.')
"
```
-Report the result to the user. If the file was created, mention they can run `/planman` to verify the effective configuration.
+Report the result to the user. If the file was created, mention they can run `/planman:status` to verify the effective configuration.
diff --git a/commands/status.md b/commands/status.md
index 1c1802e..7205782 100644
--- a/commands/status.md
+++ b/commands/status.md
@@ -4,28 +4,41 @@ description: Show planman status and effective configuration
# Planman Status
-Check the current planman configuration and codex CLI status.
+Check the current planman configuration and evaluator CLI status.
## Instructions
Run these commands and report the results:
-1. Check if codex is installed: `which codex && codex --version || echo "codex not installed"`
+1. Check if evaluator CLIs are installed:
+ `which codex && codex --version || echo "codex not installed"`
+ `which claude && claude --version || echo "claude not installed"`
2. Show effective configuration by running:
```bash
python3 -c "
- import json, sys, os; sys.path.insert(0, '${CLAUDE_PLUGIN_ROOT}/scripts')
- plugin_json = os.path.join('${CLAUDE_PLUGIN_ROOT}', '.claude-plugin', 'plugin.json')
+ import json, sys, os
+ root = os.environ.get('CLAUDE_PLUGIN_ROOT') or os.environ.get('CODEX_PLUGIN_ROOT') or os.getcwd()
+ sys.path.insert(0, os.path.join(root, 'scripts'))
+ plugin_json = os.path.join(root, '.claude-plugin', 'plugin.json')
+ if not os.path.isfile(plugin_json):
+ plugin_json = os.path.join(root, '.codex-plugin', 'plugin.json')
with open(plugin_json) as f:
ver = json.load(f)['version']
from config import load_config
+ from evaluator import detect_host, resolve_evaluator
c = load_config(cwd=os.getcwd())
+ host = detect_host({})
+ evaluator = resolve_evaluator(c, host=host)
print(f'version: {ver}')
print(f'enabled: {c.enabled}')
+ print(f'host: {host} (auto-detected)')
+ print(f'evaluator: {c.evaluator} -> {evaluator}')
print(f'threshold: {c.threshold}/10')
print(f'max_rounds: {c.max_rounds}')
print(f'min_rounds: {c.min_rounds}')
- print(f'model: {c.model or \"(codex default)\"}')
+ print(f'model: {c.model or \"(evaluator default)\"}')
+ print(f'codex_bin: {c.codex_bin}')
+ print(f'claude_bin: {c.claude_bin}')
print(f'fail_open: {c.fail_open}')
print(f'verbose: {c.verbose}')
print(f'rubric: {\"custom\" if c.rubric != __import__(\"config\").DEFAULT_RUBRIC else \"built-in\"}')
@@ -35,6 +48,7 @@ Run these commands and report the results:
print(f'auto_answer: {c.auto_answer}')
"
```
-3. Check for active sessions: `python3 ${CLAUDE_PLUGIN_ROOT}/scripts/clear_state.py list`
+3. Check for active sessions:
+ `python3 -c "import os, subprocess; root = os.environ.get('CLAUDE_PLUGIN_ROOT') or os.environ.get('CODEX_PLUGIN_ROOT') or os.getcwd(); subprocess.run(['python3', os.path.join(root, 'scripts', 'clear_state.py'), 'list'])"`
Format the output as a clean status report.
diff --git a/hooks.json b/hooks.json
new file mode 100644
index 0000000..6a3b7cb
--- /dev/null
+++ b/hooks.json
@@ -0,0 +1,17 @@
+{
+ "description": "Evaluates Codex plans using another agent",
+ "hooks": {
+ "Stop": [
+ {
+ "hooks": [
+ {
+ "type": "command",
+ "command": "python3 ./scripts/codex_stop_hook.py",
+ "timeout": 600,
+ "statusMessage": "Evaluating plan"
+ }
+ ]
+ }
+ ]
+ }
+}
diff --git a/plugins/planman/.codex-plugin/plugin.json b/plugins/planman/.codex-plugin/plugin.json
new file mode 100644
index 0000000..4e0ae21
--- /dev/null
+++ b/plugins/planman/.codex-plugin/plugin.json
@@ -0,0 +1,24 @@
+{
+ "name": "planman",
+ "version": "0.4.9",
+ "description": "Evaluates AI coding-agent plans using another agent",
+ "author": { "name": "dyn" },
+ "license": "MIT",
+ "keywords": ["plan", "evaluation", "quality-gate", "codex", "claude"],
+ "homepage": "https://github.com/RusDyn/planman",
+ "repository": "https://github.com/RusDyn/planman",
+ "interface": {
+ "displayName": "Planman",
+ "shortDescription": "Evaluates coding-agent plans before execution",
+ "longDescription": "Planman evaluates implementation plans from Codex or Claude Code with another local coding-agent CLI and blocks weak plans with actionable feedback.",
+ "developerName": "dyn",
+ "category": "Developer Tools",
+ "capabilities": ["Read", "Interactive"],
+ "defaultPrompt": [
+ "Evaluate my implementation plan",
+ "Stress-test this coding plan",
+ "Show planman status"
+ ],
+ "brandColor": "#2563EB"
+ }
+}
diff --git a/plugins/planman/hooks.json b/plugins/planman/hooks.json
new file mode 100644
index 0000000..6a3b7cb
--- /dev/null
+++ b/plugins/planman/hooks.json
@@ -0,0 +1,17 @@
+{
+ "description": "Evaluates Codex plans using another agent",
+ "hooks": {
+ "Stop": [
+ {
+ "hooks": [
+ {
+ "type": "command",
+ "command": "python3 ./scripts/codex_stop_hook.py",
+ "timeout": 600,
+ "statusMessage": "Evaluating plan"
+ }
+ ]
+ }
+ ]
+ }
+}
diff --git a/plugins/planman/schemas/evaluation.json b/plugins/planman/schemas/evaluation.json
new file mode 100644
index 0000000..17df411
--- /dev/null
+++ b/plugins/planman/schemas/evaluation.json
@@ -0,0 +1,66 @@
+{
+ "type": "object",
+ "properties": {
+ "score": {
+ "type": "integer",
+ "minimum": 1,
+ "maximum": 10,
+ "description": "Overall plan quality score (1-10). Sum of breakdown categories."
+ },
+ "breakdown": {
+ "type": "object",
+ "properties": {
+ "completeness": {
+ "type": "integer",
+ "minimum": 0,
+ "maximum": 2,
+ "description": "Does the plan cover all requirements? 0=missing major pieces, 1=partial, 2=thorough"
+ },
+ "correctness": {
+ "type": "integer",
+ "minimum": 0,
+ "maximum": 2,
+ "description": "Is the approach technically sound? 0=flawed, 1=mostly correct, 2=solid"
+ },
+ "sequencing": {
+ "type": "integer",
+ "minimum": 0,
+ "maximum": 2,
+ "description": "Are steps in the right order with dependencies respected? 0=broken, 1=mostly, 2=logical"
+ },
+ "risk_awareness": {
+ "type": "integer",
+ "minimum": 0,
+ "maximum": 2,
+ "description": "Does the plan identify risks and edge cases? 0=ignores risks, 1=some, 2=thorough"
+ },
+ "clarity": {
+ "type": "integer",
+ "minimum": 0,
+ "maximum": 2,
+ "description": "Is the plan clear and actionable? 0=vague, 1=ok, 2=precise and actionable"
+ }
+ },
+ "required": ["completeness", "correctness", "sequencing", "risk_awareness", "clarity"],
+ "additionalProperties": false
+ },
+ "weaknesses": {
+ "type": "array",
+ "items": { "type": "string" },
+ "description": "2-5 specific issues that lower the score. Reference step numbers."
+ },
+ "suggestions": {
+ "type": "array",
+ "items": { "type": "string" },
+ "description": "2-5 concrete improvements. Reference step numbers."
+ },
+ "strengths": {
+ "type": "array",
+ "items": { "type": "string" },
+ "minItems": 1,
+ "description": "1-3 things the plan does well."
+ }
+ },
+ "required": ["score", "breakdown", "weaknesses", "suggestions", "strengths"],
+ "additionalProperties": false
+}
diff --git a/plugins/planman/scripts/codex_stop_hook.py b/plugins/planman/scripts/codex_stop_hook.py
new file mode 100644
index 0000000..0214dc3
--- /dev/null
+++ b/plugins/planman/scripts/codex_stop_hook.py
@@ -0,0 +1,149 @@
+#!/usr/bin/env python3
+"""Codex Stop hook — evaluate latest plan-like content.
+
+This adapter is intentionally conservative. It only evaluates text that looks
+like a plan; if the Codex hook payload shape does not expose plan text, it
+fails open and logs the skip.
+"""
+
+import json
+import os
+import re
+import sys
+
+_scripts_dir = os.path.dirname(os.path.abspath(__file__))
+if _scripts_dir not in sys.path:
+ sys.path.insert(0, _scripts_dir)
+
+from config import load_config
+from hook_utils import log, run_evaluation, safe_session_id
+from evaluator import check_evaluator_installed
+
+
+_PLAN_RE = re.compile(
+ r"(?ims)(|implementation plan|^# .{0,80}plan\b|^\*\*summary\*\*|^\s*1\.\s+.+\n\s*2\.)"
+)
+
+
+def _walk_strings(value, path=""):
+ """Yield (path, string) pairs from nested JSON-like hook input."""
+ if isinstance(value, str):
+ yield path, value
+ elif isinstance(value, dict):
+ for key, child in value.items():
+ child_path = f"{path}.{key}" if path else str(key)
+ yield from _walk_strings(child, child_path)
+ elif isinstance(value, list):
+ for idx, child in enumerate(value):
+ yield from _walk_strings(child, f"{path}[{idx}]")
+
+
+def _extract_plan_text(hook_input):
+ """Extract the latest clearly plan-like text from a Codex hook payload."""
+ candidates = []
+ order = 0
+ for path, text in _walk_strings(hook_input):
+ order += 1
+ stripped = text.strip()
+ if len(stripped) < 80:
+ continue
+ lower_path = path.lower()
+ score = 0
+ if "plan" in lower_path:
+ score += 3
+ if any(part in lower_path for part in ("message", "content", "text", "markdown", "summary")):
+ score += 1
+ if _PLAN_RE.search(stripped):
+ score += 4
+ if score >= 4:
+ candidates.append((score, order, stripped))
+ if not candidates:
+ return None
+ candidates.sort(key=lambda item: (item[0], item[1]))
+ return candidates[-1][2]
+
+
+def _output_allow(system_message=None):
+ result = {}
+ if system_message:
+ result["systemMessage"] = system_message
+ if result:
+ json.dump(result, sys.stdout, ensure_ascii=True)
+ sys.exit(0)
+
+
+def _output_block(reason, system_message=None):
+ result = {
+ "decision": "block",
+ "reason": reason,
+ }
+ if system_message:
+ result["systemMessage"] = system_message
+ json.dump(result, sys.stdout, ensure_ascii=True)
+ sys.exit(0)
+
+
+def _main():
+ if os.environ.get("_PLANMAN_EVALUATOR"):
+ _output_allow()
+ return
+
+ try:
+ raw = sys.stdin.read()
+ except Exception:
+ raw = ""
+
+ try:
+ hook_input = json.loads(raw) if raw.strip() else {}
+ except (json.JSONDecodeError, ValueError):
+ hook_input = {}
+
+ cwd = hook_input.get("cwd") or os.getcwd()
+ config = load_config(cwd=cwd)
+ if not config.enabled:
+ _output_allow()
+ return
+
+ plan_text = _extract_plan_text(hook_input)
+ if not plan_text:
+ log("codex host: no plan-like content found in Stop payload", config, cwd)
+ _output_allow()
+ return
+
+ if not check_evaluator_installed(config, host="codex"):
+ log("codex host: evaluator CLI not installed — allowing", config, cwd)
+ _output_allow()
+ return
+
+ session_id = safe_session_id(hook_input.get("session_id") or hook_input.get("thread_id") or "codex")
+ log(f"codex host: evaluating plan-like content (session={session_id})", config, cwd)
+
+ result = run_evaluation(
+ plan_text,
+ session_id,
+ config,
+ cwd=cwd,
+ plan_path="codex:stop",
+ host="codex",
+ )
+
+ action = result["action"]
+ reason = result.get("reason")
+ sys_msg = result.get("system_message")
+
+ if action == "block":
+ _output_block(reason=reason, system_message=sys_msg)
+ else:
+ _output_allow(system_message=sys_msg)
+
+
+def main():
+ try:
+ _main()
+ except Exception as e:
+ print(f"[planman] FATAL in codex_stop_hook: {e}", file=sys.stderr)
+ sys.exit(0)
+
+
+if __name__ == "__main__":
+ main()
diff --git a/plugins/planman/scripts/config.py b/plugins/planman/scripts/config.py
new file mode 100644
index 0000000..1fc37a8
--- /dev/null
+++ b/plugins/planman/scripts/config.py
@@ -0,0 +1,287 @@
+"""Configuration loader for planman.
+
+Loads settings from two sources (env vars override file):
+ 1. .planman.jsonc or .claude/planman.jsonc (project-level, supports // comments)
+ 2. PLANMAN_* environment variables (highest priority)
+"""
+
+import json
+import os
+import re
+
+DEFAULT_RUBRIC = """\
+Score the plan on these 5 criteria (0-2 each, 10 max):
+
+1. **Completeness** (0-2): Does the plan address all stated requirements with no critical gaps?
+2. **Correctness** (0-2): Is the technical approach sound? Any flaws or misunderstandings?
+3. **Sequencing** (0-2): Are steps ordered logically? Are dependencies respected?
+4. **Risk Awareness** (0-2): Does the plan address risks proportionate to the task's scope? Simple tasks need minimal risk coverage.
+5. **Clarity** (0-2): Are steps specific and actionable? Could a developer follow them without ambiguity?
+
+The overall score MUST equal the sum of the 5 breakdown scores.
+A score of 7+ means the plan is ready to execute. Prefer simple, focused plans — do NOT penalize for omitting rollback plans, exhaustive risk analysis, or verification strategies unless the task specifically requires them.\
+"""
+
+DEFAULT_STRESS_TEST_PROMPT = """\
+Stress-test this plan. Spawn 2-3 research agents to examine it in parallel from \
+different angles (correctness, edge cases, feasibility, current best practices). \
+Have agents use web search to validate approaches and identify known pitfalls. \
+Cross-reference findings — web research may surface false positives or outdated \
+practices, so keep only what's well-supported. Then make targeted fixes to the \
+1-2 most critical weaknesses. Focus on value; don't add complexity for its own sake.\
+"""
+
+DEFAULTS = {
+ "threshold": 7,
+ "max_rounds": 3,
+ "min_rounds": 0,
+ "model": "",
+ "evaluator": "auto",
+ "codex_bin": "codex",
+ "claude_bin": "claude",
+ "fail_open": True,
+ "enabled": True,
+ "custom_rubric": "",
+ "verbose": False,
+ "stress_test": 0,
+ "context": "",
+ "source_verify": True,
+ "auto_answer": False,
+ "stress_test_prompt": "",
+ "plan_dirs": [],
+ "exec_patterns": [],
+ "skill_patterns": [],
+}
+
+_BOOL_TRUTHY = {"true", "1", "yes", "on"}
+_BOOL_FALSY = {"false", "0", "no", "off"}
+
+
+def _coerce_bool(value, key="fail_open"):
+ """Coerce a string to bool, falling back to default for the given key."""
+ if isinstance(value, bool):
+ return value
+ s = str(value).lower().strip()
+ if s in _BOOL_TRUTHY:
+ return True
+ if s in _BOOL_FALSY:
+ return False
+ return DEFAULTS.get(key, True)
+
+
+def _coerce_int(value, key):
+ """Coerce a string to int, falling back to default."""
+ try:
+ return int(value)
+ except (ValueError, TypeError):
+ return DEFAULTS.get(key, 0)
+
+
+def _coerce_stress_test(value, key="stress_test"):
+ """Coerce stress_test: False/false→0, True/true→1, number→int."""
+ if isinstance(value, bool):
+ return 1 if value else 0
+ if isinstance(value, int):
+ return max(0, value)
+ s = str(value).lower().strip()
+ if s in _BOOL_TRUTHY:
+ return 1
+ if s in _BOOL_FALSY:
+ return 0
+ try:
+ return max(0, int(s))
+ except (ValueError, TypeError):
+ return DEFAULTS.get(key, 0)
+
+
+def _coerce_list(value, key):
+ """Coerce to list[str]. Accepts JSON arrays, comma-separated strings."""
+ if isinstance(value, list):
+ return [str(item).strip() for item in value if str(item).strip()]
+ if isinstance(value, str):
+ s = value.strip()
+ if not s:
+ return []
+ # Try JSON array first (handles commas in regex patterns)
+ if s.startswith("["):
+ try:
+ parsed = json.loads(s)
+ if isinstance(parsed, list):
+ return [str(item).strip() for item in parsed if str(item).strip()]
+ except (json.JSONDecodeError, ValueError):
+ pass
+ # Fall back to comma-separated
+ return [item.strip() for item in s.split(",") if item.strip()]
+ return DEFAULTS.get(key, [])
+
+
+class Config:
+ """Planman configuration."""
+
+ __slots__ = (
+ "threshold",
+ "max_rounds",
+ "min_rounds",
+ "model",
+ "evaluator",
+ "codex_bin",
+ "claude_bin",
+ "fail_open",
+ "enabled",
+ "rubric",
+ "verbose",
+ "stress_test",
+ "context",
+ "source_verify",
+ "auto_answer",
+ "stress_test_prompt",
+ "plan_dirs",
+ "exec_patterns",
+ "skill_patterns",
+ )
+
+ def __init__(self, **kwargs):
+ self.threshold = kwargs.get("threshold", DEFAULTS["threshold"])
+ self.max_rounds = kwargs.get("max_rounds", DEFAULTS["max_rounds"])
+ self.min_rounds = kwargs.get("min_rounds", DEFAULTS["min_rounds"])
+ self.model = kwargs.get("model", DEFAULTS["model"])
+ self.evaluator = kwargs.get("evaluator", DEFAULTS["evaluator"])
+ self.codex_bin = kwargs.get("codex_bin", DEFAULTS["codex_bin"])
+ self.claude_bin = kwargs.get("claude_bin", DEFAULTS["claude_bin"])
+ self.fail_open = kwargs.get("fail_open", DEFAULTS["fail_open"])
+ self.enabled = kwargs.get("enabled", DEFAULTS["enabled"])
+ self.rubric = kwargs.get("rubric", "") or DEFAULT_RUBRIC
+ self.verbose = kwargs.get("verbose", DEFAULTS["verbose"])
+ self.stress_test = kwargs.get("stress_test", DEFAULTS["stress_test"])
+ self.context = kwargs.get("context", "")
+ self.source_verify = kwargs.get("source_verify", DEFAULTS["source_verify"])
+ self.auto_answer = kwargs.get("auto_answer", DEFAULTS["auto_answer"])
+ self.stress_test_prompt = kwargs.get("stress_test_prompt", "") or DEFAULT_STRESS_TEST_PROMPT
+ self.plan_dirs = kwargs.get("plan_dirs", DEFAULTS["plan_dirs"])
+ self.exec_patterns = kwargs.get("exec_patterns", DEFAULTS["exec_patterns"])
+ self.skill_patterns = kwargs.get("skill_patterns", DEFAULTS["skill_patterns"])
+
+
+def _strip_jsonc_comments(text):
+ """Strip // line comments from JSONC text, preserving strings."""
+ return re.sub(
+ r'("(?:[^"\\]|\\.)*")|//[^\n]*',
+ lambda m: m.group(1) if m.group(1) else "",
+ text,
+ )
+
+
+def _load_file_config(cwd=None):
+ """Load project planman config if it exists."""
+ base = cwd or "."
+ candidate_paths = (
+ os.path.join(base, ".planman.jsonc"),
+ os.path.join(base, ".planman.json"),
+ os.path.join(base, ".claude", "planman.jsonc"),
+ os.path.join(base, ".claude", "planman.json"),
+ )
+ path = next((p for p in candidate_paths if os.path.isfile(p)), None)
+ if path is None:
+ return {}
+ try:
+ with open(path, "r", encoding="utf-8") as f:
+ raw = f.read()
+ data = json.loads(_strip_jsonc_comments(raw))
+ if isinstance(data, dict):
+ return data
+ except (json.JSONDecodeError, OSError, ValueError):
+ pass
+ return {}
+
+
+def _load_env_overrides():
+ """Load PLANMAN_* environment variable overrides."""
+ overrides = {}
+ env_map = {
+ "PLANMAN_THRESHOLD": ("threshold", _coerce_int),
+ "PLANMAN_MAX_ROUNDS": ("max_rounds", _coerce_int),
+ "PLANMAN_MIN_ROUNDS": ("min_rounds", _coerce_int),
+ "PLANMAN_MODEL": ("model", str),
+ "PLANMAN_EVALUATOR": ("evaluator", str),
+ "PLANMAN_CODEX_BIN": ("codex_bin", str),
+ "PLANMAN_CLAUDE_BIN": ("claude_bin", str),
+ "PLANMAN_FAIL_OPEN": ("fail_open", _coerce_bool),
+ "PLANMAN_ENABLED": ("enabled", _coerce_bool),
+ "PLANMAN_RUBRIC": ("custom_rubric", str),
+ "PLANMAN_VERBOSE": ("verbose", _coerce_bool),
+ "PLANMAN_STRESS_TEST": ("stress_test", _coerce_stress_test),
+ "PLANMAN_CONTEXT": ("context", str),
+ "PLANMAN_SOURCE_VERIFY": ("source_verify", _coerce_bool),
+ "PLANMAN_AUTO_ANSWER": ("auto_answer", _coerce_bool),
+ "PLANMAN_PLAN_DIRS": ("plan_dirs", _coerce_list),
+ "PLANMAN_EXEC_PATTERNS": ("exec_patterns", _coerce_list),
+ "PLANMAN_SKILL_PATTERNS": ("skill_patterns", _coerce_list),
+ }
+ for env_var, (key, coerce) in env_map.items():
+ val = os.environ.get(env_var)
+ if val is not None:
+ if coerce in (_coerce_int, _coerce_bool, _coerce_stress_test, _coerce_list):
+ overrides[key] = coerce(val, key)
+ else:
+ overrides[key] = coerce(val)
+ return overrides
+
+
+def load_config(cwd=None):
+ """Load config: defaults < file < env vars."""
+ merged = dict(DEFAULTS)
+
+ # Layer 1: file config
+ file_cfg = _load_file_config(cwd=cwd)
+ for key in DEFAULTS:
+ if key in file_cfg:
+ merged[key] = file_cfg[key]
+
+ # Remap custom_rubric → rubric
+ if "custom_rubric" in file_cfg:
+ merged["custom_rubric"] = file_cfg["custom_rubric"]
+
+ # Layer 2: env overrides (highest priority)
+ env_cfg = _load_env_overrides()
+ merged.update(env_cfg)
+
+ # Clamp numeric ranges (safe coercion — invalid strings fall back to defaults)
+ merged["threshold"] = max(0, min(10, _coerce_int(merged["threshold"], "threshold")))
+ merged["max_rounds"] = max(1, min(100, _coerce_int(merged["max_rounds"], "max_rounds")))
+ merged["min_rounds"] = max(0, min(100, _coerce_int(merged["min_rounds"], "min_rounds")))
+
+ # Coerce stress_test to int (False→0, True→1, number→int)
+ merged["stress_test"] = _coerce_stress_test(merged["stress_test"], "stress_test")
+
+ evaluator = str(merged.get("evaluator", "auto")).lower().strip()
+ if evaluator not in ("auto", "codex", "claude"):
+ evaluator = DEFAULTS["evaluator"]
+ merged["evaluator"] = evaluator
+
+ # Coerce list fields
+ merged["plan_dirs"] = _coerce_list(merged.get("plan_dirs", []), "plan_dirs")
+ merged["exec_patterns"] = _coerce_list(merged.get("exec_patterns", []), "exec_patterns")
+ merged["skill_patterns"] = _coerce_list(merged.get("skill_patterns", []), "skill_patterns")
+
+ # Build Config, mapping custom_rubric to rubric
+ return Config(
+ threshold=merged["threshold"],
+ max_rounds=merged["max_rounds"],
+ min_rounds=merged["min_rounds"],
+ model=merged["model"],
+ evaluator=merged["evaluator"],
+ codex_bin=merged.get("codex_bin", "codex") or "codex",
+ claude_bin=merged.get("claude_bin", "claude") or "claude",
+ fail_open=merged["fail_open"],
+ enabled=merged["enabled"],
+ rubric=merged.get("custom_rubric", ""),
+ verbose=merged["verbose"],
+ stress_test=merged["stress_test"],
+ context=merged.get("context", ""),
+ source_verify=_coerce_bool(merged.get("source_verify", True), "source_verify"),
+ auto_answer=_coerce_bool(merged.get("auto_answer", False), "auto_answer"),
+ stress_test_prompt=merged.get("stress_test_prompt", ""),
+ plan_dirs=merged["plan_dirs"],
+ exec_patterns=merged["exec_patterns"],
+ skill_patterns=merged["skill_patterns"],
+ )
diff --git a/plugins/planman/scripts/evaluator.py b/plugins/planman/scripts/evaluator.py
new file mode 100644
index 0000000..453705d
--- /dev/null
+++ b/plugins/planman/scripts/evaluator.py
@@ -0,0 +1,375 @@
+"""Plan evaluation via coding-agent CLI subprocesses.
+
+Calls Codex or Claude Code with structured JSON scoring.
+"""
+
+import json
+import os
+import shutil
+import subprocess
+import sys
+
+PLUGIN_ROOT = (
+ os.environ.get("CLAUDE_PLUGIN_ROOT", "")
+ or os.environ.get("CODEX_PLUGIN_ROOT", "")
+ or os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
+)
+
+_CODEX_BIN = "codex"
+_CLAUDE_BIN = "claude"
+
+_codex_available = None # cached result
+_claude_available = None # cached result
+
+# Known codex error patterns → actionable messages
+_KNOWN_ERRORS = [
+ ("usage limit", "ChatGPT usage limit reached. Upgrade at https://chatgpt.com/explore/pro or wait for reset."),
+ ("rate limit", "Rate limited by OpenAI. Wait a few minutes and retry."),
+ ("authentication", "Codex authentication failed. Run `codex auth` to re-authenticate."),
+ ("could not connect", "Network error connecting to OpenAI. Check internet connection."),
+ ("context length exceeded", "Plan + rubric too large for model context. Reduce plan size."),
+ ("model not found", "Configured model not available. Check PLANMAN_MODEL setting."),
+]
+
+
+def _extract_codex_error(stderr):
+ """Extract actionable error from codex stderr, or return tail snippet."""
+ if not stderr:
+ return "no stderr output"
+ lower = stderr.lower()
+ for pattern, message in _KNOWN_ERRORS:
+ if pattern in lower:
+ return message
+ # Fallback: return tail where actual errors live (after banner + prompt echo)
+ return f"...{stderr[-1500:]}"
+
+
+def check_codex_installed(codex_path=None):
+ """Check if codex CLI is installed. Result is cached."""
+ global _codex_available
+ if codex_path is None and _codex_available is not None:
+ return _codex_available
+ available = shutil.which(codex_path or _CODEX_BIN) is not None
+ if codex_path is None:
+ _codex_available = available
+ return available
+
+
+def check_claude_installed(claude_path=None):
+ """Check if Claude Code CLI is installed. Result is cached."""
+ global _claude_available
+ if claude_path is None and _claude_available is not None:
+ return _claude_available
+ available = shutil.which(claude_path or _CLAUDE_BIN) is not None
+ if claude_path is None:
+ _claude_available = available
+ return available
+
+
+def reset_codex_cache():
+ """Reset the cached codex availability check (for testing)."""
+ global _codex_available
+ _codex_available = None
+
+
+def reset_claude_cache():
+ """Reset the cached Claude CLI availability check (for testing)."""
+ global _claude_available
+ _claude_available = None
+
+
+def detect_host(hook_input=None):
+ """Infer the host agent from hook/runtime context."""
+ if os.environ.get("CLAUDE_PLUGIN_ROOT"):
+ return "claude"
+ if os.environ.get("CODEX_HOME") or os.environ.get("CODEX_SANDBOX"):
+ return "codex"
+ if isinstance(hook_input, dict):
+ tool_name = hook_input.get("tool_name")
+ if tool_name == "ExitPlanMode" or "CLAUDE_PLUGIN_ROOT" in hook_input:
+ return "claude"
+ event = str(hook_input.get("hook_event_name") or hook_input.get("hookEventName") or "")
+ if event in ("Stop", "UserPromptSubmit", "SessionStart"):
+ return "codex"
+ if "codex" in str(hook_input.get("source", "")).lower():
+ return "codex"
+ return "claude"
+
+
+def resolve_evaluator(config, host="claude"):
+ """Resolve configured evaluator provider for a detected host."""
+ evaluator = getattr(config, "evaluator", "auto") or "auto"
+ if evaluator != "auto":
+ return evaluator
+ return "claude" if host == "codex" else "codex"
+
+
+def check_evaluator_installed(config, host="claude"):
+ """Check whether the resolved evaluator CLI is installed."""
+ provider = resolve_evaluator(config, host=host)
+ if provider == "claude":
+ return check_claude_installed(getattr(config, "claude_bin", None) or _CLAUDE_BIN)
+ return check_codex_installed(getattr(config, "codex_bin", None) or _CODEX_BIN)
+
+
+def build_prompt(plan_text, rubric, previous_feedback=None, round_number=1,
+ context=None, source_verify=True, provider="codex"):
+ """Build the evaluation prompt for codex exec."""
+ context_section = ""
+ if context:
+ context_section = f"## Project Context\n\n{context}\n\n"
+
+ prompt = (
+ "You are a senior software architect reviewing an implementation plan.\n\n"
+ f"{context_section}"
+ "Evaluate the following implementation plan using the rubric.\n\n"
+ f"{rubric}\n\n"
+ "## Feedback Guidelines\n\n"
+ "- Prioritize issues: list critical problems first, minor improvements last\n"
+ "- Be specific: reference exact steps by number\n"
+ "- Be actionable: say what to change, not just what's wrong\n\n"
+ f"## Plan to Evaluate (Round {round_number})\n\n"
+ f"{plan_text}\n"
+ )
+ if source_verify:
+ read_hint = "`cat ` or search with `rg`"
+ if provider == "claude":
+ read_hint = "read/search tools"
+ prompt += (
+ "\n## Source Verification\n\n"
+ "You have read-only access to the project filesystem (cwd = project root).\n"
+ "When evaluating correctness and completeness:\n"
+ f"1. If the plan references specific files, verify them with {read_hint}\n"
+ "2. Verify that APIs, function signatures, and module structures mentioned in the plan exist\n"
+ "3. Check that the plan's assumptions about the codebase are accurate\n"
+ "4. Note discrepancies between the plan and actual code as correctness issues\n"
+ "5. Keep file reads focused — verify key claims, don't read the entire codebase\n"
+ )
+ if previous_feedback:
+ prompt += (
+ f"\n## Previous Feedback (Round {round_number - 1})\n\n"
+ f"{previous_feedback}\n\n"
+ "Assess: Which feedback items were addressed? Which were ignored? "
+ "Focus new feedback on remaining and newly discovered issues.\n"
+ )
+ return prompt
+
+
+def _schema_path():
+ path = os.path.join(PLUGIN_ROOT, "schemas", "evaluation.json")
+ if not os.path.isfile(path):
+ return None, f"schema file not found: {path}. Check plugin root environment."
+ return path, None
+
+
+def _subprocess_env():
+ env = os.environ.copy()
+ env["_PLANMAN_EVALUATOR"] = "1"
+ return env
+
+
+def evaluate_plan(plan_text, config, previous_feedback=None, round_number=1, cwd=None, host="claude"):
+ """Evaluate a plan with the configured provider and structured output.
+
+ Returns (result_dict, error_string). On success error_string is None.
+ On failure result_dict is None and error_string describes the problem.
+ """
+ provider = resolve_evaluator(config, host=host)
+ if provider == "claude":
+ return evaluate_plan_claude(plan_text, config, previous_feedback, round_number, cwd)
+ return evaluate_plan_codex(plan_text, config, previous_feedback, round_number, cwd)
+
+
+def evaluate_plan_codex(plan_text, config, previous_feedback=None, round_number=1, cwd=None):
+ """Evaluate a plan via codex exec with structured output."""
+ codex_bin = getattr(config, "codex_bin", None) or _CODEX_BIN
+ if not check_codex_installed(codex_bin):
+ return None, f"codex CLI not found. Install: npm install -g @openai/codex"
+
+ prompt = build_prompt(
+ plan_text, config.rubric, previous_feedback, round_number,
+ context=config.context,
+ source_verify=getattr(config, "source_verify", True),
+ provider="codex",
+ )
+
+ _MAX_PROMPT_SIZE = 2_000_000 # 2MB hard cap
+ if len(prompt) > _MAX_PROMPT_SIZE:
+ return None, f"prompt too large ({len(prompt) // 1024}KB > 2MB). Reduce max_rounds or plan size."
+
+ effective_timeout = 570 # 600s hook timeout − 30s margin
+
+ schema_path, schema_error = _schema_path()
+ if schema_error:
+ return None, schema_error
+
+ cmd = [
+ codex_bin,
+ "exec", "-", # Read prompt from stdin
+ "--output-schema", schema_path,
+ "--sandbox", "read-only",
+ "--skip-git-repo-check",
+ "--ephemeral", # Don't persist session files
+ "--disable", "hooks",
+ ]
+ if config.model:
+ cmd.extend(["-m", config.model])
+
+ try:
+ result = subprocess.run(
+ cmd,
+ input=prompt, # Pass prompt via stdin
+ capture_output=True,
+ text=True,
+ errors='replace', # prevent UnicodeDecodeError on bad codex output
+ timeout=effective_timeout,
+ cwd=cwd or os.getcwd(),
+ env=_subprocess_env(),
+ )
+ except subprocess.TimeoutExpired:
+ return None, f"codex timed out ({effective_timeout}s). Try a shorter plan or check codex CLI health."
+ except FileNotFoundError:
+ reset_codex_cache()
+ return None, "codex not found. Install: npm install -g @openai/codex"
+ except (OSError, UnicodeDecodeError) as e:
+ return None, f"failed to run codex: {e}"
+
+ if config.verbose:
+ print(f"[planman] codex exit code: {result.returncode}", file=sys.stderr)
+ if result.stderr:
+ verbose_limit = 4000 if result.returncode != 0 else 2000
+ print(f"[planman] codex stderr (last {verbose_limit}): {result.stderr[-verbose_limit:]}", file=sys.stderr)
+
+ if result.returncode != 0:
+ error_detail = _extract_codex_error(result.stderr)
+ return None, f"codex exec failed (exit {result.returncode}): {error_detail}"
+
+ return parse_codex_output(result.stdout)
+
+
+def evaluate_plan_claude(plan_text, config, previous_feedback=None, round_number=1, cwd=None):
+ """Evaluate a plan via Claude Code print mode with structured output."""
+ claude_bin = getattr(config, "claude_bin", None) or _CLAUDE_BIN
+ if not check_claude_installed(claude_bin):
+ return None, "claude CLI not found. Install and authenticate Claude Code."
+
+ prompt = build_prompt(
+ plan_text, config.rubric, previous_feedback, round_number,
+ context=config.context,
+ source_verify=getattr(config, "source_verify", True),
+ provider="claude",
+ )
+
+ _MAX_PROMPT_SIZE = 2_000_000
+ if len(prompt) > _MAX_PROMPT_SIZE:
+ return None, f"prompt too large ({len(prompt) // 1024}KB > 2MB). Reduce max_rounds or plan size."
+
+ schema_path, schema_error = _schema_path()
+ if schema_error:
+ return None, schema_error
+
+ try:
+ with open(schema_path, "r", encoding="utf-8") as f:
+ schema_text = f.read()
+ except OSError as e:
+ return None, f"schema file unreadable: {e}"
+
+ cmd = [
+ claude_bin,
+ "-p",
+ "--output-format", "json",
+ "--json-schema", schema_text,
+ "--no-session-persistence",
+ "--tools", "Read,Grep,Glob",
+ ]
+ if config.model:
+ cmd.extend(["--model", config.model])
+
+ effective_timeout = 570
+ try:
+ result = subprocess.run(
+ cmd,
+ input=prompt,
+ capture_output=True,
+ text=True,
+ errors="replace",
+ timeout=effective_timeout,
+ cwd=cwd or os.getcwd(),
+ env=_subprocess_env(),
+ )
+ except subprocess.TimeoutExpired:
+ return None, f"claude timed out ({effective_timeout}s). Try a shorter plan or check Claude Code CLI health."
+ except FileNotFoundError:
+ reset_claude_cache()
+ return None, "claude not found. Install and authenticate Claude Code."
+ except (OSError, UnicodeDecodeError) as e:
+ return None, f"failed to run claude: {e}"
+
+ if config.verbose:
+ print(f"[planman] claude exit code: {result.returncode}", file=sys.stderr)
+ if result.stderr:
+ verbose_limit = 4000 if result.returncode != 0 else 2000
+ print(f"[planman] claude stderr (last {verbose_limit}): {result.stderr[-verbose_limit:]}", file=sys.stderr)
+
+ if result.returncode != 0:
+ return None, f"claude -p failed (exit {result.returncode}): {_extract_codex_error(result.stderr)}"
+
+ return parse_codex_output(result.stdout)
+
+
+def parse_codex_output(stdout):
+ """Parse structured JSON from codex exec stdout.
+
+ Returns (result_dict, error_string).
+ """
+ if not stdout or not stdout.strip():
+ return None, "codex returned empty output"
+
+ try:
+ data = json.loads(stdout)
+ except json.JSONDecodeError as e:
+ return None, f"codex returned malformed output. Set PLANMAN_VERBOSE=true for details."
+
+ # Claude's JSON mode can wrap the final structured response.
+ if isinstance(data, dict) and "result" in data and not any(k in data for k in ("score", "breakdown")):
+ wrapped = data.get("result")
+ if isinstance(wrapped, str):
+ try:
+ data = json.loads(wrapped)
+ except json.JSONDecodeError:
+ return None, "claude returned malformed structured result"
+ elif isinstance(wrapped, dict):
+ data = wrapped
+
+ # Validate required fields
+ if not isinstance(data, dict):
+ return None, "codex output is not a JSON object"
+
+ required = ("score", "breakdown", "weaknesses", "suggestions", "strengths")
+ missing = [k for k in required if k not in data]
+ if missing:
+ return None, f"codex output missing fields: {', '.join(missing)}"
+
+ score = data.get("score")
+ if not isinstance(score, int) or score < 1 or score > 10:
+ return None, f"invalid score: {score} (must be integer 1-10)"
+
+ breakdown = data.get("breakdown", {})
+ for key in ("completeness", "correctness", "sequencing", "risk_awareness", "clarity"):
+ val = breakdown.get(key)
+ if not isinstance(val, int) or val < 0 or val > 2:
+ return None, f"invalid breakdown.{key}: {val} (must be integer 0-2)"
+
+ # Validate score equals breakdown sum
+ expected_sum = sum(breakdown.get(k, 0) for k in
+ ("completeness", "correctness", "sequencing", "risk_awareness", "clarity"))
+ if score != expected_sum:
+ return None, f"score mismatch: score={score} but breakdown sum={expected_sum}"
+
+ # Validate array contents
+ if not data.get("strengths"):
+ return None, "no strengths listed"
+ if score < 10 and not data.get("weaknesses"):
+ return None, "score < 10 but no weaknesses listed"
+
+ return data, None
diff --git a/plugins/planman/scripts/hook_utils.py b/plugins/planman/scripts/hook_utils.py
new file mode 100644
index 0000000..fef1229
--- /dev/null
+++ b/plugins/planman/scripts/hook_utils.py
@@ -0,0 +1,505 @@
+"""Shared evaluation helpers for planman hooks (PreToolUse(ExitPlanMode)).
+
+Contains the common evaluation flow:
+ detect plan -> load state -> check round limit -> assess -> format output
+
+Used by pre_exit_plan_hook.py and post_tool_hook.py.
+"""
+
+import json
+import os
+import sys
+import tempfile
+import time
+from datetime import datetime, timezone
+
+try:
+ import fcntl
+except ImportError:
+ fcntl = None
+
+import glob
+
+from config import load_config
+from evaluator import evaluate_plan
+from path_utils import normalize_path
+from state import (
+ compute_plan_hash,
+ load_state,
+ record_feedback,
+ save_state,
+ update_for_plan,
+)
+
+# ── Shared constants used by both hooks ──────────────────────────────
+
+MARKER_TEMPLATE = os.path.join(tempfile.gettempdir(), "planman-plan-{session_id}.json")
+
+
+def safe_session_id(session_id):
+ """Sanitize session_id for use in file paths. Falls back to 'default' if empty."""
+ safe = "".join(c for c in session_id if c.isalnum() or c in "-_")
+ safe = safe[:100]
+ return safe or "default"
+
+
+def _log_to_file(msg, cwd):
+ """Append a timestamped message to planman.log (race-safe)."""
+ if cwd:
+ log_path = os.path.join(cwd, ".claude", "planman.log")
+ else:
+ log_path = os.path.join(tempfile.gettempdir(), "planman.log")
+ try:
+ os.makedirs(os.path.dirname(log_path), exist_ok=True)
+ with open(log_path, "a", encoding="utf-8") as f:
+ if fcntl:
+ fcntl.flock(f, fcntl.LOCK_EX)
+ try:
+ ts = datetime.now(timezone.utc).isoformat()
+ f.write(f"[{ts}] {msg}\n")
+ finally:
+ if fcntl:
+ fcntl.flock(f, fcntl.LOCK_UN)
+ except Exception:
+ pass # Never crash on log failure
+
+
+def log(msg, config, cwd=None):
+ """Always log to file; also print to stderr if verbose."""
+ _log_to_file(msg, cwd)
+ if config and config.verbose:
+ print(f"[planman] {msg}", file=sys.stderr)
+
+
+def is_plan_filename(basename):
+ """Return True if basename looks like an actual plan file (not metadata)."""
+ lower = basename.lower()
+ if lower.startswith("."):
+ return False
+ skip_prefixes = ("readme", "template", "sample", "example", "backup")
+ for prefix in skip_prefixes:
+ if lower.startswith(prefix):
+ return False
+ return True
+
+
+def read_plan_text(path):
+ """Read plan file, return (text, skip_reason).
+
+ text is None when the file is empty/oversized/unreadable.
+ skip_reason is set only when the file was found but explicitly rejected.
+ """
+ _MAX_PLAN_SIZE = 1_000_000 # 1 MB
+ try:
+ size = os.path.getsize(path)
+ if size > _MAX_PLAN_SIZE:
+ return None, f"Plan file too large (>{_MAX_PLAN_SIZE // 1_000_000} MB): {path}"
+ with open(path, "r", encoding="utf-8") as f:
+ text = f.read()
+ return (text, None) if text.strip() else (None, None)
+ except (OSError, UnicodeDecodeError):
+ return None, None
+
+
+def read_marker_metadata(session_id):
+ """Read marker file and return (normalized_path_or_None, timestamp_float).
+
+ Returns (None, 0) for: missing file, corrupt JSON, missing keys,
+ non-numeric timestamp, future timestamp (clamped to 0).
+ """
+ safe_id = safe_session_id(session_id)
+ marker_path = MARKER_TEMPLATE.format(session_id=safe_id)
+ try:
+ with open(marker_path, "r", encoding="utf-8") as f:
+ marker = json.load(f)
+ except (OSError, json.JSONDecodeError):
+ return (None, 0)
+
+ if not isinstance(marker, dict):
+ return (None, 0)
+
+ plan_path = marker.get("plan_file_path")
+ if not plan_path or not isinstance(plan_path, str):
+ return (None, 0)
+
+ ts = marker.get("timestamp")
+ if ts is None:
+ return (None, 0)
+ try:
+ ts = float(ts)
+ except (ValueError, TypeError):
+ return (None, 0)
+
+ # Future timestamp → clamp to 0 (marker still trusted if file exists)
+ if ts > time.time():
+ ts = 0
+
+ return (normalize_path(plan_path), ts)
+
+
+def scan_plan_dirs(cwd, plan_dirs=None, project_local_only=False):
+ """Scan plan directories for most recently modified .md plan file.
+
+ Checks {cwd}/.claude/plans/, configured plan_dirs (resolved relative
+ to cwd), and optionally ~/.claude/plans/.
+ Returns the path of the best candidate or None.
+ """
+ home_plans = os.path.expanduser("~/.claude/plans")
+ cwd_plans = os.path.join(cwd, ".claude", "plans") if cwd else None
+
+ scan_dirs_list = []
+ if cwd_plans and os.path.isdir(cwd_plans):
+ scan_dirs_list.append(cwd_plans)
+
+ # Additional plan_dirs (resolved relative to cwd)
+ if plan_dirs and cwd:
+ for d in plan_dirs:
+ resolved = os.path.join(cwd, d) if not os.path.isabs(d) else d
+ resolved = os.path.realpath(resolved)
+ if os.path.isdir(resolved) and resolved not in scan_dirs_list:
+ scan_dirs_list.append(resolved)
+
+ if not project_local_only:
+ if os.path.isdir(home_plans):
+ real_home = os.path.realpath(home_plans)
+ if real_home not in [os.path.realpath(d) for d in scan_dirs_list]:
+ scan_dirs_list.append(home_plans)
+
+ md_files = []
+ for plans_dir in scan_dirs_list:
+ for f in glob.glob(os.path.join(plans_dir, "*.md")):
+ if is_plan_filename(os.path.basename(f)):
+ md_files.append(f)
+
+ if not md_files:
+ return None
+
+ return max(md_files, key=os.path.getmtime)
+
+
+def find_plan_file(session_id, cwd, config):
+ """Find the plan file path via session marker or fallback scan.
+
+ Returns (plan_file_path, plan_text, skip_reason) where skip_reason
+ is a human-readable message when the plan was found but rejected
+ (e.g. oversized), or None on success / when no plan exists at all.
+ """
+ _MARKER_TTL = 7200 # 2 hours
+ _STALENESS_TOLERANCE = 2 # seconds
+
+ plan_dirs = getattr(config, "plan_dirs", None)
+
+ # ── Gate: debug marker-only mode (internal escape hatch) ──
+ if os.environ.get("_PLANMAN_DEBUG_MARKER_ONLY"):
+ marker_path, _ = read_marker_metadata(session_id)
+ if marker_path and os.path.isfile(marker_path):
+ text, skip = read_plan_text(marker_path)
+ if text:
+ return (marker_path, text, None)
+ return (None, None, skip)
+ return (None, None, None)
+
+ # ── Step 1: Try marker (authoritative when fresh) ──
+ marker_plan_path, marker_ts = read_marker_metadata(session_id)
+ now = time.time()
+ expired = marker_ts > 0 and (now - marker_ts) > (_MARKER_TTL + _STALENESS_TOLERANCE)
+
+ if marker_plan_path and os.path.isfile(marker_plan_path) and not expired:
+ text, skip = read_plan_text(marker_plan_path)
+ if text:
+ log(
+ f"plan detection: source=marker, path={marker_plan_path}, "
+ f"session={safe_session_id(session_id)}",
+ config, cwd,
+ )
+ return (marker_plan_path, text, None)
+ if skip:
+ return (None, None, skip)
+
+ # ── Step 2: Scan fallback (marker missing/expired/file deleted) ──
+ scan_path = scan_plan_dirs(cwd, plan_dirs=plan_dirs, project_local_only=True)
+ if not scan_path:
+ scan_path = scan_plan_dirs(cwd, plan_dirs=plan_dirs, project_local_only=False)
+
+ reason = (
+ "marker_expired" if expired
+ else "marker_file_deleted" if marker_plan_path
+ else "no_marker"
+ )
+ if scan_path:
+ text, skip = read_plan_text(scan_path)
+ if text:
+ log(
+ f"plan detection: source=scan_fallback({reason}), "
+ f"path={scan_path}, session={safe_session_id(session_id)}",
+ config, cwd,
+ )
+ return (scan_path, text, None)
+ if skip:
+ return (None, None, skip)
+
+ return (None, None, None)
+
+
+def format_trend(history, current_score):
+ """Format score trend from history + current score.
+
+ Called BEFORE record_feedback() persists current_score to history,
+ so history contains only previous rounds' data.
+ """
+ if not history:
+ return ""
+ prev_score = history[-1].get("score")
+ if prev_score is None or current_score is None:
+ return ""
+ delta = current_score - prev_score
+ sign = "+" if delta > 0 else ""
+ return f"- **Previous**: {prev_score}/10 → {current_score}/10 ({sign}{delta})"
+
+
+def format_feedback(data, threshold, round_num, max_rounds, first_round=False, trend="",
+ min_rounds_remaining=None):
+ """Format evaluation result into structured, actionable feedback.
+
+ Args:
+ data: Evaluation result dict with score, breakdown, weaknesses, etc.
+ threshold: Minimum score to pass.
+ round_num: Current round number.
+ max_rounds: Maximum evaluation rounds.
+ first_round: Whether this is the first round (mandatory rejection).
+ trend: Trend line string from format_trend() (empty on round 1).
+ min_rounds_remaining: If set, number of rounds still required before plan can pass.
+ """
+ score = data.get("score", "?")
+ breakdown = data.get("breakdown") or {}
+ weaknesses = data.get("weaknesses") or []
+ suggestions = data.get("suggestions") or []
+ strengths = data.get("strengths") or []
+
+ # Header
+ lines = ["## Evaluation Result"]
+ if first_round:
+ lines.append(
+ f"- **Score**: {score}/10 (threshold: {threshold}) | "
+ f"**First-round review** — Round 1/{max_rounds}"
+ )
+ else:
+ lines.append(
+ f"- **Score**: {score}/10 (threshold: {threshold}) | "
+ f"Round {round_num}/{max_rounds}"
+ )
+ if trend:
+ lines.append(trend)
+ if min_rounds_remaining:
+ lines.append(f"- **Min rounds**: {min_rounds_remaining} more round(s) required before plan can pass")
+
+ # Issues — must fix
+ if weaknesses:
+ lines.append("")
+ lines.append("## Issues (must fix)")
+ for w in weaknesses:
+ lines.append(f"- {w}")
+
+ # Suggestions — improvements
+ if suggestions:
+ lines.append("")
+ lines.append("## Suggestions (improvements)")
+ for s in suggestions:
+ lines.append(f"- {s}")
+
+ return "\n".join(lines)
+
+
+def truncate_for_system_message(score, round_num, max_rounds, trend,
+ issues, suggestions, strengths, limit=2000):
+ """Build systemMessage: score + round + trend only (details are in reason)."""
+ parts = [f"Planman: {score}/10 | Round {round_num}/{max_rounds}"]
+ if trend:
+ parts.append(trend)
+ return "\n".join(parts)
+
+
+def format_approval(data):
+ """Format approval message."""
+ score = data.get("score", "?")
+ return f"Plan approved (score: {score}/10)."
+
+
+def run_evaluation(plan_text, session_id, config, cwd=None, plan_path=None, host="claude"):
+ """Run the full plan evaluation flow.
+
+ Returns a dict with keys:
+ action: "pass" | "block" | "skip"
+ reason: str or None (for block)
+ system_message: str or None
+ """
+ # Empty text — skip
+ if not plan_text or not plan_text.strip():
+ return {"action": "skip", "reason": None, "system_message": "Planman: Plan is empty — nothing to evaluate."}
+
+ # Load state and update round counter
+ state = load_state(session_id)
+
+ # ── Plan-mode path: full multi-round flow ────
+ state = update_for_plan(state, plan_text, plan_path)
+ log(f"round {state['round_count']}/{config.max_rounds}", config, cwd)
+
+ # Check round limit — pass through with clear proceed signal
+ if state["round_count"] > config.max_rounds:
+ log("max rounds exceeded — auto-approving", config, cwd)
+ state["plan_approved"] = True
+ try:
+ save_state(state)
+ except (OSError, ValueError) as e:
+ log(f"failed to save state: {e}", config, cwd)
+ last_score = state.get('last_score')
+ if last_score is not None:
+ score_msg = f"Last score was {last_score}/10 (threshold: {config.threshold}). "
+ else:
+ score_msg = ""
+ return {
+ "action": "pass",
+ "reason": None,
+ "system_message": (
+ f"Planman: Max evaluation rounds ({config.max_rounds}) reached. "
+ f"{score_msg}"
+ "Plan accepted — proceed with implementation."
+ ),
+ }
+
+ # Stress-test mode: skip external evaluation for the first N rounds.
+ # Round N+1 continues with normal evaluator scoring.
+ if config.stress_test and state["round_count"] <= config.stress_test:
+ prompt = config.stress_test_prompt
+ state = record_feedback(state, None, prompt, None)
+ try:
+ save_state(state)
+ except (OSError, ValueError) as e:
+ log(f"failed to save state: {e}", config, cwd)
+ log(f"stress-test mode: round {state['round_count']}/{config.stress_test} rejected without evaluation", config, cwd)
+ return {
+ "action": "block",
+ "reason": prompt,
+ "system_message": (
+ f"Planman: Stress-test mode — plan rejected for deep revision. "
+ f"Stress-test round {state['round_count']}/{config.stress_test} | "
+ f"Round {state['round_count']}/{config.max_rounds}."
+ ),
+ }
+
+ # Assess via resolved evaluator
+ previous_feedback = state.get("last_feedback")
+ result, error = evaluate_plan(
+ plan_text, config, previous_feedback, state["round_count"], cwd=cwd, host=host
+ )
+
+ if error:
+ log(f"evaluation error: {error}", config, cwd)
+ error_brief = error[:500] if len(error) > 500 else error
+ if config.fail_open:
+ state["plan_approved"] = True
+ try:
+ save_state(state)
+ except (OSError, ValueError) as e:
+ log(f"failed to save state: {e}", config, cwd)
+ return {
+ "action": "pass",
+ "reason": None,
+ "system_message": f"Planman: Evaluation failed ({error_brief}). Passing through (fail-open).",
+ }
+ else:
+ # Persist state so round_count advances (prevents infinite retry loops)
+ try:
+ save_state(state)
+ except (OSError, ValueError) as e:
+ log(f"failed to save state: {e}", config, cwd)
+ return {
+ "action": "block",
+ "reason": f"Planman evaluation failed: {error_brief}. Set PLANMAN_FAIL_OPEN=true to pass through on errors.",
+ "system_message": None,
+ }
+
+ assessment_score = result["score"]
+ weaknesses = result.get("weaknesses") or []
+ suggestions = result.get("suggestions") or []
+ strengths = result.get("strengths") or []
+
+ # Compute trend BEFORE recording feedback (history has prev rounds only)
+ trend = format_trend(state.get("history", []), assessment_score)
+
+ # First-round mandatory rejection
+ if state["round_count"] == 1:
+ feedback_text = format_feedback(
+ result, config.threshold, state["round_count"], config.max_rounds,
+ first_round=True, trend=trend,
+ )
+ sys_msg = truncate_for_system_message(
+ assessment_score, state["round_count"], config.max_rounds,
+ trend, weaknesses, suggestions, strengths,
+ )
+ state = record_feedback(state, assessment_score, feedback_text, result.get("breakdown"))
+ try:
+ save_state(state)
+ except OSError as e:
+ log(f"failed to save state: {e}", config, cwd)
+ log(f"first round: mandatory review ({assessment_score}/10)", config, cwd)
+ return {
+ "action": "block",
+ "reason": feedback_text,
+ "system_message": sys_msg,
+ }
+
+ # Min rounds enforcement: block even if score would pass
+ if config.min_rounds and state["round_count"] < config.min_rounds:
+ feedback_text = format_feedback(
+ result, config.threshold, state["round_count"], config.max_rounds,
+ trend=trend, min_rounds_remaining=config.min_rounds - state["round_count"],
+ )
+ sys_msg = truncate_for_system_message(
+ assessment_score, state["round_count"], config.max_rounds,
+ trend, weaknesses, suggestions, strengths,
+ )
+ state = record_feedback(state, assessment_score, feedback_text, result.get("breakdown"))
+ try:
+ save_state(state)
+ except OSError as e:
+ log(f"failed to save state: {e}", config, cwd)
+ log(f"min rounds: {state['round_count']}/{config.min_rounds} — blocking", config, cwd)
+ return {"action": "block", "reason": feedback_text, "system_message": sys_msg}
+
+ if assessment_score >= config.threshold:
+ # Plan passes (round >= 2) — preserve state but null feedback
+ state = record_feedback(state, assessment_score, None, result.get("breakdown"))
+ state["plan_approved"] = True
+ try:
+ save_state(state)
+ except OSError as e:
+ log(f"failed to save state: {e}", config, cwd)
+ log(f"plan accepted: {assessment_score}/10", config, cwd)
+ return {
+ "action": "pass",
+ "reason": None,
+ "system_message": f"Planman: {format_approval(result)}",
+ }
+ else:
+ # Plan rejected
+ feedback_text = format_feedback(
+ result, config.threshold, state["round_count"], config.max_rounds,
+ trend=trend,
+ )
+ sys_msg = truncate_for_system_message(
+ assessment_score, state["round_count"], config.max_rounds,
+ trend, weaknesses, suggestions, strengths,
+ )
+ state = record_feedback(state, assessment_score, feedback_text, result.get("breakdown"))
+ try:
+ save_state(state)
+ except OSError as e:
+ log(f"failed to save state: {e}", config, cwd)
+
+ log(f"plan rejected: {assessment_score}/{config.threshold}", config, cwd)
+ return {
+ "action": "block",
+ "reason": feedback_text,
+ "system_message": sys_msg,
+ }
diff --git a/plugins/planman/scripts/path_utils.py b/plugins/planman/scripts/path_utils.py
new file mode 100644
index 0000000..b97bd14
--- /dev/null
+++ b/plugins/planman/scripts/path_utils.py
@@ -0,0 +1,16 @@
+"""Shared path normalization — no planman imports to avoid cycles."""
+
+import os
+
+
+def normalize_path(path):
+ """Normalize a file path for consistent comparison.
+
+ Expands ~, resolves symlinks, returns absolute path.
+ """
+ if not path:
+ return path
+ try:
+ return os.path.realpath(os.path.expanduser(path))
+ except (ValueError, OSError):
+ return path
diff --git a/plugins/planman/scripts/state.py b/plugins/planman/scripts/state.py
new file mode 100644
index 0000000..ccb5202
--- /dev/null
+++ b/plugins/planman/scripts/state.py
@@ -0,0 +1,138 @@
+"""Multi-round session state tracking.
+
+State file per session at /planman-{session_id}.json.
+Tracks round count, last score/feedback, and plan hash for change detection.
+"""
+
+import hashlib
+import json
+import os
+import tempfile
+import time
+
+from path_utils import normalize_path as _normalize
+
+
+def _state_path(session_id):
+ """Return the state file path for a session."""
+ safe_id = "".join(c for c in session_id if c.isalnum() or c in "-_")
+ return os.path.join(tempfile.gettempdir(), f"planman-{safe_id or 'default'}.json")
+
+
+def compute_plan_hash(plan_text):
+ """Compute a short hash of the plan text for change detection.
+
+ Normalizes whitespace so minor formatting changes don't reset rounds.
+ """
+ normalized = " ".join(plan_text.split())
+ return hashlib.sha256(normalized.encode("utf-8")).hexdigest()[:16]
+
+
+def load_state(session_id):
+ """Load session state, returning a dict.
+
+ Returns default state if file doesn't exist or is corrupt.
+ """
+ path = _state_path(session_id)
+ try:
+ with open(path, "r", encoding="utf-8") as f:
+ data = json.load(f)
+ if isinstance(data, dict) and "session_id" in data:
+ return data
+ except (OSError, json.JSONDecodeError, ValueError):
+ pass
+
+ return {
+ "session_id": session_id,
+ "round_count": 0,
+ "last_score": None,
+ "last_feedback": None,
+ "plan_hash": None,
+ "history": [],
+ }
+
+
+def save_state(state):
+ """Save session state atomically via write-to-temp + os.replace()."""
+ session_id = state.get("session_id", "unknown")
+ path = _state_path(session_id)
+
+ tmp_fd, tmp_path = tempfile.mkstemp(
+ dir=tempfile.gettempdir(), prefix="planman-tmp-"
+ )
+ try:
+ with os.fdopen(tmp_fd, "w", encoding="utf-8") as f:
+ json.dump(state, f, indent=2, allow_nan=False)
+ os.replace(tmp_path, path)
+ except (OSError, ValueError):
+ # Best-effort cleanup
+ try:
+ os.unlink(tmp_path)
+ except OSError:
+ pass
+ raise
+
+
+def clear_state(session_id):
+ """Remove session state file."""
+ path = _state_path(session_id)
+ try:
+ os.unlink(path)
+ except OSError:
+ pass
+
+
+def update_for_plan(state, plan_text, plan_path=None):
+ """Update state for a new plan evaluation.
+
+ Resets round counter when:
+ - plan_approved flag is set (previous plan was approved)
+ - plan_path differs from stored path (new plan file)
+ Otherwise increments.
+ """
+ new_hash = compute_plan_hash(plan_text)
+ normalized_plan_path = _normalize(plan_path) if plan_path else None
+
+ # Fallback reset: previous plan was approved but PostToolUse didn't fire
+ # (e.g. user accepted with "clear context"). Consume the flag.
+ if state.get("plan_approved"):
+ state["round_count"] = 1
+ state["history"] = []
+ state.pop("plan_approved", None)
+ else:
+ stored_path = _normalize(state.get("plan_file_path")) if state.get("plan_file_path") else None
+
+ if normalized_plan_path and normalized_plan_path != stored_path:
+ # New plan file (or first file) = new plan
+ state["round_count"] = 1
+ state["history"] = []
+ else:
+ # Same file = revision
+ state["round_count"] = state.get("round_count", 0) + 1
+
+ state["plan_hash"] = new_hash
+ state["last_eval_time"] = time.time()
+ if normalized_plan_path:
+ state["plan_file_path"] = normalized_plan_path
+ return state
+
+
+def record_feedback(state, score, feedback, breakdown=None):
+ """Record evaluation results in state."""
+ state["last_score"] = score
+ state["last_feedback"] = feedback
+ if breakdown:
+ state["last_breakdown"] = breakdown
+ # Append to history
+ history = state.get("history", [])
+ history.append({
+ "round": state.get("round_count"),
+ "score": score,
+ "breakdown": breakdown,
+ "timestamp": time.time(),
+ })
+ # Cap at 20 entries
+ if len(history) > 20:
+ history = history[-20:]
+ state["history"] = history
+ return state
diff --git a/pytest.ini b/pytest.ini
new file mode 100644
index 0000000..2e51ca4
--- /dev/null
+++ b/pytest.ini
@@ -0,0 +1,11 @@
+[pytest]
+testpaths = tests
+pythonpath = scripts
+norecursedirs =
+ benchmark
+ logs
+ submission
+ .git
+ .claude
+ .codex
+ .pytest_cache
diff --git a/scripts/codex_stop_hook.py b/scripts/codex_stop_hook.py
new file mode 100644
index 0000000..0214dc3
--- /dev/null
+++ b/scripts/codex_stop_hook.py
@@ -0,0 +1,149 @@
+#!/usr/bin/env python3
+"""Codex Stop hook — evaluate latest plan-like content.
+
+This adapter is intentionally conservative. It only evaluates text that looks
+like a plan; if the Codex hook payload shape does not expose plan text, it
+fails open and logs the skip.
+"""
+
+import json
+import os
+import re
+import sys
+
+_scripts_dir = os.path.dirname(os.path.abspath(__file__))
+if _scripts_dir not in sys.path:
+ sys.path.insert(0, _scripts_dir)
+
+from config import load_config
+from hook_utils import log, run_evaluation, safe_session_id
+from evaluator import check_evaluator_installed
+
+
+_PLAN_RE = re.compile(
+ r"(?ims)(|implementation plan|^# .{0,80}plan\b|^\*\*summary\*\*|^\s*1\.\s+.+\n\s*2\.)"
+)
+
+
+def _walk_strings(value, path=""):
+ """Yield (path, string) pairs from nested JSON-like hook input."""
+ if isinstance(value, str):
+ yield path, value
+ elif isinstance(value, dict):
+ for key, child in value.items():
+ child_path = f"{path}.{key}" if path else str(key)
+ yield from _walk_strings(child, child_path)
+ elif isinstance(value, list):
+ for idx, child in enumerate(value):
+ yield from _walk_strings(child, f"{path}[{idx}]")
+
+
+def _extract_plan_text(hook_input):
+ """Extract the latest clearly plan-like text from a Codex hook payload."""
+ candidates = []
+ order = 0
+ for path, text in _walk_strings(hook_input):
+ order += 1
+ stripped = text.strip()
+ if len(stripped) < 80:
+ continue
+ lower_path = path.lower()
+ score = 0
+ if "plan" in lower_path:
+ score += 3
+ if any(part in lower_path for part in ("message", "content", "text", "markdown", "summary")):
+ score += 1
+ if _PLAN_RE.search(stripped):
+ score += 4
+ if score >= 4:
+ candidates.append((score, order, stripped))
+ if not candidates:
+ return None
+ candidates.sort(key=lambda item: (item[0], item[1]))
+ return candidates[-1][2]
+
+
+def _output_allow(system_message=None):
+ result = {}
+ if system_message:
+ result["systemMessage"] = system_message
+ if result:
+ json.dump(result, sys.stdout, ensure_ascii=True)
+ sys.exit(0)
+
+
+def _output_block(reason, system_message=None):
+ result = {
+ "decision": "block",
+ "reason": reason,
+ }
+ if system_message:
+ result["systemMessage"] = system_message
+ json.dump(result, sys.stdout, ensure_ascii=True)
+ sys.exit(0)
+
+
+def _main():
+ if os.environ.get("_PLANMAN_EVALUATOR"):
+ _output_allow()
+ return
+
+ try:
+ raw = sys.stdin.read()
+ except Exception:
+ raw = ""
+
+ try:
+ hook_input = json.loads(raw) if raw.strip() else {}
+ except (json.JSONDecodeError, ValueError):
+ hook_input = {}
+
+ cwd = hook_input.get("cwd") or os.getcwd()
+ config = load_config(cwd=cwd)
+ if not config.enabled:
+ _output_allow()
+ return
+
+ plan_text = _extract_plan_text(hook_input)
+ if not plan_text:
+ log("codex host: no plan-like content found in Stop payload", config, cwd)
+ _output_allow()
+ return
+
+ if not check_evaluator_installed(config, host="codex"):
+ log("codex host: evaluator CLI not installed — allowing", config, cwd)
+ _output_allow()
+ return
+
+ session_id = safe_session_id(hook_input.get("session_id") or hook_input.get("thread_id") or "codex")
+ log(f"codex host: evaluating plan-like content (session={session_id})", config, cwd)
+
+ result = run_evaluation(
+ plan_text,
+ session_id,
+ config,
+ cwd=cwd,
+ plan_path="codex:stop",
+ host="codex",
+ )
+
+ action = result["action"]
+ reason = result.get("reason")
+ sys_msg = result.get("system_message")
+
+ if action == "block":
+ _output_block(reason=reason, system_message=sys_msg)
+ else:
+ _output_allow(system_message=sys_msg)
+
+
+def main():
+ try:
+ _main()
+ except Exception as e:
+ print(f"[planman] FATAL in codex_stop_hook: {e}", file=sys.stderr)
+ sys.exit(0)
+
+
+if __name__ == "__main__":
+ main()
diff --git a/scripts/config.py b/scripts/config.py
index 3058c95..1fc37a8 100644
--- a/scripts/config.py
+++ b/scripts/config.py
@@ -1,7 +1,7 @@
"""Configuration loader for planman.
Loads settings from two sources (env vars override file):
- 1. .claude/planman.jsonc (project-level, supports // comments)
+ 1. .planman.jsonc or .claude/planman.jsonc (project-level, supports // comments)
2. PLANMAN_* environment variables (highest priority)
"""
@@ -36,6 +36,9 @@
"max_rounds": 3,
"min_rounds": 0,
"model": "",
+ "evaluator": "auto",
+ "codex_bin": "codex",
+ "claude_bin": "claude",
"fail_open": True,
"enabled": True,
"custom_rubric": "",
@@ -120,6 +123,9 @@ class Config:
"max_rounds",
"min_rounds",
"model",
+ "evaluator",
+ "codex_bin",
+ "claude_bin",
"fail_open",
"enabled",
"rubric",
@@ -139,6 +145,9 @@ def __init__(self, **kwargs):
self.max_rounds = kwargs.get("max_rounds", DEFAULTS["max_rounds"])
self.min_rounds = kwargs.get("min_rounds", DEFAULTS["min_rounds"])
self.model = kwargs.get("model", DEFAULTS["model"])
+ self.evaluator = kwargs.get("evaluator", DEFAULTS["evaluator"])
+ self.codex_bin = kwargs.get("codex_bin", DEFAULTS["codex_bin"])
+ self.claude_bin = kwargs.get("claude_bin", DEFAULTS["claude_bin"])
self.fail_open = kwargs.get("fail_open", DEFAULTS["fail_open"])
self.enabled = kwargs.get("enabled", DEFAULTS["enabled"])
self.rubric = kwargs.get("rubric", "") or DEFAULT_RUBRIC
@@ -163,12 +172,16 @@ def _strip_jsonc_comments(text):
def _load_file_config(cwd=None):
- """Load .claude/planman.jsonc (or .json fallback) if it exists."""
+ """Load project planman config if it exists."""
base = cwd or "."
- path = os.path.join(base, ".claude", "planman.jsonc")
- if not os.path.isfile(path):
- path = os.path.join(base, ".claude", "planman.json")
- if not os.path.isfile(path):
+ candidate_paths = (
+ os.path.join(base, ".planman.jsonc"),
+ os.path.join(base, ".planman.json"),
+ os.path.join(base, ".claude", "planman.jsonc"),
+ os.path.join(base, ".claude", "planman.json"),
+ )
+ path = next((p for p in candidate_paths if os.path.isfile(p)), None)
+ if path is None:
return {}
try:
with open(path, "r", encoding="utf-8") as f:
@@ -189,6 +202,9 @@ def _load_env_overrides():
"PLANMAN_MAX_ROUNDS": ("max_rounds", _coerce_int),
"PLANMAN_MIN_ROUNDS": ("min_rounds", _coerce_int),
"PLANMAN_MODEL": ("model", str),
+ "PLANMAN_EVALUATOR": ("evaluator", str),
+ "PLANMAN_CODEX_BIN": ("codex_bin", str),
+ "PLANMAN_CLAUDE_BIN": ("claude_bin", str),
"PLANMAN_FAIL_OPEN": ("fail_open", _coerce_bool),
"PLANMAN_ENABLED": ("enabled", _coerce_bool),
"PLANMAN_RUBRIC": ("custom_rubric", str),
@@ -237,6 +253,11 @@ def load_config(cwd=None):
# Coerce stress_test to int (False→0, True→1, number→int)
merged["stress_test"] = _coerce_stress_test(merged["stress_test"], "stress_test")
+ evaluator = str(merged.get("evaluator", "auto")).lower().strip()
+ if evaluator not in ("auto", "codex", "claude"):
+ evaluator = DEFAULTS["evaluator"]
+ merged["evaluator"] = evaluator
+
# Coerce list fields
merged["plan_dirs"] = _coerce_list(merged.get("plan_dirs", []), "plan_dirs")
merged["exec_patterns"] = _coerce_list(merged.get("exec_patterns", []), "exec_patterns")
@@ -248,6 +269,9 @@ def load_config(cwd=None):
max_rounds=merged["max_rounds"],
min_rounds=merged["min_rounds"],
model=merged["model"],
+ evaluator=merged["evaluator"],
+ codex_bin=merged.get("codex_bin", "codex") or "codex",
+ claude_bin=merged.get("claude_bin", "claude") or "claude",
fail_open=merged["fail_open"],
enabled=merged["enabled"],
rubric=merged.get("custom_rubric", ""),
diff --git a/scripts/evaluator.py b/scripts/evaluator.py
index 17c4dc0..453705d 100644
--- a/scripts/evaluator.py
+++ b/scripts/evaluator.py
@@ -1,7 +1,6 @@
-"""Codex CLI evaluation via subprocess.
+"""Plan evaluation via coding-agent CLI subprocesses.
-Calls `codex exec` with `--output-schema` for structured JSON scoring.
-No API keys required — uses ChatGPT subscription auth via the codex CLI.
+Calls Codex or Claude Code with structured JSON scoring.
"""
import json
@@ -10,13 +9,17 @@
import subprocess
import sys
-PLUGIN_ROOT = os.environ.get("CLAUDE_PLUGIN_ROOT", "") or os.path.dirname(
- os.path.dirname(os.path.abspath(__file__))
+PLUGIN_ROOT = (
+ os.environ.get("CLAUDE_PLUGIN_ROOT", "")
+ or os.environ.get("CODEX_PLUGIN_ROOT", "")
+ or os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
)
_CODEX_BIN = "codex"
+_CLAUDE_BIN = "claude"
_codex_available = None # cached result
+_claude_available = None # cached result
# Known codex error patterns → actionable messages
_KNOWN_ERRORS = [
@@ -44,10 +47,23 @@ def _extract_codex_error(stderr):
def check_codex_installed(codex_path=None):
"""Check if codex CLI is installed. Result is cached."""
global _codex_available
- if _codex_available is not None:
+ if codex_path is None and _codex_available is not None:
return _codex_available
- _codex_available = shutil.which(codex_path or _CODEX_BIN) is not None
- return _codex_available
+ available = shutil.which(codex_path or _CODEX_BIN) is not None
+ if codex_path is None:
+ _codex_available = available
+ return available
+
+
+def check_claude_installed(claude_path=None):
+ """Check if Claude Code CLI is installed. Result is cached."""
+ global _claude_available
+ if claude_path is None and _claude_available is not None:
+ return _claude_available
+ available = shutil.which(claude_path or _CLAUDE_BIN) is not None
+ if claude_path is None:
+ _claude_available = available
+ return available
def reset_codex_cache():
@@ -56,8 +72,48 @@ def reset_codex_cache():
_codex_available = None
+def reset_claude_cache():
+ """Reset the cached Claude CLI availability check (for testing)."""
+ global _claude_available
+ _claude_available = None
+
+
+def detect_host(hook_input=None):
+ """Infer the host agent from hook/runtime context."""
+ if os.environ.get("CLAUDE_PLUGIN_ROOT"):
+ return "claude"
+ if os.environ.get("CODEX_HOME") or os.environ.get("CODEX_SANDBOX"):
+ return "codex"
+ if isinstance(hook_input, dict):
+ tool_name = hook_input.get("tool_name")
+ if tool_name == "ExitPlanMode" or "CLAUDE_PLUGIN_ROOT" in hook_input:
+ return "claude"
+ event = str(hook_input.get("hook_event_name") or hook_input.get("hookEventName") or "")
+ if event in ("Stop", "UserPromptSubmit", "SessionStart"):
+ return "codex"
+ if "codex" in str(hook_input.get("source", "")).lower():
+ return "codex"
+ return "claude"
+
+
+def resolve_evaluator(config, host="claude"):
+ """Resolve configured evaluator provider for a detected host."""
+ evaluator = getattr(config, "evaluator", "auto") or "auto"
+ if evaluator != "auto":
+ return evaluator
+ return "claude" if host == "codex" else "codex"
+
+
+def check_evaluator_installed(config, host="claude"):
+ """Check whether the resolved evaluator CLI is installed."""
+ provider = resolve_evaluator(config, host=host)
+ if provider == "claude":
+ return check_claude_installed(getattr(config, "claude_bin", None) or _CLAUDE_BIN)
+ return check_codex_installed(getattr(config, "codex_bin", None) or _CODEX_BIN)
+
+
def build_prompt(plan_text, rubric, previous_feedback=None, round_number=1,
- context=None, source_verify=True):
+ context=None, source_verify=True, provider="codex"):
"""Build the evaluation prompt for codex exec."""
context_section = ""
if context:
@@ -76,11 +132,14 @@ def build_prompt(plan_text, rubric, previous_feedback=None, round_number=1,
f"{plan_text}\n"
)
if source_verify:
+ read_hint = "`cat ` or search with `rg`"
+ if provider == "claude":
+ read_hint = "read/search tools"
prompt += (
"\n## Source Verification\n\n"
"You have read-only access to the project filesystem (cwd = project root).\n"
"When evaluating correctness and completeness:\n"
- "1. If the plan references specific files, read them with `cat ` or search with `rg`\n"
+ f"1. If the plan references specific files, verify them with {read_hint}\n"
"2. Verify that APIs, function signatures, and module structures mentioned in the plan exist\n"
"3. Check that the plan's assumptions about the codebase are accurate\n"
"4. Note discrepancies between the plan and actual code as correctness issues\n"
@@ -96,19 +155,42 @@ def build_prompt(plan_text, rubric, previous_feedback=None, round_number=1,
return prompt
-def evaluate_plan(plan_text, config, previous_feedback=None, round_number=1, cwd=None):
- """Evaluate a plan via codex exec with structured output.
+def _schema_path():
+ path = os.path.join(PLUGIN_ROOT, "schemas", "evaluation.json")
+ if not os.path.isfile(path):
+ return None, f"schema file not found: {path}. Check plugin root environment."
+ return path, None
+
+
+def _subprocess_env():
+ env = os.environ.copy()
+ env["_PLANMAN_EVALUATOR"] = "1"
+ return env
+
+
+def evaluate_plan(plan_text, config, previous_feedback=None, round_number=1, cwd=None, host="claude"):
+ """Evaluate a plan with the configured provider and structured output.
Returns (result_dict, error_string). On success error_string is None.
On failure result_dict is None and error_string describes the problem.
"""
- if not check_codex_installed():
- return None, "codex CLI not found. Install: npm install -g @openai/codex"
+ provider = resolve_evaluator(config, host=host)
+ if provider == "claude":
+ return evaluate_plan_claude(plan_text, config, previous_feedback, round_number, cwd)
+ return evaluate_plan_codex(plan_text, config, previous_feedback, round_number, cwd)
+
+
+def evaluate_plan_codex(plan_text, config, previous_feedback=None, round_number=1, cwd=None):
+ """Evaluate a plan via codex exec with structured output."""
+ codex_bin = getattr(config, "codex_bin", None) or _CODEX_BIN
+ if not check_codex_installed(codex_bin):
+ return None, f"codex CLI not found. Install: npm install -g @openai/codex"
prompt = build_prompt(
plan_text, config.rubric, previous_feedback, round_number,
context=config.context,
source_verify=getattr(config, "source_verify", True),
+ provider="codex",
)
_MAX_PROMPT_SIZE = 2_000_000 # 2MB hard cap
@@ -117,18 +199,18 @@ def evaluate_plan(plan_text, config, previous_feedback=None, round_number=1, cwd
effective_timeout = 570 # 600s hook timeout − 30s margin
- schema_path = os.path.join(PLUGIN_ROOT, "schemas", "evaluation.json")
-
- if not os.path.isfile(schema_path):
- return None, f"schema file not found: {schema_path}. Check CLAUDE_PLUGIN_ROOT."
+ schema_path, schema_error = _schema_path()
+ if schema_error:
+ return None, schema_error
cmd = [
- _CODEX_BIN,
+ codex_bin,
"exec", "-", # Read prompt from stdin
"--output-schema", schema_path,
"--sandbox", "read-only",
"--skip-git-repo-check",
"--ephemeral", # Don't persist session files
+ "--disable", "hooks",
]
if config.model:
cmd.extend(["-m", config.model])
@@ -142,6 +224,7 @@ def evaluate_plan(plan_text, config, previous_feedback=None, round_number=1, cwd
errors='replace', # prevent UnicodeDecodeError on bad codex output
timeout=effective_timeout,
cwd=cwd or os.getcwd(),
+ env=_subprocess_env(),
)
except subprocess.TimeoutExpired:
return None, f"codex timed out ({effective_timeout}s). Try a shorter plan or check codex CLI health."
@@ -164,6 +247,76 @@ def evaluate_plan(plan_text, config, previous_feedback=None, round_number=1, cwd
return parse_codex_output(result.stdout)
+def evaluate_plan_claude(plan_text, config, previous_feedback=None, round_number=1, cwd=None):
+ """Evaluate a plan via Claude Code print mode with structured output."""
+ claude_bin = getattr(config, "claude_bin", None) or _CLAUDE_BIN
+ if not check_claude_installed(claude_bin):
+ return None, "claude CLI not found. Install and authenticate Claude Code."
+
+ prompt = build_prompt(
+ plan_text, config.rubric, previous_feedback, round_number,
+ context=config.context,
+ source_verify=getattr(config, "source_verify", True),
+ provider="claude",
+ )
+
+ _MAX_PROMPT_SIZE = 2_000_000
+ if len(prompt) > _MAX_PROMPT_SIZE:
+ return None, f"prompt too large ({len(prompt) // 1024}KB > 2MB). Reduce max_rounds or plan size."
+
+ schema_path, schema_error = _schema_path()
+ if schema_error:
+ return None, schema_error
+
+ try:
+ with open(schema_path, "r", encoding="utf-8") as f:
+ schema_text = f.read()
+ except OSError as e:
+ return None, f"schema file unreadable: {e}"
+
+ cmd = [
+ claude_bin,
+ "-p",
+ "--output-format", "json",
+ "--json-schema", schema_text,
+ "--no-session-persistence",
+ "--tools", "Read,Grep,Glob",
+ ]
+ if config.model:
+ cmd.extend(["--model", config.model])
+
+ effective_timeout = 570
+ try:
+ result = subprocess.run(
+ cmd,
+ input=prompt,
+ capture_output=True,
+ text=True,
+ errors="replace",
+ timeout=effective_timeout,
+ cwd=cwd or os.getcwd(),
+ env=_subprocess_env(),
+ )
+ except subprocess.TimeoutExpired:
+ return None, f"claude timed out ({effective_timeout}s). Try a shorter plan or check Claude Code CLI health."
+ except FileNotFoundError:
+ reset_claude_cache()
+ return None, "claude not found. Install and authenticate Claude Code."
+ except (OSError, UnicodeDecodeError) as e:
+ return None, f"failed to run claude: {e}"
+
+ if config.verbose:
+ print(f"[planman] claude exit code: {result.returncode}", file=sys.stderr)
+ if result.stderr:
+ verbose_limit = 4000 if result.returncode != 0 else 2000
+ print(f"[planman] claude stderr (last {verbose_limit}): {result.stderr[-verbose_limit:]}", file=sys.stderr)
+
+ if result.returncode != 0:
+ return None, f"claude -p failed (exit {result.returncode}): {_extract_codex_error(result.stderr)}"
+
+ return parse_codex_output(result.stdout)
+
+
def parse_codex_output(stdout):
"""Parse structured JSON from codex exec stdout.
@@ -177,6 +330,17 @@ def parse_codex_output(stdout):
except json.JSONDecodeError as e:
return None, f"codex returned malformed output. Set PLANMAN_VERBOSE=true for details."
+ # Claude's JSON mode can wrap the final structured response.
+ if isinstance(data, dict) and "result" in data and not any(k in data for k in ("score", "breakdown")):
+ wrapped = data.get("result")
+ if isinstance(wrapped, str):
+ try:
+ data = json.loads(wrapped)
+ except json.JSONDecodeError:
+ return None, "claude returned malformed structured result"
+ elif isinstance(wrapped, dict):
+ data = wrapped
+
# Validate required fields
if not isinstance(data, dict):
return None, "codex output is not a JSON object"
diff --git a/scripts/hook_utils.py b/scripts/hook_utils.py
index e25809c..fef1229 100644
--- a/scripts/hook_utils.py
+++ b/scripts/hook_utils.py
@@ -21,7 +21,7 @@
import glob
from config import load_config
-from evaluator import check_codex_installed, evaluate_plan
+from evaluator import evaluate_plan
from path_utils import normalize_path
from state import (
compute_plan_hash,
@@ -325,7 +325,7 @@ def format_approval(data):
return f"Plan approved (score: {score}/10)."
-def run_evaluation(plan_text, session_id, config, cwd=None, plan_path=None):
+def run_evaluation(plan_text, session_id, config, cwd=None, plan_path=None, host="claude"):
"""Run the full plan evaluation flow.
Returns a dict with keys:
@@ -367,8 +367,8 @@ def run_evaluation(plan_text, session_id, config, cwd=None, plan_path=None):
),
}
- # Stress-test mode: skip Codex for the first N rounds, block with prompt as reason.
- # Round N+1 continues with normal Codex evaluation.
+ # Stress-test mode: skip external evaluation for the first N rounds.
+ # Round N+1 continues with normal evaluator scoring.
if config.stress_test and state["round_count"] <= config.stress_test:
prompt = config.stress_test_prompt
state = record_feedback(state, None, prompt, None)
@@ -387,10 +387,10 @@ def run_evaluation(plan_text, session_id, config, cwd=None, plan_path=None):
),
}
- # Assess via codex
+ # Assess via resolved evaluator
previous_feedback = state.get("last_feedback")
result, error = evaluate_plan(
- plan_text, config, previous_feedback, state["round_count"], cwd=cwd
+ plan_text, config, previous_feedback, state["round_count"], cwd=cwd, host=host
)
if error:
diff --git a/scripts/post_exit_plan_hook.py b/scripts/post_exit_plan_hook.py
index cac0c9c..3f876df 100644
--- a/scripts/post_exit_plan_hook.py
+++ b/scripts/post_exit_plan_hook.py
@@ -31,6 +31,9 @@
def _main():
+ if os.environ.get("_PLANMAN_EVALUATOR"):
+ sys.exit(0)
+
try:
raw = sys.stdin.read()
except Exception:
diff --git a/scripts/post_tool_hook.py b/scripts/post_tool_hook.py
index 02a4131..de297fd 100644
--- a/scripts/post_tool_hook.py
+++ b/scripts/post_tool_hook.py
@@ -74,6 +74,9 @@ def _write_marker(file_path, session_id):
def _main():
+ if os.environ.get("_PLANMAN_EVALUATOR"):
+ sys.exit(0)
+
# Read hook input from stdin
try:
raw = sys.stdin.read()
diff --git a/scripts/pre_ask_hook.py b/scripts/pre_ask_hook.py
index c990a9a..82cba24 100644
--- a/scripts/pre_ask_hook.py
+++ b/scripts/pre_ask_hook.py
@@ -52,6 +52,10 @@ def _output_allow(system_message=None):
def _main():
+ if os.environ.get("_PLANMAN_EVALUATOR"):
+ _output_allow()
+ return
+
try:
raw = sys.stdin.read()
except Exception:
diff --git a/scripts/pre_exec_gate_hook.py b/scripts/pre_exec_gate_hook.py
index b29f8ab..6f47e7c 100644
--- a/scripts/pre_exec_gate_hook.py
+++ b/scripts/pre_exec_gate_hook.py
@@ -83,6 +83,10 @@ def _get_patterns_for_tool(tool_name, config):
def _main():
+ if os.environ.get("_PLANMAN_EVALUATOR"):
+ _fast_exit_allow()
+ return
+
# Read stdin
try:
raw = sys.stdin.read()
@@ -142,12 +146,12 @@ def _main():
# ── Evaluation path: pattern matched ──
from hook_utils import find_plan_file, log, run_evaluation
- from evaluator import check_codex_installed
+ from evaluator import check_evaluator_installed
log(f"exec gate: {tool_name} pattern matched, text={matchable!r}", config, cwd)
- if not check_codex_installed():
- log("codex CLI not installed — allowing", config, cwd)
+ if not check_evaluator_installed(config, host="claude"):
+ log("evaluator CLI not installed — allowing", config, cwd)
_output_allow()
return
@@ -173,7 +177,7 @@ def _main():
log(f"evaluating plan from {plan_path} before exec (session={session_id})", config, cwd)
result = run_evaluation(
- plan_text, session_id, config, cwd=cwd, plan_path=plan_path
+ plan_text, session_id, config, cwd=cwd, plan_path=plan_path, host="claude"
)
action = result["action"]
diff --git a/scripts/pre_exit_plan_hook.py b/scripts/pre_exit_plan_hook.py
index bafd827..70da871 100644
--- a/scripts/pre_exit_plan_hook.py
+++ b/scripts/pre_exit_plan_hook.py
@@ -26,7 +26,7 @@
sys.path.insert(0, _scripts_dir)
from config import load_config
-from evaluator import check_codex_installed
+from evaluator import check_evaluator_installed
from hook_utils import find_plan_file, log, run_evaluation, safe_session_id
@@ -57,6 +57,10 @@ def _output_allow(system_message=None):
def _main():
+ if os.environ.get("_PLANMAN_EVALUATOR"):
+ _output_allow()
+ return
+
# Read hook input from stdin
try:
raw = sys.stdin.read()
@@ -81,8 +85,8 @@ def _main():
_output_allow()
return
- if not check_codex_installed():
- log("codex CLI not installed — passing through", config, cwd)
+ if not check_evaluator_installed(config, host="claude"):
+ log("evaluator CLI not installed — passing through", config, cwd)
_output_allow()
return
@@ -106,7 +110,7 @@ def _main():
# Run evaluation
result = run_evaluation(
- plan_text, session_id, config, cwd=cwd, plan_path=plan_path
+ plan_text, session_id, config, cwd=cwd, plan_path=plan_path, host="claude"
)
action = result["action"]
diff --git a/tests/test_codex_plugin_package.py b/tests/test_codex_plugin_package.py
new file mode 100644
index 0000000..132f579
--- /dev/null
+++ b/tests/test_codex_plugin_package.py
@@ -0,0 +1,59 @@
+"""Tests for the packaged Codex plugin wrapper."""
+
+import filecmp
+import json
+import os
+import unittest
+
+
+ROOT = os.path.abspath(os.path.join(os.path.dirname(__file__), ".."))
+PACKAGE_ROOT = os.path.join(ROOT, "plugins", "planman")
+
+
+class TestCodexPluginPackage(unittest.TestCase):
+ def test_marketplace_points_to_packaged_plugin(self):
+ path = os.path.join(ROOT, ".agents", "plugins", "marketplace.json")
+ with open(path, encoding="utf-8") as f:
+ marketplace = json.load(f)
+
+ plugin = marketplace["plugins"][0]
+ self.assertEqual(plugin["name"], "planman")
+ self.assertEqual(plugin["source"]["path"], "./plugins/planman")
+
+ def test_packaged_manifest_matches_root_manifest(self):
+ self.assertTrue(
+ filecmp.cmp(
+ os.path.join(ROOT, ".codex-plugin", "plugin.json"),
+ os.path.join(PACKAGE_ROOT, ".codex-plugin", "plugin.json"),
+ shallow=False,
+ )
+ )
+
+ def test_packaged_hook_config_matches_root_hook_config(self):
+ self.assertTrue(
+ filecmp.cmp(
+ os.path.join(ROOT, "hooks.json"),
+ os.path.join(PACKAGE_ROOT, "hooks.json"),
+ shallow=False,
+ )
+ )
+
+ def test_packaged_runtime_files_match_sources(self):
+ files = [
+ ("scripts/codex_stop_hook.py", "scripts/codex_stop_hook.py"),
+ ("scripts/config.py", "scripts/config.py"),
+ ("scripts/evaluator.py", "scripts/evaluator.py"),
+ ("scripts/hook_utils.py", "scripts/hook_utils.py"),
+ ("scripts/path_utils.py", "scripts/path_utils.py"),
+ ("scripts/state.py", "scripts/state.py"),
+ ("schemas/evaluation.json", "schemas/evaluation.json"),
+ ]
+ for source, packaged in files:
+ with self.subTest(source=source):
+ self.assertTrue(
+ filecmp.cmp(
+ os.path.join(ROOT, source),
+ os.path.join(PACKAGE_ROOT, packaged),
+ shallow=False,
+ )
+ )
diff --git a/tests/test_codex_stop_hook.py b/tests/test_codex_stop_hook.py
new file mode 100644
index 0000000..ebc7525
--- /dev/null
+++ b/tests/test_codex_stop_hook.py
@@ -0,0 +1,120 @@
+"""Tests for Codex Stop hook adapter."""
+
+import json
+import os
+import sys
+import unittest
+from io import StringIO
+from unittest.mock import patch
+
+sys.path.insert(0, os.path.join(os.path.dirname(__file__), "..", "scripts"))
+
+
+PLAN_TEXT = """\
+# Implementation Plan
+
+## Summary
+Add cross-agent plan evaluation.
+
+1. Refactor the evaluator provider.
+2. Add Claude provider support.
+3. Add Codex host adapter.
+"""
+
+
+class TestCodexStopHook(unittest.TestCase):
+ def setUp(self):
+ self._saved = {}
+ for k in list(os.environ):
+ if k.startswith("PLANMAN_") or k == "_PLANMAN_EVALUATOR":
+ self._saved[k] = os.environ.pop(k)
+ os.environ["PLANMAN_ENABLED"] = "true"
+
+ def tearDown(self):
+ for k in list(os.environ):
+ if k.startswith("PLANMAN_") or k == "_PLANMAN_EVALUATOR":
+ del os.environ[k]
+ os.environ.update(self._saved)
+
+ def _run_hook(self, hook_input):
+ import codex_stop_hook
+
+ stdout_capture = StringIO()
+ with patch("sys.stdin", StringIO(json.dumps(hook_input))), \
+ patch("sys.stdout", stdout_capture), \
+ self.assertRaises(SystemExit) as ctx:
+ codex_stop_hook.main()
+ return stdout_capture.getvalue(), ctx.exception.code
+
+ def test_extract_plan_text_from_payload(self):
+ import codex_stop_hook
+
+ text = codex_stop_hook._extract_plan_text({
+ "hookEventName": "Stop",
+ "assistant": {"message": PLAN_TEXT},
+ })
+ self.assertEqual(text, PLAN_TEXT.strip())
+
+ def test_extract_plan_text_prefers_latest_candidate(self):
+ import codex_stop_hook
+
+ older_plan = PLAN_TEXT + "\n" + ("Extra detail.\n" * 20)
+ latest_plan = """Here is the revised plan:
+
+# Plan
+
+1. Keep the Codex hook contract.
+2. Use authenticated Claude print mode.
+3. Verify package parity.
+"""
+ text = codex_stop_hook._extract_plan_text({
+ "messages": [
+ {"content": older_plan},
+ {"content": latest_plan},
+ ],
+ })
+ self.assertEqual(text, latest_plan.strip())
+
+ def test_extract_plan_text_matches_multiline_heading(self):
+ import codex_stop_hook
+
+ text = codex_stop_hook._extract_plan_text({
+ "assistant": {"message": "Intro text.\n\n" + PLAN_TEXT},
+ })
+ self.assertEqual(text, ("Intro text.\n\n" + PLAN_TEXT).strip())
+
+ def test_no_plan_exits_silently(self):
+ output, code = self._run_hook({"hookEventName": "Stop", "cwd": os.getcwd(), "message": "done"})
+ self.assertEqual(code, 0)
+ self.assertEqual(output, "")
+
+ @patch("codex_stop_hook.check_evaluator_installed", return_value=True)
+ @patch("codex_stop_hook.run_evaluation")
+ def test_plan_found_blocks_with_feedback(self, mock_eval, mock_check):
+ mock_eval.return_value = {
+ "action": "block",
+ "reason": "Fix the plan",
+ "system_message": "Planman: 6/10",
+ }
+ output, code = self._run_hook({
+ "hookEventName": "Stop",
+ "thread_id": "abc",
+ "cwd": os.getcwd(),
+ "assistant": {"message": PLAN_TEXT},
+ })
+ self.assertEqual(code, 0)
+ parsed = json.loads(output)
+ self.assertEqual(parsed["decision"], "block")
+ self.assertIn("Fix the plan", parsed["reason"])
+ mock_eval.assert_called_once()
+ self.assertEqual(mock_eval.call_args[1]["host"], "codex")
+
+ def test_evaluator_sentinel_exits_silently(self):
+ os.environ["_PLANMAN_EVALUATOR"] = "1"
+ output, code = self._run_hook({"hookEventName": "Stop", "assistant": {"message": PLAN_TEXT}})
+ self.assertEqual(code, 0)
+ self.assertEqual(output, "")
+
+
+if __name__ == "__main__":
+ unittest.main()
diff --git a/tests/test_config.py b/tests/test_config.py
index 7000607..a0382fd 100644
--- a/tests/test_config.py
+++ b/tests/test_config.py
@@ -121,6 +121,23 @@ def test_model_override(self):
cfg = load_config()
self.assertEqual(cfg.model, "gpt-4o")
+ def test_evaluator_override(self):
+ os.environ["PLANMAN_EVALUATOR"] = "claude"
+ cfg = load_config()
+ self.assertEqual(cfg.evaluator, "claude")
+
+ def test_invalid_evaluator_falls_back_to_auto(self):
+ os.environ["PLANMAN_EVALUATOR"] = "self"
+ cfg = load_config()
+ self.assertEqual(cfg.evaluator, "auto")
+
+ def test_evaluator_bins_override(self):
+ os.environ["PLANMAN_CODEX_BIN"] = "/opt/bin/codex"
+ os.environ["PLANMAN_CLAUDE_BIN"] = "/opt/bin/claude"
+ cfg = load_config()
+ self.assertEqual(cfg.codex_bin, "/opt/bin/codex")
+ self.assertEqual(cfg.claude_bin, "/opt/bin/claude")
+
def test_verbose_true(self):
os.environ["PLANMAN_VERBOSE"] = "1"
cfg = load_config()
@@ -152,6 +169,13 @@ def tearDown(self):
shutil.rmtree(self._tmpdir, ignore_errors=True)
def test_file_config_loads(self):
+ with open(".planman.jsonc", "w") as f:
+ json.dump({"threshold": 8, "max_rounds": 5}, f)
+ cfg = load_config()
+ self.assertEqual(cfg.threshold, 8)
+ self.assertEqual(cfg.max_rounds, 5)
+
+ def test_legacy_claude_file_config_loads(self):
os.makedirs(".claude", exist_ok=True)
with open(".claude/planman.jsonc", "w") as f:
json.dump({"threshold": 8, "max_rounds": 5}, f)
@@ -159,6 +183,15 @@ def test_file_config_loads(self):
self.assertEqual(cfg.threshold, 8)
self.assertEqual(cfg.max_rounds, 5)
+ def test_neutral_config_takes_priority_over_legacy_claude_config(self):
+ with open(".planman.jsonc", "w") as f:
+ json.dump({"threshold": 6}, f)
+ os.makedirs(".claude", exist_ok=True)
+ with open(".claude/planman.jsonc", "w") as f:
+ json.dump({"threshold": 9}, f)
+ cfg = load_config()
+ self.assertEqual(cfg.threshold, 6)
+
def test_env_overrides_file(self):
os.makedirs(".claude", exist_ok=True)
with open(".claude/planman.jsonc", "w") as f:
@@ -190,18 +223,16 @@ def test_jsonc_with_comments(self):
def test_json_fallback(self):
"""planman.json is loaded if planman.jsonc doesn't exist."""
- os.makedirs(".claude", exist_ok=True)
- with open(".claude/planman.json", "w") as f:
+ with open(".planman.json", "w") as f:
json.dump({"threshold": 6}, f)
cfg = load_config()
self.assertEqual(cfg.threshold, 6)
def test_jsonc_takes_priority_over_json(self):
"""planman.jsonc is preferred when both exist."""
- os.makedirs(".claude", exist_ok=True)
- with open(".claude/planman.jsonc", "w") as f:
+ with open(".planman.jsonc", "w") as f:
json.dump({"threshold": 9}, f)
- with open(".claude/planman.json", "w") as f:
+ with open(".planman.json", "w") as f:
json.dump({"threshold": 4}, f)
cfg = load_config()
self.assertEqual(cfg.threshold, 9)
diff --git a/tests/test_evaluator.py b/tests/test_evaluator.py
index 8cb8ed1..bbdd9c5 100644
--- a/tests/test_evaluator.py
+++ b/tests/test_evaluator.py
@@ -11,9 +11,14 @@
from evaluator import (
_extract_codex_error,
build_prompt,
+ check_claude_installed,
check_codex_installed,
+ detect_host,
+ evaluate_plan_claude,
evaluate_plan,
parse_codex_output,
+ resolve_evaluator,
+ reset_claude_cache,
reset_codex_cache,
)
from config import Config
@@ -26,6 +31,9 @@ def _make_config(**overrides):
"threshold": 7,
"max_rounds": 3,
"model": "",
+ "evaluator": "auto",
+ "codex_bin": "codex",
+ "claude_bin": "claude",
"fail_open": True,
"enabled": True,
"rubric": "Score it 1-10.",
@@ -192,6 +200,29 @@ def test_empty_weaknesses_with_perfect_score_accepted(self):
self.assertIsNone(error)
self.assertEqual(result["score"], 10)
+ def test_claude_wrapped_json_result_accepted(self):
+ stdout = json.dumps({"type": "result", "result": json.dumps(VALID_RESULT)})
+ result, error = parse_codex_output(stdout)
+ self.assertIsNone(error)
+ self.assertEqual(result["score"], 8)
+
+
+class TestEvaluatorRouting(unittest.TestCase):
+ def test_auto_claude_host_uses_codex(self):
+ self.assertEqual(resolve_evaluator(_make_config(), host="claude"), "codex")
+
+ def test_auto_codex_host_uses_claude(self):
+ self.assertEqual(resolve_evaluator(_make_config(), host="codex"), "claude")
+
+ def test_explicit_evaluator_override(self):
+ self.assertEqual(resolve_evaluator(_make_config(evaluator="claude"), host="claude"), "claude")
+
+ def test_detect_host_from_claude_tool(self):
+ self.assertEqual(detect_host({"tool_name": "ExitPlanMode"}), "claude")
+
+ def test_detect_host_from_codex_event(self):
+ self.assertEqual(detect_host({"hookEventName": "Stop"}), "codex")
+
@patch("evaluator.PLUGIN_ROOT", _PROJECT_ROOT)
class TestEvaluatePlan(unittest.TestCase):
@@ -247,6 +278,22 @@ def test_ephemeral_flag_passed(self, mock_check, mock_run):
cmd = mock_run.call_args[0][0]
self.assertIn("--ephemeral", cmd)
+ @patch("evaluator.subprocess.run")
+ @patch("evaluator.check_codex_installed", return_value=True)
+ def test_codex_disables_hooks_and_sets_sentinel(self, mock_check, mock_run):
+ mock_run.return_value = MagicMock(
+ returncode=0,
+ stdout=json.dumps(VALID_RESULT),
+ stderr="",
+ )
+ config = _make_config()
+ evaluate_plan("My plan", config)
+ cmd = mock_run.call_args[0][0]
+ self.assertIn("--disable", cmd)
+ self.assertIn("hooks", cmd)
+ env = mock_run.call_args[1]["env"]
+ self.assertEqual(env["_PLANMAN_EVALUATOR"], "1")
+
@patch("evaluator.subprocess.run")
@patch("evaluator.check_codex_installed", return_value=True)
def test_model_flag_passed(self, mock_check, mock_run):
@@ -394,6 +441,56 @@ def test_caching(self, mock_which):
mock_which.assert_called_once()
+class TestCheckClaudeInstalled(unittest.TestCase):
+ def setUp(self):
+ reset_claude_cache()
+
+ def tearDown(self):
+ reset_claude_cache()
+
+ @patch("evaluator.shutil.which", return_value="/usr/local/bin/claude")
+ def test_found(self, mock_which):
+ self.assertTrue(check_claude_installed())
+
+ @patch("evaluator.shutil.which", return_value=None)
+ def test_not_found(self, mock_which):
+ self.assertFalse(check_claude_installed())
+
+
+@patch("evaluator.PLUGIN_ROOT", _PROJECT_ROOT)
+class TestEvaluatePlanClaude(unittest.TestCase):
+ def setUp(self):
+ reset_claude_cache()
+
+ def tearDown(self):
+ reset_claude_cache()
+
+ @patch("evaluator.subprocess.run")
+ @patch("evaluator.check_claude_installed", return_value=True)
+ def test_successful_claude_evaluation(self, mock_check, mock_run):
+ mock_run.return_value = MagicMock(
+ returncode=0,
+ stdout=json.dumps({"result": json.dumps(VALID_RESULT)}),
+ stderr="",
+ )
+ result, error = evaluate_plan_claude("My plan", _make_config(evaluator="claude"))
+ self.assertIsNone(error)
+ self.assertEqual(result["score"], 8)
+ cmd = mock_run.call_args[0][0]
+ self.assertIn("claude", cmd[0])
+ self.assertIn("-p", cmd)
+ self.assertIn("--json-schema", cmd)
+ self.assertIn("--no-session-persistence", cmd)
+ self.assertNotIn("--bare", cmd)
+ self.assertEqual(mock_run.call_args[1]["env"]["_PLANMAN_EVALUATOR"], "1")
+
+ @patch("evaluator.check_claude_installed", return_value=False)
+ def test_claude_not_installed(self, mock_check):
+ result, error = evaluate_plan_claude("My plan", _make_config(evaluator="claude"))
+ self.assertIsNone(result)
+ self.assertIn("claude CLI not found", error)
+
+
class TestPluginRoot(unittest.TestCase):
def test_empty_plugin_root_uses_file_based_fallback(self):
"""When CLAUDE_PLUGIN_ROOT is empty string, should use file-based fallback."""
@@ -428,7 +525,7 @@ def test_missing_schema_returns_error(self, mock_check):
result, error = evaluate_plan("My plan", config)
self.assertIsNone(result)
self.assertIn("schema file not found", error)
- self.assertIn("CLAUDE_PLUGIN_ROOT", error)
+ self.assertIn("plugin root", error)
class TestExtractCodexError(unittest.TestCase):